diff --git a/CHANGELOG.md b/CHANGELOG.md
index 39cbf1a..7645791 100644
--- a/CHANGELOG.md
+++ b/CHANGELOG.md
@@ -6,6 +6,63 @@ All notable changes to Tessera are recorded here. The format follows
## [Unreleased]
+### Fixed
+
+- **The MAFFT backend ignored strand.** A genome, or one contig of a draft assembly, on the
+ opposite strand to the backbone was aligned as given and came out at chance-level identity
+ (about 0.40 against 0.97 for the same genome in forward orientation), which the scan then
+ read as a divergent region. MAFFT now runs with `--adjustdirection`.
+- **`reassort` never applied its alignment-fraction filter.** The threshold was written as a
+ fraction (0.5) and compared with skani's percentage, so a tip aligning over a fifth of a
+ segment could outrank a full-length match and become the segment's nearest strain. The
+ threshold is now 50 %.
+- **`fill-references` did not align the last round's downloads.** On a `--max-rounds` exit the
+ final downloads were counted in `fill_summary.tsv`, the report and `lineages.tsv` but were
+ absent from `panel.msa.fasta`, so detection ran without them. One more alignment
+ (`final.msa.fasta`) is now built from them, without a further search.
+- **`fill-references --curate` did nothing unless a round both found a gap and downloaded a
+ reference.** A supplied collection that contains a whole-genome sibling of the query has
+ no coverage gap, so the loop converged before curation ran. Curation now runs before the
+ first alignment for a supplied collection and before each later build for that round's
+ downloads. A panel seeded from scratch is unchanged: seeding has its own sibling filter.
+ One limit remains: the sibling test is anchored on the query's closest whole-genome
+ relative, so when a sibling is itself the closest genome in the collection it becomes
+ that anchor and is kept (further siblings are removed). The log names the anchor chosen.
+- **`fill-references --curate --reference X` could delete X** and fail the next round with
+ "Reference 'X' not found". The given reference is now never removed by curation (it is
+ listed as `sibling-kept` / `redundant-kept` where it would have been). The sibling test
+ stays anchored on the query's closest relative, not on the reference.
+- **A run could delete an existing `
'
+ )
+```
+
+with:
+
+```python
+ f'bases to judge; breakpoint = windows straddling a called '
+ f'breakpoint, where the two parents together explain the query (not a missing '
+ f'reference, and not counted in the caveat above).'
+ )
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ fig = build_interactive_figure(
+ ctx.site_result, datasets, regions, ctx.gaps,
+```
+
+with:
+
+```python
+ fig = build_interactive_figure(
+ ctx.site_result, datasets, regions, ctx.caveat_gaps,
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ gaps = ctx.gaps
+ lineage_map = ctx.lineage_map
+ threshold = ctx.coverage_threshold
+ fig = build_interactive_figure(result, datasets, regions, gaps)
+```
+
+with:
+
+```python
+ gaps = ctx.gaps
+ # Breakpoint gaps are tabulated but are not poorly covered stretches.
+ caveat_gaps = ctx.caveat_gaps
+ lineage_map = ctx.lineage_map
+ threshold = ctx.coverage_threshold
+ fig = build_interactive_figure(result, datasets, regions, caveat_gaps)
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ f"{_caveat_html(gaps, threshold)}"
+```
+
+with:
+
+```python
+ f"{_caveat_html(caveat_gaps, threshold)}"
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ f"{_mosaic_html(regions, colors, s, gaps, lineage_map)}"
+```
+
+with:
+
+```python
+ f"{_mosaic_html(regions, colors, s, caveat_gaps, lineage_map)}"
+```
+
+In `src/tessera/recomb/report_text.py`, replace:
+
+```python
+from .coverage import CoverageGap
+```
+
+with:
+
+```python
+from .coverage import BREAKPOINT_KIND, CoverageGap
+```
+
+In `src/tessera/recomb/report_text.py`, replace:
+
+```python
+ print_formatted_table(rows, header=COVERAGE_HEADER, echo=echo)
+ echo(" ^ the closest reference here is poor; the true source may be missing. "
+ "Run 'tessera find-references' to search NCBI.")
+ echo("")
+```
+
+with:
+
+```python
+ print_formatted_table(rows, header=COVERAGE_HEADER, echo=echo)
+ if any(g.kind != BREAKPOINT_KIND for g in gaps):
+ echo(" ^ the closest reference here is poor; the true source may be missing. "
+ "Run 'tessera find-references' to search NCBI.")
+ if any(g.kind == BREAKPOINT_KIND for g in gaps):
+ echo(" breakpoint = windows straddling a called breakpoint; the two parents "
+ "together explain the query there (not a missing reference).")
+ echo("")
+```
+
+In `docs/detection-methods.md`, replace:
+
+```markdown
+| `coverage_gaps.tsv` | Stretches where even the closest reference is a poor match -- possible missing references |
+```
+
+with:
+
+```markdown
+| `coverage_gaps.tsv` | Stretches where even the closest reference is below the best-similarity threshold, with a `kind`: `divergent` (the query is far from every reference -- a possible missing reference), `low_information` (too few comparable bases to judge), or `breakpoint` (windows straddling a called breakpoint, where the region's two parents together explain the query; not a missing reference, and it does not caveat the region) |
+```
+
+In `example_data/README.md`, replace:
+
+```markdown
+Both callers call `parent_B` over the insert (q-value ~1e-29) with a sharp breakpoint,
+so the region is flagged as agreeing (high confidence); the similarity plot shows an
+obvious crossover.
+```
+
+with:
+
+```markdown
+The four default callers all call `parent_B` over the insert with a sharp breakpoint,
+so the region is flagged as agreeing (high confidence); the similarity plot shows an
+obvious crossover. The two short stretches listed under reference coverage are the
+windows straddling the breakpoints (kind `breakpoint`), not missing references.
+```
+
+- [ ] **Step 4: Run the tests to verify they pass**
+
+Run: `pytest tests/unit/test_breakpoint_gaps.py tests/unit/test_coverage.py tests/unit/test_coverage_reconcile.py tests/unit/test_output_contract.py tests/integration/test_recomb_pipeline.py -q`
+Expected: PASS -- `41 passed`
+
+Run: `ruff check src tests validation && mypy src`
+Expected: `All checks passed!` and `Success: no issues found in 79 source files`
+
+Then run the shipped example and read the result:
+
+```bash
+tessera recomb --msa example_data/divergent.msa.fasta --query query \
+ --output /tmp/tessera-plan-b/divergent --window-size 300 --window-step 30 --plot-format png
+cut -f1,2,5,6,21,22,23 /tmp/tessera-plan-b/divergent/recombination_regions.tsv
+cat /tmp/tessera-plan-b/divergent/coverage_gaps.tsv
+```
+
+Expected: one region `parent_B parent_A 960 2010 no no hmm,3seq,maxchi,bootscan`; two gaps (`810-1170`, `1860-2190`) both of kind `breakpoint`; the log line `2 low-similarity stretch(es) sit on a called breakpoint ...`; and the report verdict ends "high confidence." with no "Possible missing reference" box.
+
+- [ ] **Step 5: Commit**
+
+```bash
+git add tests/unit/test_breakpoint_gaps.py src/tessera/recomb/coverage.py src/tessera/recomb/run.py src/tessera/recomb/report_context.py src/tessera/recomb/report_html.py src/tessera/recomb/report_text.py docs/detection-methods.md example_data/README.md
+git commit -m "Do not report breakpoint-straddling windows as missing references" -m "Co-Authored-By: Claude Fable 5.1 "
+```
+
+---
+
+### Task 4: Report the union of overlapping regions (B3)
+
+The verdict and the "Query recombinant" card add region lengths. The ensemble keeps overlapping regions that name different donors as separate rows, so a 2.2 kb union was reported as "3.3 kb (54.6 %)".
+
+**Files:**
+- Create: `tests/unit/test_report_summary.py`
+- Modify: `src/tessera/recomb/report_html.py`
+
+**Interfaces:**
+- Consumes: nothing new.
+- Produces: `_union_length(spans: list[tuple[int, int]]) -> int` in `recomb/report_html.py`.
+
+- [ ] **Step 1: Write the failing tests**
+
+Create `tests/unit/test_report_summary.py`:
+
+```python
+"""The report's headline numbers agree with the regions they summarise."""
+
+from __future__ import annotations
+
+import numpy as np
+
+from tessera.recomb.regions import Region
+from tessera.recomb.report_html import _summary, _verdict_html
+
+
+class _Result:
+ """The one attribute ``_summary`` reads: a 6000-base, gap-free query."""
+
+ query = "query"
+ query_cumulative = np.arange(6001)
+
+
+def _region(minor: str, start: int, end: int, *, donor_absent: bool = False) -> Region:
+ return Region(
+ minor_parent=minor, major_parent="backbone", msa_start=start, msa_end=end,
+ query_start=start, query_end=end, n_windows=5,
+ mean_sim_minor=0.99, mean_sim_major=0.92, margin=0.07,
+ qvalue=1e-9, support=0.95, methods=("hmm", "3seq"), donor_absent=donor_absent,
+ )
+
+
+def test_overlapping_regions_are_counted_once() -> None:
+ """The ensemble keeps overlapping regions that name different donors separate.
+ Adding their lengths reported a 2.2 kb union as 3.3 kb (54.6 % of the query)."""
+ regions = [_region("donorB", 1900, 3100), _region("donorA", 2025, 4100)]
+ s = _summary(_Result(), regions, ["backbone", "donorA", "donorB"])
+ assert s["recomb_bp"] == 2200 # the union 1900-4100, not 1200 + 2075
+ assert round(s["pct"], 1) == 36.7
+ assert s["n_regions"] == 2 # both regions are still listed
+
+
+def test_recombinant_fraction_never_exceeds_the_query() -> None:
+ regions = [_region(f"donor{i}", 0, 6000) for i in range(3)]
+ s = _summary(_Result(), regions, ["backbone"])
+ assert s["recomb_bp"] == 6000
+ assert s["pct"] == 100.0
+
+
+def test_touching_and_nested_regions() -> None:
+ touching = [_region("donorA", 1000, 2000), _region("donorB", 2000, 3000)]
+ assert _summary(_Result(), touching, ["backbone"])["recomb_bp"] == 2000
+ nested = [_region("donorA", 1000, 4000), _region("donorB", 2000, 2500)]
+ assert _summary(_Result(), nested, ["backbone"])["recomb_bp"] == 3000
+
+
+def test_disjoint_regions_still_add_up() -> None:
+ regions = [_region("donorA", 500, 1500), _region("donorB", 3000, 3500)]
+ assert _summary(_Result(), regions, ["backbone"])["recomb_bp"] == 1500
+
+
+def test_donor_absent_regions_stay_out_of_the_recombinant_span() -> None:
+ regions = [_region("donorA", 500, 1500), _region("x", 3000, 5000, donor_absent=True)]
+ s = _summary(_Result(), regions, ["backbone"])
+ assert s["recomb_bp"] == 1000
+ assert s["n_absent"] == 1
+
+
+def test_verdict_states_the_union() -> None:
+ regions = [_region("donorB", 1900, 3100), _region("donorA", 2025, 4100)]
+ s = _summary(_Result(), regions, ["backbone", "donorA", "donorB"])
+ verdict = _verdict_html(s, "query", {})
+ assert "2.2 kb" in verdict
+ assert "36.7%" in verdict
+```
+
+- [ ] **Step 2: Run the tests to verify they fail**
+
+Run: `pytest tests/unit/test_report_summary.py -q`
+Expected: FAIL --
+
+```text
+FAILED tests/unit/test_report_summary.py::test_overlapping_regions_are_counted_once
+FAILED tests/unit/test_report_summary.py::test_recombinant_fraction_never_exceeds_the_query
+FAILED tests/unit/test_report_summary.py::test_touching_and_nested_regions
+FAILED tests/unit/test_report_summary.py::test_verdict_states_the_union
+4 failed, 2 passed in 0.45s
+```
+
+- [ ] **Step 3: Implement**
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+def _summary(
+ result: WindowSimilarity, regions: list[Region], datasets: list[str]
+) -> dict:
+```
+
+with:
+
+```python
+def _union_length(spans: list[tuple[int, int]]) -> int:
+ """Total length covered by ``spans`` (half-open intervals), overlaps counted once."""
+ total = 0
+ covered_to: int | None = None
+ for start, end in sorted(spans):
+ if end <= start:
+ continue
+ if covered_to is None or start > covered_to:
+ total += end - start
+ covered_to = end
+ elif end > covered_to:
+ total += end - covered_to
+ covered_to = end
+ return total
+
+
+def _summary(
+ result: WindowSimilarity, regions: list[Region], datasets: list[str]
+) -> dict:
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ recomb_bp = sum(max(0, r.query_end - r.query_start) for r in present)
+```
+
+with:
+
+```python
+ # The union, not the sum: overlapping regions that name different donors are kept
+ # as separate rows, and adding their lengths counts the shared stretch twice.
+ recomb_bp = _union_length([(r.query_start, r.query_end) for r in present])
+```
+
+- [ ] **Step 4: Run the tests to verify they pass**
+
+Run: `pytest tests/unit/test_report_summary.py tests/unit/test_report_typed.py -q`
+Expected: PASS -- `9 passed`
+
+Run: `ruff check src tests validation && mypy src`
+Expected: `All checks passed!` and `Success: no issues found in 79 source files`
+
+- [ ] **Step 5: Commit**
+
+```bash
+git add tests/unit/test_report_summary.py src/tessera/recomb/report_html.py
+git commit -m "Report the union of overlapping regions in the verdict" -m "Co-Authored-By: Claude Fable 5.1 "
+```
+
+---
+
+### Task 5: Keep the method comparison in step with re-attribution (B4)
+
+`--reattribute-donors` re-labels the region but not the per-method breakdown rows, so `recombination_methods.tsv` and the report's method table name the donor from before re-attribution.
+
+The spec suggests returning an old-to-new mapping. That is not needed: `reattribute_donors` returns exactly one region per input region in the same order, and `regions` and `method_breakdown` are parallel lists (built together, sorted alike, filtered together), so the rows can be updated in step with a strict `zip`.
+
+**Files:**
+- Modify: `src/tessera/recomb/run.py`
+- Test (extend): `tests/unit/test_reattribute.py`
+
+**Interfaces:**
+- Consumes: `regions` and `method_breakdown` in `run_recomb` being parallel lists.
+- Produces: nothing new.
+
+- [ ] **Step 1: Write the failing tests**
+
+In `tests/unit/test_reattribute.py`, replace the imports:
+
+```python
+import numpy as np
+
+from tessera.recomb.reattribute import reattribute_donors
+from tessera.recomb.regions import Region
+```
+
+with:
+
+```python
+import csv
+import random
+from pathlib import Path
+
+import numpy as np
+
+from tessera.recomb.reattribute import reattribute_donors
+from tessera.recomb.regions import Region
+from tessera.recomb.run import RecombParams, run_recomb
+
+from ..conftest import write_fasta
+```
+
+Append to the end of `tests/unit/test_reattribute.py`:
+
+```python
+# --- the run's other outputs follow the re-attribution ---------------------
+
+def _reattribution_panel(tmp_path: Path) -> tuple[Path, dict[str, str]]:
+ """A cowpox backbone with a variola insert at 2000-4000. The insert is copied from
+ ``variolaA``, but ``variolaA`` shares clade V with two distant genomes, so clade V's
+ consensus matches the insert worse than clade W's (``variolaB`` alone)."""
+ rng = random.Random(11)
+
+ def mutate(seq: str, frac: float) -> str:
+ chars = list(seq)
+ for i in range(len(chars)):
+ if rng.random() < frac:
+ chars[i] = rng.choice("ACGT")
+ return "".join(chars)
+
+ base = "".join(rng.choice("ACGT") for _ in range(6000))
+ cowpox, variola = mutate(base, 0.03), mutate(base, 0.08)
+ variola_b = mutate(variola, 0.01)
+ query = list(cowpox)
+ query[2000:4000] = list(variola[2000:4000])
+ msa = write_fasta(tmp_path / "panel.fasta", {
+ "query": "".join(query), "cowpox": cowpox, "variolaA": variola,
+ "variolaB": variola_b, "junk1": mutate(base, 0.15), "junk2": mutate(base, 0.15),
+ })
+ lineages = {"cowpox": "CPX", "variolaA": "V", "junk1": "V", "junk2": "V", "variolaB": "W"}
+ return msa, lineages
+
+
+def test_method_comparison_names_the_reattributed_donor(tmp_path: Path, logger) -> None:
+ """Re-attribution re-labelled the region but not the per-method breakdown, so
+ `recombination_methods.tsv` and the report's method table kept the old donor."""
+ msa, lineages = _reattribution_panel(tmp_path)
+ out = tmp_path / "out"
+ run_recomb(
+ RecombParams(msa=msa, output=out, query="query", plot_format="png",
+ lineage_map=lineages, reattribute_donors=True, cluster_lineages=False),
+ logger,
+ )
+ regions = list(csv.DictReader((out / "recombination_regions.tsv").open(), delimiter="\t"))
+ methods = list(csv.DictReader((out / "recombination_methods.tsv").open(), delimiter="\t"))
+ assert [r["minor_parent"] for r in regions] == ["variolaB"] # re-attributed from variolaA
+ assert [m["minor_parent"] for m in methods] == ["variolaB"]
+ assert [(m["query_start"], m["query_end"]) for m in methods] == [
+ (r["query_start"], r["query_end"]) for r in regions
+ ]
+```
+
+- [ ] **Step 2: Run the tests to verify they fail**
+
+Run: `pytest tests/unit/test_reattribute.py -q`
+Expected: FAIL --
+
+```text
+FAILED tests/unit/test_reattribute.py::test_method_comparison_names_the_reattributed_donor
+1 failed, 7 passed in 2.72s
+```
+
+- [ ] **Step 3: Implement**
+
+In `src/tessera/recomb/run.py`, replace:
+
+```python
+ margin=params.reattribute_margin, logger=logger,
+ )
+```
+
+with:
+
+```python
+ margin=params.reattribute_margin, logger=logger,
+ )
+ # The breakdown rows were built from the regions before re-attribution and are
+ # written to recombination_methods.tsv and the report's method table. Keep the
+ # donor they name in step with the region. reattribute_donors returns one region
+ # per input region in the same order, so the two lists stay parallel.
+ for region, row in zip(regions, method_breakdown, strict=True):
+ row["minor_parent"] = region.minor_parent
+```
+
+- [ ] **Step 4: Run the tests to verify they pass**
+
+Run: `pytest tests/unit/test_reattribute.py tests/unit/test_ensemble.py -q`
+Expected: PASS -- `24 passed`
+
+Run: `ruff check src tests validation && mypy src`
+Expected: `All checks passed!` and `Success: no issues found in 79 source files`
+
+- [ ] **Step 5: Commit**
+
+```bash
+git add tests/unit/test_reattribute.py src/tessera/recomb/run.py
+git commit -m "Keep the method comparison in step with donor re-attribution" -m "Co-Authored-By: Claude Fable 5.1 "
+```
+
+---
+
+### Task 6: Reject a --lineage-map that does not exist (B5)
+
+`--lineage-map` pointing at a file that does not exist is ignored (exit 0, untyped report) in five commands, because the readers treat a missing lineage file as "no typed names". The barcode caller and donor re-attribution then do nothing, silently.
+
+This task creates `tests/unit/test_cli_input_checks.py`; Task 12 appends to it. Its `no_pipeline` fixture is what keeps these tests off the network.
+
+**Files:**
+- Create: `tests/unit/test_cli_input_checks.py`
+- Modify: `src/tessera/cli/main.py`
+- Modify: `src/tessera/cli/cmd_recomb.py`
+- Modify: `src/tessera/cli/cmd_type_lineages.py`
+- Modify: `src/tessera/cli/cmd_detect.py`
+- Modify: `src/tessera/cli/cmd_build_panel.py`
+- Modify: `src/tessera/cli/cmd_fill_references.py`
+
+**Interfaces:**
+- Consumes: `_require_file` in `cli/main.py`.
+- Produces: `_require_lineage_map(path: Path | None) -> None` in `cli/main.py`. Task 12 extends the import lists this task writes in `cmd_detect.py`, `cmd_build_panel.py`, `cmd_fill_references.py` and `cmd_recomb.py`.
+
+- [ ] **Step 1: Write the failing tests**
+
+Create `tests/unit/test_cli_input_checks.py`:
+
+```python
+"""Inputs the CLI must reject before any work starts.
+
+Each case used to be accepted: the option was silently ignored, or the run failed
+later under "Unexpected error". None of these tests may reach the network -- the
+panel-building entry points are replaced with a function that fails the test if the
+command gets that far.
+"""
+
+from __future__ import annotations
+
+from pathlib import Path
+
+import pytest
+from typer.testing import CliRunner
+
+from tessera.cli.main import app
+
+runner = CliRunner()
+
+EXAMPLE = "example_data/divergent.msa.fasta"
+
+
+def _must_not_run(*args, **kwargs):
+ raise AssertionError("the command reached the pipeline; it should have been rejected")
+
+
+@pytest.fixture
+def no_pipeline(monkeypatch):
+ """Fail the test if a command gets past validation into panel building or typing."""
+ monkeypatch.setattr("tessera.discover.iterate.fill_references", _must_not_run)
+ monkeypatch.setattr("tessera.discover.lineage_assign.assign_lineages", _must_not_run)
+ monkeypatch.setattr("tessera.discover.run.find_references", _must_not_run)
+
+
+def _query(tmp_path: Path) -> Path:
+ path = tmp_path / "query.fasta"
+ path.write_text(">query\nACGTACGTACGT\n")
+ return path
+
+
+def _collection(tmp_path: Path) -> Path:
+ directory = tmp_path / "collection"
+ directory.mkdir()
+ (directory / "ref.fasta").write_text(">ref\nACGTACGTACGT\n")
+ return directory
+
+
+def _rejected(result, needle: str) -> None:
+ assert result.exit_code == 1, result.output
+ assert "Unexpected error" not in result.output
+ assert needle in " ".join(result.output.split())
+
+
+# --- --lineage-map must exist ---------------------------------------------
+
+def test_recomb_rejects_a_missing_lineage_map(tmp_path: Path) -> None:
+ """A mistyped path used to be ignored: the run exited 0 with an untyped report, and
+ the barcode caller and donor re-attribution silently did nothing."""
+ result = runner.invoke(app, [
+ "recomb", "--msa", EXAMPLE, "--query", "query",
+ "--output", str(tmp_path / "out"), "--lineage-map", str(tmp_path / "nope.tsv"),
+ ])
+ _rejected(result, "--lineage-map file not found")
+ assert not (tmp_path / "out" / "report.html").exists()
+
+
+def test_recomb_rejects_a_lineage_map_that_is_a_directory(tmp_path: Path) -> None:
+ result = runner.invoke(app, [
+ "recomb", "--msa", EXAMPLE, "--query", "query",
+ "--output", str(tmp_path / "out"), "--lineage-map", str(tmp_path),
+ ])
+ _rejected(result, "is a directory, not a file")
+
+
+def test_type_lineages_rejects_a_missing_lineage_map(tmp_path: Path, no_pipeline) -> None:
+ result = runner.invoke(app, [
+ "type-lineages", "--collection", str(_collection(tmp_path)),
+ "--output", str(tmp_path / "out"), "--lineage-map", str(tmp_path / "nope.tsv"),
+ ])
+ _rejected(result, "--lineage-map file not found")
+
+
+@pytest.mark.parametrize("command", ["detect", "fill-references", "build-panel"])
+def test_panel_commands_reject_a_missing_lineage_map(
+ tmp_path: Path, no_pipeline, command: str
+) -> None:
+ result = runner.invoke(app, [
+ command, "--query", str(_query(tmp_path)), "--output", str(tmp_path / "out"),
+ "--lineage-map", str(tmp_path / "nope.tsv"),
+ ])
+ _rejected(result, "--lineage-map file not found")
+```
+
+- [ ] **Step 2: Run the tests to verify they fail**
+
+Run: `pytest tests/unit/test_cli_input_checks.py -q`
+Expected: FAIL --
+
+```text
+FAILED tests/unit/test_cli_input_checks.py::test_recomb_rejects_a_missing_lineage_map
+FAILED tests/unit/test_cli_input_checks.py::test_recomb_rejects_a_lineage_map_that_is_a_directory
+FAILED tests/unit/test_cli_input_checks.py::test_type_lineages_rejects_a_missing_lineage_map
+FAILED tests/unit/test_cli_input_checks.py::test_panel_commands_reject_a_missing_lineage_map[detect]
+FAILED tests/unit/test_cli_input_checks.py::test_panel_commands_reject_a_missing_lineage_map[fill-references]
+FAILED tests/unit/test_cli_input_checks.py::test_panel_commands_reject_a_missing_lineage_map[build-panel]
+6 failed in 2.52s
+```
+
+- [ ] **Step 3: Implement**
+
+In `src/tessera/cli/main.py`, replace:
+
+```python
+def _require_directory(path: Path, label: str) -> None:
+```
+
+with:
+
+```python
+def _require_lineage_map(path: Path | None) -> None:
+ """Reject a ``--lineage-map`` that does not exist.
+
+ The readers treat a missing lineage file as "no typed names", which is right for
+ the automatically discovered ``lineages.tsv`` and wrong for a path the user typed:
+ the run would finish with an untyped report, and the barcode caller and donor
+ re-attribution would do nothing, without a word.
+ """
+ if path is not None:
+ _require_file(path, "--lineage-map file")
+
+
+def _require_directory(path: Path, label: str) -> None:
+```
+
+In `src/tessera/cli/cmd_recomb.py`, replace:
+
+```python
+ _require_file,
+ _require_range,
+```
+
+with:
+
+```python
+ _require_file,
+ _require_lineage_map,
+ _require_range,
+```
+
+In `src/tessera/cli/cmd_recomb.py`, replace:
+
+```python
+ _require_file(msa, "MSA file")
+```
+
+with:
+
+```python
+ _require_file(msa, "MSA file")
+ _require_lineage_map(lineage_map)
+```
+
+In `src/tessera/cli/cmd_type_lineages.py`, replace:
+
+```python
+from .main import _require_directory, app, get_logger, stage_errors
+```
+
+with:
+
+```python
+from .main import _require_directory, _require_lineage_map, app, get_logger, stage_errors
+```
+
+In `src/tessera/cli/cmd_type_lineages.py`, replace:
+
+```python
+ _require_directory(collection, "Collection directory")
+```
+
+with:
+
+```python
+ _require_directory(collection, "Collection directory")
+ _require_lineage_map(lineage_map)
+```
+
+In `src/tessera/cli/cmd_detect.py`, replace:
+
+```python
+from .main import _require_choice, _require_file, app, get_logger, stage_errors
+```
+
+with:
+
+```python
+from .main import (
+ _require_choice,
+ _require_file,
+ _require_lineage_map,
+ app,
+ get_logger,
+ stage_errors,
+)
+```
+
+In `src/tessera/cli/cmd_detect.py`, replace:
+
+```python
+ _require_file(query, "Query file")
+```
+
+with:
+
+```python
+ _require_file(query, "Query file")
+ _require_lineage_map(lineage_map)
+```
+
+In `src/tessera/cli/cmd_build_panel.py`, replace:
+
+```python
+from .main import _require_choice, _require_file, app, get_logger, stage_errors
+```
+
+with:
+
+```python
+from .main import (
+ _require_choice,
+ _require_file,
+ _require_lineage_map,
+ app,
+ get_logger,
+ stage_errors,
+)
+```
+
+In `src/tessera/cli/cmd_build_panel.py`, replace:
+
+```python
+ _require_file(query, "Query file")
+```
+
+with:
+
+```python
+ _require_file(query, "Query file")
+ _require_lineage_map(lineage_map)
+```
+
+In `src/tessera/cli/cmd_fill_references.py`, replace:
+
+```python
+ _require_file,
+ app,
+```
+
+with:
+
+```python
+ _require_file,
+ _require_lineage_map,
+ app,
+```
+
+In `src/tessera/cli/cmd_fill_references.py`, replace:
+
+```python
+ _require_file(query, "Query file")
+```
+
+with:
+
+```python
+ _require_file(query, "Query file")
+ _require_lineage_map(lineage_map)
+```
+
+- [ ] **Step 4: Run the tests to verify they pass**
+
+Run: `pytest tests/unit/test_cli_input_checks.py tests/unit/test_cli_commands.py tests/unit/test_cli_validation.py -q`
+Expected: PASS -- `54 passed`
+
+Run: `ruff check src tests validation && mypy src`
+Expected: `All checks passed!` and `Success: no issues found in 79 source files`
+
+- [ ] **Step 5: Commit**
+
+```bash
+git add tests/unit/test_cli_input_checks.py src/tessera/cli/main.py src/tessera/cli/cmd_recomb.py src/tessera/cli/cmd_type_lineages.py src/tessera/cli/cmd_detect.py src/tessera/cli/cmd_build_panel.py src/tessera/cli/cmd_fill_references.py
+git commit -m "Reject a --lineage-map that does not exist" -m "Co-Authored-By: Claude Fable 5.1 "
+```
+
+---
+
+### Task 7: Report a barcode caller that could not run as not run (B6)
+
+On an untyped panel the barcode caller returns no regions and names the first record of the alignment as major parent. A `--method barcode` run then reports that record as the backbone and a clean negative; in an ensemble the method table shows barcode as `no`.
+
+**Files:**
+- Modify: `src/tessera/recomb/barcode.py`
+- Modify: `src/tessera/recomb/run.py`
+- Modify: `src/tessera/recomb/report_context.py`
+- Modify: `src/tessera/recomb/report_text.py`
+- Modify: `src/tessera/recomb/report.py`
+- Modify: `src/tessera/recomb/report_html.py`
+- Modify: `docs/detection-methods.md`
+- Test (extend): `tests/unit/test_barcode.py`
+
+**Interfaces:**
+- Consumes: `majors: dict[str, str | None]` in `run_recomb`; `recombinant_msa` and `write_fasta` from `tests/conftest.py`.
+- Produces: `call_regions_barcode` returns `([], None, [])` when it cannot run; locals `not_run: tuple[str, ...]` and `n_ran: int` in `run_recomb`; `ReportContext.methods_not_run: tuple[str, ...] = ()`; `write_methods_tsv(..., methods_not_run=())`; `_method_comparison_html(..., methods_not_run=())` and `_method_section(..., methods_not_run=())`; provenance key `callers not run`. Task 8 edits the `methods_not_run=not_run,` line; Task 10 edits the `write_methods_tsv(...)` call block.
+
+- [ ] **Step 1: Write the failing tests**
+
+In `tests/unit/test_barcode.py`, replace the imports:
+
+```python
+from pathlib import Path
+
+import numpy as np
+
+from tessera.recomb.analyze import analyze
+from tessera.recomb.barcode import clade_markers
+from tessera.recomb.regions import RegionParams, call_regions
+from tessera.recomb.similarity import compute_similarity
+
+from ..conftest import write_fasta
+```
+
+with:
+
+```python
+import csv
+import json
+import logging
+from pathlib import Path
+
+import numpy as np
+import pytest
+
+from tessera.core.errors import UserInputError
+from tessera.recomb.analyze import analyze
+from tessera.recomb.barcode import clade_markers
+from tessera.recomb.regions import RegionParams, call_regions
+from tessera.recomb.run import RecombParams, run_recomb
+from tessera.recomb.similarity import compute_similarity
+
+from ..conftest import recombinant_msa, write_fasta
+```
+
+Append to the end of `tests/unit/test_barcode.py`:
+
+```python
+# --- "could not run" is not "found nothing" --------------------------------
+
+def test_barcode_names_no_major_parent_when_it_cannot_run(tmp_path: Path) -> None:
+ """It used to return the first record of the alignment as the major parent, which a
+ barcode-only run then reported as the backbone."""
+ result = compute_similarity(str(_panel(tmp_path)), "query", window_size=300, window_step=30)
+ params = RegionParams.with_defaults(300, method="barcode") # no lineage map
+ regions, major, _ = call_regions(result, analyze(result), 300, params)
+ assert regions == []
+ assert major is None
+
+
+def _run(tmp_path: Path, logger, **kwargs) -> Path:
+ out = tmp_path / "out"
+ run_recomb(
+ RecombParams(msa=recombinant_msa(tmp_path, recombinant=True), output=out,
+ query="query", plot_format="png", **kwargs),
+ logger,
+ )
+ return out
+
+
+def test_barcode_only_run_on_an_untyped_panel_is_refused(tmp_path: Path, logger) -> None:
+ """Nothing was tested, so there is no result to report -- not a clean negative."""
+ with pytest.raises(UserInputError, match="typed references"):
+ _run(tmp_path, logger, methods=("barcode",))
+ assert not (tmp_path / "out" / "report.html").exists()
+
+
+def _collecting_logger() -> tuple[logging.Logger, list[str]]:
+ """A logger outside the ``tessera`` hierarchy that keeps its warnings in a list.
+
+ Not ``caplog``: once a CLI test has configured the ``tessera`` logger it stops
+ propagating to the root logger, and whether that has happened depends on test order.
+ """
+ messages: list[str] = []
+
+ class _Collect(logging.Handler):
+ def emit(self, record: logging.LogRecord) -> None:
+ messages.append(record.getMessage())
+
+ log = logging.getLogger("test_barcode.collect")
+ log.handlers = [_Collect(level=logging.WARNING)]
+ log.propagate = False
+ return log, messages
+
+
+def test_ensemble_reports_barcode_as_not_run_on_an_untyped_panel(tmp_path: Path) -> None:
+ log, warnings = _collecting_logger()
+ out = _run(tmp_path, log, methods=("3seq", "barcode"))
+ assert any("barcode" in message and "could not run" in message for message in warnings)
+
+ rows = list(csv.DictReader((out / "recombination_methods.tsv").open(), delimiter="\t"))
+ assert rows, "3seq should still call the insert"
+ assert {row["3seq"] for row in rows} == {"yes"}
+ assert {row["barcode"] for row in rows} == {"not run"} # was "no"
+
+ run = json.loads((out / "run_provenance.json").read_text())["run"]
+ assert "barcode" in run["callers not run"]
+ assert "not run" in (out / "report.html").read_text()
+
+
+def test_agreement_gate_counts_only_the_callers_that_ran(tmp_path: Path, logger) -> None:
+ """--min-methods is clamped to the callers actually run. A caller that could not run
+ cannot agree, so it must not make the gate unreachable."""
+ out = _run(tmp_path, logger, methods=("3seq", "barcode"), min_methods=2)
+ rows = list(csv.DictReader((out / "recombination_regions.tsv").open(), delimiter="\t"))
+ assert [row["methods"] for row in rows] == ["3seq"]
+
+
+def test_barcode_not_run_on_a_typed_panel_without_marked_clades(tmp_path: Path) -> None:
+ """Typed, but every reference is in one clade: there are no clade markers to
+ compete, so the caller cannot run and the warning says why."""
+ log, warnings = _collecting_logger()
+ one_clade = {"A": "L1", "B": "L1", "other": "L1"}
+ out = _run(tmp_path, log, methods=("3seq", "barcode"), lineage_map=one_clade)
+ assert any("fewer than two typed clades" in message for message in warnings)
+ rows = list(csv.DictReader((out / "recombination_methods.tsv").open(), delimiter="\t"))
+ assert {row["barcode"] for row in rows} == {"not run"}
+```
+
+The warning test uses its own logger instead of `caplog`: once a CLI test has configured the `tessera` logger it stops propagating to the root logger, so `caplog` would make the test depend on test order.
+
+- [ ] **Step 2: Run the tests to verify they fail**
+
+Run: `pytest tests/unit/test_barcode.py -q`
+Expected: FAIL --
+
+```text
+FAILED tests/unit/test_barcode.py::test_barcode_names_no_major_parent_when_it_cannot_run
+FAILED tests/unit/test_barcode.py::test_barcode_only_run_on_an_untyped_panel_is_refused
+FAILED tests/unit/test_barcode.py::test_ensemble_reports_barcode_as_not_run_on_an_untyped_panel
+FAILED tests/unit/test_barcode.py::test_agreement_gate_counts_only_the_callers_that_ran
+FAILED tests/unit/test_barcode.py::test_barcode_not_run_on_a_typed_panel_without_marked_clades
+5 failed, 4 passed in 1.84s
+```
+
+- [ ] **Step 3: Implement**
+
+In `src/tessera/recomb/barcode.py`, replace:
+
+```python
+ that donor clade. Returns ``(regions, major, [])`` to match ``call_regions``.
+ """
+```
+
+with:
+
+```python
+ that donor clade. Returns ``(regions, major, [])`` to match ``call_regions``.
+
+ When the caller cannot run -- the panel is untyped, or fewer than two clades carry
+ enough markers -- it returns ``([], None, [])``. ``major is None`` is the signal that
+ nothing was tested; the pipeline reports the caller as not run rather than as having
+ found nothing.
+ """
+```
+
+In `src/tessera/recomb/barcode.py`, replace:
+
+```python
+ labels = list(result.similarities)
+ default_major = labels[0] if labels else None
+ lineage_map = getattr(params, "lineage_map", None)
+ if not lineage_map:
+ return [], default_major, []
+ cols, alleles, rep = clade_markers(result.rows, result.query, lineage_map)
+ if not cols:
+ return [], default_major, []
+```
+
+with:
+
+```python
+ lineage_map = getattr(params, "lineage_map", None)
+ if not lineage_map:
+ return [], None, []
+ cols, alleles, rep = clade_markers(result.rows, result.query, lineage_map)
+ if not cols:
+ return [], None, []
+```
+
+In `src/tessera/recomb/run.py`, replace:
+
+```python
+ major_parent, per_major = reconcile_major(majors, window_wins=analysis_bp.winners_with_ties)
+```
+
+with:
+
+```python
+ # The barcode caller needs typed references and returns no major parent when it
+ # cannot run. That is "could not test", which must not be reported as "tested and
+ # found nothing": refuse a run that selected nothing else, and say so otherwise.
+ not_run = tuple(m for m in params.methods if m == "barcode" and majors[m] is None)
+ if not_run:
+ reason = (
+ "fewer than two typed clades carry enough characteristic markers"
+ if lineage_map else
+ "it needs typed references (a lineage map: --lineage-map, or a lineages.tsv "
+ "beside the output or the MSA) and none was found"
+ )
+ if len(not_run) == len(params.methods):
+ raise UserInputError(
+ f"The barcode caller could not run: {reason}. No other caller was "
+ "selected, so this scan could not test for recombination. Supply typed "
+ "references or choose another --method."
+ )
+ logger.warning(
+ "The barcode caller could not run (%s); it is reported as 'not run', not as "
+ "a negative.", reason,
+ )
+ n_ran = len(params.methods) - len(not_run)
+
+ major_parent, per_major = reconcile_major(majors, window_wins=analysis_bp.winners_with_ties)
+```
+
+In `src/tessera/recomb/run.py`, replace:
+
+```python
+ min_agree = max(1, min(params.min_methods, len(params.methods)))
+```
+
+with:
+
+```python
+ min_agree = max(1, min(params.min_methods, n_ran))
+```
+
+In `src/tessera/recomb/run.py`, replace:
+
+```python
+ if excluded_siblings:
+ provenance["excluded siblings (query's own lineage)"] = ", ".join(
+```
+
+with:
+
+```python
+ if not_run:
+ provenance["callers not run"] = ", ".join(not_run) + " (needs typed references)"
+ if excluded_siblings:
+ provenance["excluded siblings (query's own lineage)"] = ", ".join(
+```
+
+In `src/tessera/recomb/run.py`, replace:
+
+```python
+ methods_run=params.methods, method_breakdown=method_breakdown, per_major=per_major,
+```
+
+with:
+
+```python
+ methods_run=params.methods, method_breakdown=method_breakdown, per_major=per_major,
+ methods_not_run=not_run,
+```
+
+In `src/tessera/recomb/report_context.py`, replace:
+
+```python
+ methods_run: tuple[str, ...] = ()
+```
+
+with:
+
+```python
+ methods_run: tuple[str, ...] = ()
+ # Selected callers that could not run (barcode on an untyped panel). They stay in
+ # ``methods_run`` so the method table keeps a column for them, marked "not run".
+ methods_not_run: tuple[str, ...] = ()
+```
+
+In `src/tessera/recomb/report_text.py`, replace:
+
+```python
+def write_methods_tsv(
+ breakdown: list[dict], methods_run: tuple[str, ...], output_dir: Path,
+ logger: logging.Logger,
+) -> None:
+ """Write the per-region x per-method agreement matrix (the ensemble breakdown)."""
+```
+
+with:
+
+```python
+def write_methods_tsv(
+ breakdown: list[dict], methods_run: tuple[str, ...], output_dir: Path,
+ logger: logging.Logger, methods_not_run: tuple[str, ...] = (),
+) -> None:
+ """Write the per-region x per-method agreement matrix (the ensemble breakdown).
+
+ A cell is ``yes`` / ``no`` for a caller that ran, and ``not run`` for one that was
+ selected but could not run -- a ``no`` there would read as a caller that looked and
+ found nothing.
+ """
+```
+
+In `src/tessera/recomb/report_text.py`, replace:
+
+```python
+ cells = [("yes" if m in called else "no") for m in methods_run]
+```
+
+with:
+
+```python
+ cells = [
+ "not run" if m in methods_not_run else ("yes" if m in called else "no")
+ for m in methods_run
+ ]
+```
+
+In `src/tessera/recomb/report.py`, replace:
+
+```python
+ write_methods_tsv(ctx.method_breakdown, ctx.methods_run, output_dir, logger)
+```
+
+with:
+
+```python
+ write_methods_tsv(
+ ctx.method_breakdown, ctx.methods_run, output_dir, logger, ctx.methods_not_run
+ )
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+def _method_comparison_html(
+ breakdown: list[dict], methods_run: tuple[str, ...], per_major: dict[str, str],
+ lineage_map: LineageMap | None = None,
+) -> str:
+ """A compact region x method agreement matrix; omitted for a single-method run."""
+ if len(methods_run) < 2:
+ return ""
+```
+
+with:
+
+```python
+def _method_comparison_html(
+ breakdown: list[dict], methods_run: tuple[str, ...], per_major: dict[str, str],
+ lineage_map: LineageMap | None = None, methods_not_run: tuple[str, ...] = (),
+) -> str:
+ """A compact region x method agreement matrix; omitted for a single-method run."""
+ if len(methods_run) < 2:
+ return ""
+ not_run_note = ""
+ if methods_not_run:
+ names = ", ".join(html.escape(m) for m in methods_not_run)
+ not_run_note = (
+ f'
Not run: {names}. The caller needs typed '
+ f'references and none were available, so it tested nothing; its column below '
+ f'is not a negative result.
'
+ )
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ f'Backbone per method — {majors}.'
+ )
+ if not breakdown:
+```
+
+with:
+
+```python
+ f'Backbone per method — {majors}.{not_run_note}'
+ )
+ if not breakdown:
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ cells = "".join(
+ f'
{"✓" if m in b["per_method_support"] else "·"}
'
+ for m in methods_run
+ )
+```
+
+with:
+
+```python
+ cells = "".join(
+ '
not run
' if m in methods_not_run else
+ f'
{"✓" if m in b["per_method_support"] else "·"}
'
+ for m in methods_run
+ )
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+def _method_section(
+ method_breakdown: list[dict] | None, methods_run: tuple[str, ...],
+ per_major: dict[str, str] | None, lineage_map: LineageMap | None,
+) -> str:
+ """Wrap the method-comparison table in a report section (empty for one method)."""
+ body = _method_comparison_html(
+ method_breakdown or [], methods_run, per_major or {}, lineage_map
+ )
+```
+
+with:
+
+```python
+def _method_section(
+ method_breakdown: list[dict] | None, methods_run: tuple[str, ...],
+ per_major: dict[str, str] | None, lineage_map: LineageMap | None,
+ methods_not_run: tuple[str, ...] = (),
+) -> str:
+ """Wrap the method-comparison table in a report section (empty for one method)."""
+ body = _method_comparison_html(
+ method_breakdown or [], methods_run, per_major or {}, lineage_map, methods_not_run
+ )
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ f"{_method_section(ctx.method_breakdown, ctx.methods_run, ctx.per_major, lineage_map)}"
+```
+
+with:
+
+```python
+ f"{method_section}"
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ doc = (
+ '
+```
+
+with:
+
+```python
+ method_section = _method_section(
+ ctx.method_breakdown, ctx.methods_run, ctx.per_major, lineage_map,
+ ctx.methods_not_run,
+ )
+
+ doc = (
+ '
+```
+
+In `docs/detection-methods.md`, replace:
+
+```markdown
+the genome-level callers at low divergence. It is silent on untyped panels, and opt-in
+(needs typed references). It composes with `--pool-consensus` but needs only a typed panel.
+```
+
+with:
+
+```markdown
+the genome-level callers at low divergence. It is opt-in and needs typed references. On an
+untyped panel it cannot run, which is not the same as finding nothing: in an ensemble it is
+logged and shown as `not run` in the method comparison, and a run that selected only
+`barcode` is refused. It composes with `--pool-consensus` but needs only a typed panel.
+```
+
+In `docs/detection-methods.md`, replace:
+
+```markdown
+one row per region with a Y/n per method and the parent-free flag |
+```
+
+with:
+
+```markdown
+one row per region with `yes` / `no` per method (`not run` for a selected caller that could not run) and the parent-free flag |
+```
+
+- [ ] **Step 4: Run the tests to verify they pass**
+
+Run: `pytest tests/unit/test_barcode.py tests/unit/test_ensemble.py tests/unit/test_run.py tests/unit/test_output_contract.py tests/unit/test_report_typed.py -q`
+Expected: PASS -- `53 passed`
+
+Run: `ruff check src tests validation && mypy src`
+Expected: `All checks passed!` and `Success: no issues found in 79 source files`
+
+- [ ] **Step 5: Commit**
+
+```bash
+git add tests/unit/test_barcode.py src/tessera/recomb/barcode.py src/tessera/recomb/run.py src/tessera/recomb/report_context.py src/tessera/recomb/report_text.py src/tessera/recomb/report.py src/tessera/recomb/report_html.py docs/detection-methods.md
+git commit -m "Report a barcode caller that could not run as not run" -m "Co-Authored-By: Claude Fable 5.1 "
+```
+
+---
+
+### Task 8: PHI: use the run's alpha and say when it is not testable (B7)
+
+Two defects in the parent-free signal. (i) The report states and applies alpha 0.05 whatever `--alpha` is, while the per-region flag uses `--alpha`. (ii) With `z` informative sites and a window of at least `z - 1` ranks, every pair of sites is inside the window, the statistic cannot change under permutation, and p is 1 for any data; this is reported as "no significant signal".
+
+The default window is not changed here (that is plan C, item C5).
+
+**Files:**
+- Create: `tests/unit/test_report_signal.py`
+- Modify: `src/tessera/recomb/diagnostics.py`
+- Modify: `src/tessera/recomb/run.py`
+- Modify: `src/tessera/recomb/report_context.py`
+- Modify: `src/tessera/recomb/report_text.py`
+- Modify: `src/tessera/recomb/report_html.py`
+- Modify: `docs/detection-methods.md`
+- Modify: `example_data/README.md`
+- Test (extend): `tests/unit/test_diagnostics.py`
+- Test (extend): `tests/unit/test_harness_scoring.py`
+
+**Interfaces:**
+- Consumes: `ReportContext` construction in `run_recomb` (the line Task 7 wrote); `corroborating_intervals`, which already returns `[]` for `phi_p is None`.
+- Produces: `RecombinationSignal.phi_p: float | None` (`None` = not testable); `ReportContext.alpha: float = 0.05`.
+
+- [ ] **Step 1: Write the failing tests**
+
+Append to the end of `tests/unit/test_diagnostics.py`:
+
+```python
+# --- "not testable" is not "no signal" --------------------------------------
+
+def test_phi_is_not_testable_when_every_site_pair_is_inside_the_window() -> None:
+ """With z informative sites and a window of at least z - 1 ranks, the statistic
+ averages over every pair of sites, so reordering the sites cannot change it and the
+ permutation p-value is 1 whatever the data. That was reported as "no signal"."""
+ rows = _block_alignment(recombinant=True) # 24 informative columns
+ signal = recombination_signal(rows, "s0", lambda c: c, window=100, seed=1)
+ assert signal is not None
+ assert signal.n_informative == 24
+ assert signal.phi_p is None # was 1.0
+ assert signal.rmin >= 1 # Rmin does not depend on the window
+ assert corroborating_intervals(signal, alpha=0.05) == []
+
+
+def test_phi_becomes_testable_one_rank_below_the_site_count() -> None:
+ rows = _block_alignment(recombinant=True) # z = 24, so z - 1 = 23
+ at_limit = recombination_signal(rows, "s0", lambda c: c, window=23, seed=1)
+ below = recombination_signal(rows, "s0", lambda c: c, window=22, seed=1)
+ assert at_limit is not None and below is not None
+ assert at_limit.phi_p is None
+ assert below.phi_p is not None
+```
+
+Create `tests/unit/test_report_signal.py`:
+
+```python
+"""The parent-free signal section states the alpha the run used, and says "not
+testable" when the PHI test could not have rejected."""
+
+from __future__ import annotations
+
+import json
+import logging
+
+from tessera.recomb.diagnostics import RecombinationSignal
+from tessera.recomb.report_html import _signal_html
+from tessera.recomb.report_text import write_profile_tsv
+from tessera.recomb.run import RecombParams, run_recomb
+
+
+def _signal(phi_p: float | None, n_informative: int = 500) -> RecombinationSignal:
+ return RecombinationSignal(
+ n_informative=n_informative, phi_p=phi_p, phi_observed=0.1, phi_window=100,
+ rmin=2, rmin_intervals=[(100, 200), (900, 1000)], profile=[(150, 150, 0.2)],
+ )
+
+
+def test_signal_section_says_not_testable() -> None:
+ html = _signal_html(_signal(None, n_informative=54), alpha=0.05)
+ assert "not testable" in html
+ assert "p =" not in html
+ assert "no significant recombination signal" not in html
+ assert "54 informative sites" in html
+ assert "--phi-window" in html # tells the reader what to change
+
+
+def test_profile_tsv_header_marks_an_untestable_phi(tmp_path) -> None:
+ write_profile_tsv(_signal(None, n_informative=54), tmp_path, logging.getLogger("tessera"))
+ header = (tmp_path / "recombination_profile.tsv").read_text().splitlines()[0]
+ assert header.split("\t")[1] == "NA"
+ assert "not testable" in header
+
+
+def test_run_reports_an_untestable_phi(example_data, tmp_path, logger) -> None:
+ # cryptic_insert has 54 informative sites: fewer than the default window of 100.
+ out = tmp_path / "cryptic"
+ run_recomb(
+ RecombParams(msa=example_data / "cryptic_insert.msa.fasta", output=out,
+ query="query", plot_format="png"),
+ logger,
+ )
+ run = json.loads((out / "run_provenance.json").read_text())["run"]
+ assert run["recombination signal (PHI)"].startswith("not testable")
+ assert "not testable" in (out / "report.html").read_text()
+
+
+def test_report_states_the_alpha_the_run_used(example_data, tmp_path, logger) -> None:
+ """The section always printed "alpha 0.05" and judged significance at 0.05, while
+ the per-region PHI flag used --alpha."""
+ out = tmp_path / "divergent"
+ run_recomb(
+ RecombParams(msa=example_data / "divergent.msa.fasta", output=out, query="query",
+ window_size=300, window_step=30, plot_format="png", alpha=0.2,
+ phi_window=20),
+ logger,
+ )
+ report = (out / "report.html").read_text()
+ assert "(alpha 0.2;" in report
+ assert "(alpha 0.05;" not in report
+```
+
+Append to the end of `tests/unit/test_harness_scoring.py`:
+
+```python
+def test_parse_signal_reads_an_untestable_phi_header(tmp_path):
+ """The profile header carries NA when the PHI test could not have rejected; the
+ harness must still find the Rmin that follows it."""
+ import logging
+
+ from tessera.recomb.diagnostics import RecombinationSignal
+ from tessera.recomb.report_text import write_profile_tsv
+
+ signal = RecombinationSignal(
+ n_informative=54, phi_p=None, phi_observed=0.1, phi_window=100, rmin=2,
+ )
+ write_profile_tsv(signal, tmp_path, logging.getLogger("tessera"))
+ assert rh.parse_signal(tmp_path / "recombination_profile.tsv") == ("NA", "2")
+```
+
+`validation/run_benchmark.py` and `run_coalescent_benchmark.py` already treat a `None` p-value as "untestable" and count it separately, so their denominators change for small alignments; that is the intended reading.
+
+- [ ] **Step 2: Run the tests to verify they fail**
+
+Run: `pytest tests/unit/test_diagnostics.py tests/unit/test_report_signal.py tests/unit/test_harness_scoring.py -q`
+Expected: FAIL --
+
+```text
+FAILED tests/unit/test_diagnostics.py::test_phi_is_not_testable_when_every_site_pair_is_inside_the_window
+FAILED tests/unit/test_diagnostics.py::test_phi_becomes_testable_one_rank_below_the_site_count
+FAILED tests/unit/test_report_signal.py::test_signal_section_says_not_testable
+FAILED tests/unit/test_report_signal.py::test_profile_tsv_header_marks_an_untestable_phi
+FAILED tests/unit/test_report_signal.py::test_run_reports_an_untestable_phi
+FAILED tests/unit/test_report_signal.py::test_report_states_the_alpha_the_run_used
+FAILED tests/unit/test_harness_scoring.py::test_parse_signal_reads_an_untestable_phi_header
+7 failed, 26 passed in 3.02s
+```
+
+- [ ] **Step 3: Implement**
+
+In `src/tessera/recomb/diagnostics.py`, replace:
+
+```python
+ phi_p: float # PHI permutation p-value (one-sided; small = recombination)
+```
+
+with:
+
+```python
+ # PHI permutation p-value (one-sided; small = recombination). None when the test
+ # could not have rejected: see recombination_signal.
+ phi_p: float | None
+```
+
+In `src/tessera/recomb/diagnostics.py`, replace:
+
+```python
+ informative columns (too little variation to test). The PHI p-value is reported
+ as-is; the significance threshold is a reporting concern, applied downstream.
+ ``query_label`` is accepted for interface symmetry; the statistics use every
+ sequence in ``rows``.
+ """
+```
+
+with:
+
+```python
+ informative columns (too little variation to test). The PHI p-value is reported
+ as-is; the significance threshold is a reporting concern, applied downstream.
+ ``query_label`` is accepted for interface symmetry; the statistics use every
+ sequence in ``rows``.
+
+ ``phi_p`` is ``None`` when the test is not testable at this window: with ``z``
+ informative columns and ``window >= z - 1`` every pair of columns falls inside the
+ window, the statistic is the mean over all pairs, and no reordering of the columns
+ can change it -- the permutation p-value would be 1 whatever the data. Rmin, the
+ intervals and the profile do not depend on the permutation and are still returned.
+ """
+```
+
+In `src/tessera/recomb/diagnostics.py`, replace:
+
+```python
+ p, observed = phi_pvalue(incompatible, window, seed=seed)
+```
+
+with:
+
+```python
+ p: float | None
+ if z - 1 > window:
+ p, observed = phi_pvalue(incompatible, window, seed=seed)
+ else:
+ p, observed = None, phi(incompatible, window)
+```
+
+In `src/tessera/recomb/run.py`, replace:
+
+```python
+ if signal is not None:
+ logger.info(
+ "Recombination signal (parent-free): PHI p=%.4g, Rmin=%d (%d informative "
+ "sites).", signal.phi_p, signal.rmin, signal.n_informative,
+ )
+```
+
+with:
+
+```python
+ if signal is not None and signal.phi_p is None:
+ logger.info(
+ "Recombination signal (parent-free): PHI not testable (%d informative "
+ "site(s) do not exceed the window of %d ranks; lower --phi-window), Rmin=%d.",
+ signal.n_informative, signal.phi_window, signal.rmin,
+ )
+ elif signal is not None:
+ logger.info(
+ "Recombination signal (parent-free): PHI p=%.4g, Rmin=%d (%d informative "
+ "sites).", signal.phi_p, signal.rmin, signal.n_informative,
+ )
+```
+
+In `src/tessera/recomb/run.py`, replace:
+
+```python
+ provenance["recombination signal (PHI)"] = (
+ f"p={signal.phi_p:.4g} ({signal.n_informative} informative sites, "
+ f"window {signal.phi_window})"
+ )
+```
+
+with:
+
+```python
+ phi_text = "not testable" if signal.phi_p is None else f"p={signal.phi_p:.4g}"
+ provenance["recombination signal (PHI)"] = (
+ f"{phi_text} ({signal.n_informative} informative sites, "
+ f"window {signal.phi_window})"
+ )
+```
+
+In `src/tessera/recomb/run.py`, replace:
+
+```python
+ methods_not_run=not_run,
+```
+
+with:
+
+```python
+ methods_not_run=not_run, alpha=params.alpha,
+```
+
+In `src/tessera/recomb/report_context.py`, replace:
+
+```python
+ signal: RecombinationSignal | None = None
+```
+
+with:
+
+```python
+ signal: RecombinationSignal | None = None
+ # The run's significance level, so the report judges the PHI p-value at the same
+ # alpha the per-region corroboration used.
+ alpha: float = 0.05
+```
+
+In `src/tessera/recomb/report_text.py`, replace:
+
+```python
+ fo.write(
+ f"# PHI p-value\t{signal.phi_p:.4g}\t(window {signal.phi_window} "
+ f"informative sites, {signal.n_informative} sites, Rmin {signal.rmin})\n"
+ )
+```
+
+with:
+
+```python
+ # "NA" when the PHI test could not have rejected (too few informative sites
+ # for the window). The Rmin stays last in the note: the harness reads it there.
+ phi_p = "NA" if signal.phi_p is None else f"{signal.phi_p:.4g}"
+ note = "" if signal.phi_p is not None else "not testable at this window; "
+ fo.write(
+ f"# PHI p-value\t{phi_p}\t({note}window {signal.phi_window} "
+ f"informative sites, {signal.n_informative} sites, Rmin {signal.rmin})\n"
+ )
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ significant = signal.phi_p < alpha
+ verdict = (
+ 'significant recombination signal' if significant
+ else 'no significant recombination signal'
+ )
+```
+
+with:
+
+```python
+ if signal.phi_p is None:
+ # Every pair of informative sites is inside one window, so the permutation test
+ # cannot reject whatever the data. Say so instead of printing p = 1.
+ p_cell = "not testable"
+ verdict = (
+ f'{signal.n_informative} informative sites do not exceed the window of '
+ f'{signal.phi_window} ranks, so the permutation test cannot reject here; '
+ f'this is not evidence against recombination. Lower '
+ f'--phi-window to test'
+ )
+ else:
+ p_cell = f"p = {signal.phi_p:.4g}"
+ verdict = (
+ 'significant recombination signal' if signal.phi_p < alpha
+ else 'no significant recombination signal'
+ ) + f' (alpha {alpha:g}; {signal.n_informative} informative sites)'
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ f'
'
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ f"{_signal_html(ctx.signal)}"
+```
+
+with:
+
+```python
+ f"{_signal_html(ctx.signal, ctx.alpha)}"
+```
+
+In `docs/detection-methods.md`, replace:
+
+```markdown
+parent-free outputs. The diagnostic runs for every `--method`; disable with
+`--no-phi`, or widen its window with `--phi-window`.
+```
+
+with:
+
+```markdown
+parent-free outputs. The diagnostic runs for every `--method`; disable with
+`--no-phi`, or widen its window with `--phi-window`.
+
+The PHI p-value is judged at the run's `--alpha`, in the report and for the per-region
+flag alike. The test is **not testable** when the alignment has too few informative sites
+for the window (`--phi-window`, default 100 site ranks; at least window + 2 sites are
+needed): every pair of sites then falls inside one window, so reordering the sites cannot
+change the statistic and the permutation p-value would be 1 whatever the data. Tessera
+reports this as `not testable` (`NA` in the `recombination_profile.tsv` header) rather
+than as a non-significant result, and no region is flagged `parent_free_support`; lower
+`--phi-window` to test such an alignment. Just above that limit the test is defined but
+has little power: the shipped `divergent` example (159 informative sites) gives p = 1 at
+the default window and p = 0.001 at `--phi-window 20`.
+```
+
+In `docs/detection-methods.md`, replace:
+
+```markdown
+| `recombination_profile.tsv` | Parent-free signal: header with the PHI p-value and Rmin, then
+```
+
+with:
+
+```markdown
+| `recombination_profile.tsv` | Parent-free signal: header with the PHI p-value (`NA` when not testable) and Rmin, then
+```
+
+In `example_data/README.md`, replace:
+
+```markdown
+Both runs also report the parent-free PHI / Rmin signal in `recombination_profile.tsv`.
+```
+
+with:
+
+```markdown
+Both runs also write the parent-free PHI / Rmin signal to `recombination_profile.tsv`.
+The cryptic example has 54 informative sites, fewer than the default PHI window of 100, so
+its PHI test is reported as `not testable`; add `--phi-window 5` to test it.
+```
+
+- [ ] **Step 4: Run the tests to verify they pass**
+
+Run: `pytest tests/unit/test_diagnostics.py tests/unit/test_report_signal.py tests/unit/test_harness_scoring.py tests/unit/test_benchmark_scoring.py tests/unit/test_coalescent_benchmark.py tests/integration -q`
+Expected: PASS -- `59 passed, 1 skipped`
+
+Run: `ruff check src tests validation && mypy src`
+Expected: `All checks passed!` and `Success: no issues found in 79 source files`
+
+- [ ] **Step 5: Commit**
+
+```bash
+git add tests/unit/test_diagnostics.py tests/unit/test_harness_scoring.py tests/unit/test_report_signal.py src/tessera/recomb/diagnostics.py src/tessera/recomb/run.py src/tessera/recomb/report_context.py src/tessera/recomb/report_text.py src/tessera/recomb/report_html.py docs/detection-methods.md example_data/README.md
+git commit -m "Judge PHI at the run's alpha and report an untestable test as such" -m "Co-Authored-By: Claude Fable 5.1 "
+```
+
+---
+
+### Task 9: Plots: donor-absent bands, the pair plot, and colours (B8)
+
+The plots label a donor-absent region "recombinant: " in the backbone's colour; `similarity_pair` shows the two leading window winners, which are the backbone and a near-duplicate of it when the panel holds one; and with `--top-n 1` the donor is outside the colour map and drawn grey.
+
+**Files:**
+- Create: `tests/unit/test_report_plots.py`
+- Modify: `src/tessera/recomb/report_plots.py`
+- Modify: `src/tessera/recomb/report.py`
+- Modify: `src/tessera/recomb/report_html.py`
+- Modify: `docs/detection-methods.md`
+
+**Interfaces:**
+- Consumes: `Region.donor_absent`, `Region.length_bp`.
+- Produces: `region_labels(datasets: list[str], regions: list[Region]) -> list[str]` and `ABSENT_LABEL = "donor absent"` in `recomb/report_plots.py`; `pair_datasets(regions: list[Region], ranked: list[str]) -> list[str]` in `recomb/report.py`. Task 10 edits the block containing `pair = pair_datasets(...)`.
+
+- [ ] **Step 1: Write the failing tests**
+
+Create `tests/unit/test_report_plots.py`:
+
+```python
+"""The plots label what the regions table says, in the same colours."""
+
+from __future__ import annotations
+
+from pathlib import Path
+
+from matplotlib.figure import Figure
+
+from tessera.recomb.regions import Region
+from tessera.recomb.report import pair_datasets
+from tessera.recomb.report_plots import (
+ GREY,
+ _color_map,
+ _palette,
+ _shade_regions,
+ build_interactive_figure,
+ region_labels,
+)
+from tessera.recomb.similarity import compute_similarity
+
+from ..conftest import recombinant_msa
+
+
+def _region(minor: str, major: str = "A", *, start: int = 2000, end: int = 4000,
+ donor_absent: bool = False) -> Region:
+ return Region(
+ minor_parent=minor, major_parent=major, msa_start=start, msa_end=end,
+ query_start=start, query_end=end, n_windows=10,
+ mean_sim_minor=0.99, mean_sim_major=0.95, margin=0.04, donor_absent=donor_absent,
+ )
+
+
+# --- colours cover every label a region names ------------------------------
+
+def test_region_labels_appends_region_parents_to_the_plotted_datasets() -> None:
+ labels = region_labels(["A"], [_region("B"), _region("other", donor_absent=True)])
+ assert labels == ["A", "B"] # the donor is added; a donor-absent stand-in is not
+ assert region_labels(["A", "B"], [_region("B")]) == ["A", "B"] # no duplicates
+
+
+def test_donor_outside_the_top_n_keeps_a_colour(tmp_path: Path) -> None:
+ """With --top-n 1 only the backbone was in the colour map, so the donor band and
+ its label were drawn grey."""
+ result = compute_similarity(str(recombinant_msa(tmp_path, recombinant=True)), "query")
+ fig = build_interactive_figure(result, ["A"], [_region("B")])
+ donor_colour = _color_map(["A", "B"])["B"]
+ band = next(a for a in fig.layout.annotations if a.text == "B")
+ assert band.font.color == donor_colour
+ assert donor_colour != GREY
+
+
+# --- a donor-absent region is not a recombinant from the backbone ----------
+
+def test_static_plot_does_not_call_a_donor_absent_region_recombinant() -> None:
+ """The stand-in label of a donor-absent region is the closest reference, often the
+ backbone itself: the legend read "recombinant: " in the backbone's colour."""
+ ax = Figure().subplots() # no pyplot state, no display backend needed
+ _shade_regions(ax, [_region("A", donor_absent=True)], _palette(["A", "B"]))
+ labels = ax.get_legend_handles_labels()[1]
+ assert labels == ["donor absent (no close reference)"]
+
+
+def test_static_plot_still_labels_a_called_donor() -> None:
+ ax = Figure().subplots()
+ _shade_regions(ax, [_region("B")], _palette(["A", "B"]))
+ labels = ax.get_legend_handles_labels()[1]
+ assert labels == ["recombinant: B"]
+
+
+def test_interactive_plot_marks_a_donor_absent_region(tmp_path: Path) -> None:
+ result = compute_similarity(str(recombinant_msa(tmp_path, recombinant=True)), "query")
+ fig = build_interactive_figure(result, ["A", "B"], [_region("A", donor_absent=True)])
+ texts = [a.text for a in fig.layout.annotations]
+ assert texts == ["donor absent"]
+ assert fig.layout.annotations[0].font.color == GREY
+
+
+# --- the pair plot is major versus leading minor ---------------------------
+
+def test_pair_plot_shows_the_backbone_and_the_leading_donor() -> None:
+ """It showed the top two window winners, which are the backbone and a
+ near-duplicate of it when the panel holds one."""
+ regions = [_region("variola", "cowpox", start=2000, end=2600),
+ _region("camelpox", "cowpox", start=4000, end=4100)]
+ assert pair_datasets(regions, ["cowpox", "cowpox2"]) == ["cowpox", "variola"]
+
+
+def test_pair_plot_falls_back_to_the_window_ranking_without_a_called_donor() -> None:
+ assert pair_datasets([], ["cowpox", "cowpox2"]) == ["cowpox", "cowpox2"]
+ absent = [_region("cowpox", "cowpox", donor_absent=True)]
+ assert pair_datasets(absent, ["cowpox", "cowpox2"]) == ["cowpox", "cowpox2"]
+```
+
+The plot file names do not change (`similarity_top{N}` still counts the plotted datasets): colours come from the extended label list, the plotted lines do not.
+
+- [ ] **Step 2: Run the tests to verify they fail**
+
+Run: `pytest tests/unit/test_report_plots.py -q`
+Expected: FAIL --
+
+```text
+ImportError: cannot import name 'pair_datasets' from 'tessera.recomb.report'
+ERROR tests/unit/test_report_plots.py
+1 error in 0.68s
+```
+
+- [ ] **Step 3: Implement**
+
+In `src/tessera/recomb/report_plots.py`, replace:
+
+```python
+def _palette(datasets: list[str]):
+```
+
+with:
+
+```python
+def region_labels(datasets: list[str], regions: list[Region]) -> list[str]:
+ """``datasets`` followed by any parent of a called region that is not among them.
+
+ The plots draw the top-N datasets, but a region can name a parent outside that
+ list (``--top-n 1`` leaves out every donor). Colours are assigned from this
+ extended list so a region is never drawn in the fallback grey; the plotted
+ datasets come first, so their colours do not depend on which regions were called.
+ Donor-absent regions are skipped: their label is a stand-in, not a called parent.
+ """
+ labels = list(datasets)
+ for region in regions:
+ if region.donor_absent:
+ continue
+ for label in (region.major_parent, region.minor_parent):
+ if label not in labels:
+ labels.append(label)
+ return labels
+
+
+def _palette(datasets: list[str]):
+```
+
+In `src/tessera/recomb/report_plots.py`, replace:
+
+```python
+def _shade_regions(ax, regions: list[Region], colors: dict) -> None:
+ seen: set[str] = set()
+ for region in regions:
+ color = colors.get(region.minor_parent, "grey")
+ label = f"recombinant: {region.minor_parent}" if region.minor_parent not in seen else None
+ seen.add(region.minor_parent)
+ ax.axvspan(region.msa_start, region.msa_end, color=color, alpha=0.12, label=label)
+```
+
+with:
+
+```python
+ABSENT_LABEL = "donor absent"
+
+
+def _shade_regions(ax, regions: list[Region], colors: dict) -> None:
+ seen: set[str] = set()
+ for region in regions:
+ if region.donor_absent:
+ # The stand-in label is the closest reference -- often the backbone itself --
+ # so naming it here would read as "recombinant from the backbone".
+ key, color = ABSENT_LABEL, GREY
+ text = f"{ABSENT_LABEL} (no close reference)"
+ else:
+ key, color = region.minor_parent, colors.get(region.minor_parent, GREY)
+ text = f"recombinant: {region.minor_parent}"
+ label = text if key not in seen else None
+ seen.add(key)
+ ax.axvspan(region.msa_start, region.msa_end, color=color, alpha=0.12, label=label)
+```
+
+In `src/tessera/recomb/report_plots.py`, replace:
+
+```python
+ subset = df.loc[available]
+ colors = _palette(available)
+```
+
+with:
+
+```python
+ subset = df.loc[available]
+ colors = _palette(region_labels(available, regions))
+```
+
+In `src/tessera/recomb/report_plots.py`, replace:
+
+```python
+ colors = _palette([seq1, seq2])
+```
+
+with:
+
+```python
+ colors = _palette(region_labels([seq1, seq2], regions))
+```
+
+In `src/tessera/recomb/report_plots.py`, replace:
+
+```python
+ df = result.to_dataframe()
+ colors = _color_map(datasets)
+ fig = go.Figure()
+```
+
+with:
+
+```python
+ df = result.to_dataframe()
+ colors = _color_map(region_labels(datasets, regions))
+ fig = go.Figure()
+```
+
+In `src/tessera/recomb/report_plots.py`, replace:
+
+```python
+ for region in regions:
+ color = colors.get(region.minor_parent, GREY)
+ fig.add_vrect(
+ x0=region.msa_start, x1=region.msa_end,
+ fillcolor=color, opacity=0.12, line_width=0, layer="below",
+ annotation_text=("" if region.minor_parent in seen else region.minor_parent),
+ annotation_position="top left",
+ annotation_font_size=11, annotation_font_color=color,
+ )
+ seen.add(region.minor_parent)
+```
+
+with:
+
+```python
+ for region in regions:
+ if region.donor_absent:
+ key, color = ABSENT_LABEL, GREY
+ else:
+ key, color = region.minor_parent, colors.get(region.minor_parent, GREY)
+ fig.add_vrect(
+ x0=region.msa_start, x1=region.msa_end,
+ fillcolor=color, opacity=0.12, line_width=0, layer="below",
+ annotation_text=("" if key in seen else key),
+ annotation_position="top left",
+ annotation_font_size=11, annotation_font_color=color,
+ )
+ seen.add(key)
+```
+
+In `src/tessera/recomb/report.py`, replace:
+
+```python
+- ``similarity_pair.{fmt}`` static major-vs-minor pairwise plot
+```
+
+with:
+
+```python
+- ``similarity_pair.{fmt}`` static pairwise plot: the major parent against the
+ donor of the longest called region (the two leading
+ window winners when no donor was called)
+```
+
+In `src/tessera/recomb/report.py`, replace:
+
+```python
+def write_reports(
+```
+
+with:
+
+```python
+def pair_datasets(regions: list[Region], ranked: list[str]) -> list[str]:
+ """The two datasets of the pairwise plot: major parent, then the leading donor.
+
+ The leading donor is the donor of the longest called, donor-present region. Without
+ one there is no minor parent to show and the plot falls back to ``ranked`` (the two
+ leading window winners) -- which is not "major versus minor" when the panel holds a
+ near-duplicate of the backbone, hence the region-based choice whenever possible.
+ """
+ present = [r for r in regions if not r.donor_absent]
+ if not present:
+ return ranked
+ longest = max(present, key=lambda r: r.length_bp)
+ return [longest.major_parent, longest.minor_parent]
+
+
+def write_reports(
+```
+
+In `src/tessera/recomb/report.py`, replace:
+
+```python
+ pair = rank_datasets(ranking, 2)
+```
+
+with:
+
+```python
+ pair = pair_datasets(regions, rank_datasets(ranking, 2))
+```
+
+In `src/tessera/recomb/report.py`, replace:
+
+```python
+ "write_html_report", "write_reports",
+]
+```
+
+with:
+
+```python
+ "write_html_report", "write_reports", "pair_datasets",
+]
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+from .report_plots import GREY, _color_map, build_interactive_figure
+```
+
+with:
+
+```python
+from .report_plots import GREY, _color_map, build_interactive_figure, region_labels
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ colors = _color_map(datasets)
+ s = _summary(result, regions, datasets)
+```
+
+with:
+
+```python
+ # Colour every label a region names, not only the top-N: with --top-n 1 the donor
+ # is outside the plotted datasets and its swatch and mosaic segment were grey.
+ colors = _color_map(region_labels(datasets, regions))
+ s = _summary(result, regions, datasets)
+```
+
+In `docs/detection-methods.md`, replace:
+
+```markdown
+| `similarity_pair.{fmt}` | Static plot of the major vs leading minor parent, region shaded |
+```
+
+with:
+
+```markdown
+| `similarity_pair.{fmt}` | Static plot of the major parent against the donor of the longest called region, regions shaded (the two leading window winners when no donor was called) |
+```
+
+- [ ] **Step 4: Run the tests to verify they pass**
+
+Run: `pytest tests/unit/test_report_plots.py tests/unit/test_report_typed.py tests/unit/test_output_contract.py tests/integration -q`
+Expected: PASS -- `47 passed, 1 skipped`
+
+Run: `ruff check src tests validation && mypy src`
+Expected: `All checks passed!` and `Success: no issues found in 79 source files`
+
+- [ ] **Step 5: Commit**
+
+```bash
+git add tests/unit/test_report_plots.py src/tessera/recomb/report_plots.py src/tessera/recomb/report.py src/tessera/recomb/report_html.py docs/detection-methods.md
+git commit -m "Label donor-absent regions and the pair plot for what they show" -m "Co-Authored-By: Claude Fable 5.1 "
+```
+
+---
+
+### Task 10: Replace stale report, help and documentation text (B9)
+
+Text that describes an earlier version of the tool: the report footer hard-codes `.pdf` plots and lists files that were not written; the methods paragraph and references describe a two-caller ensemble; `--method` help says "all but the legacy heuristic" and omits `geneconv`; `docs/reference-panels.md` describes a fetch cap that no longer exists; `docs/detection-methods.md` does not describe lineage clustering.
+
+The clustering section documents a limitation found by the audit (spec item C1: on a panel below about 1.5 % divergence all references pool and the HMM cannot call). It is described as it behaves today; plan C revises it.
+
+**Files:**
+- Create: `tests/unit/test_report_stale_text.py`
+- Modify: `src/tessera/recomb/report.py`
+- Modify: `src/tessera/recomb/report_html.py`
+- Modify: `src/tessera/recomb/report_assets.py`
+- Modify: `src/tessera/cli/cmd_recomb.py`
+- Modify: `src/tessera/cli/cmd_detect.py`
+- Modify: `src/tessera/cli/cmd_fill_references.py`
+- Modify: `docs/detection-methods.md`
+- Modify: `docs/reference-panels.md`
+- Modify: `example_data/README.md`
+
+**Interfaces:**
+- Consumes: `pair_datasets` (Task 9) and the `write_methods_tsv` call (Task 7) inside `write_reports`; the plot functions returning `Path | None`.
+- Produces: `write_html_report(..., ctx, companion_files: list[str] | None = None)` and `_footer_html(provenance, companion_files=None)`.
+
+- [ ] **Step 1: Write the failing tests**
+
+Create `tests/unit/test_report_stale_text.py`:
+
+```python
+"""Report and help text that describes what the run did, not an earlier version of it."""
+
+from __future__ import annotations
+
+import re
+from pathlib import Path
+
+from typer.testing import CliRunner
+
+from tessera.cli.main import app
+from tessera.recomb.report_assets import _REFERENCES
+from tessera.recomb.run import RecombParams, run_recomb
+
+
+def _footer(out: Path) -> str:
+ report = (out / "report.html").read_text()
+ return re.search(r"", report, flags=re.S).group(0)
+
+
+def _run(example_data: Path, out: Path, logger, **kwargs) -> Path:
+ run_recomb(
+ RecombParams(msa=example_data / "divergent.msa.fasta", output=out, query="query",
+ window_size=300, window_step=30, **kwargs),
+ logger,
+ )
+ return out
+
+
+def test_footer_lists_the_plots_in_the_format_they_were_written(
+ example_data, tmp_path, logger
+) -> None:
+ out = tmp_path / "out"
+ out.mkdir()
+ (out / "similarity_pair.pdf").write_text("left over from an earlier run")
+ footer = _footer(_run(example_data, out, logger, plot_format="png"))
+ assert "similarity_pair.png" in footer
+ assert "similarity_top3.png" in footer
+ # Was hard-coded; and a file this run did not write is not listed even if present.
+ assert ".pdf" not in footer
+
+
+def test_footer_lists_only_files_that_exist(example_data, tmp_path, logger) -> None:
+ out = _run(example_data, tmp_path / "out", logger, plot_format="png",
+ methods=("hmm",), phi=False)
+ footer = _footer(out)
+ listed = re.findall(r"(.*?)", footer)
+ assert listed, "the footer should list the companion files"
+ for name in listed:
+ assert (out / name).exists(), f"{name} is listed but was not written"
+ # A single-caller run writes no method comparison; --no-phi writes no profile.
+ assert "recombination_methods.tsv" not in listed
+ assert "recombination_profile.tsv" not in listed
+ assert "run_provenance.json" in listed
+
+
+def test_methods_paragraph_describes_the_default_ensemble(
+ example_data, tmp_path, logger
+) -> None:
+ report = (_run(example_data, tmp_path / "out", logger, plot_format="png")
+ / "report.html").read_text()
+ methods = re.search(r'.*?', report, flags=re.S).group(0)
+ assert "(HMM and the 3SEQ triplet test)" not in methods # the two-caller description
+ for caller in ("3SEQ", "MaxChi", "Bootscan"):
+ assert caller in methods
+
+
+def test_references_cite_every_default_caller() -> None:
+ titles = " ".join(title for title, _ in _REFERENCES)
+ assert "MaxChi" in titles
+ assert "Bootscan" in titles
+
+
+def test_method_help_lists_every_caller_and_the_real_default() -> None:
+ result = CliRunner().invoke(app, ["recomb", "--help"], env={"COLUMNS": "200"})
+ assert result.exit_code == 0
+ text = " ".join(result.output.replace("│", " ").split())
+ assert "geneconv" in text # run by `all`, and was missing from the list
+ assert "all but the legacy heuristic" not in text # geneconv and barcode are opt-in too
+```
+
+Not changed here: the comment `# cap a broad NCBI Virus fetch` at `src/tessera/discover/iterate.py:89` is also stale, but plan A rewrites that file; leave it to plan A to avoid a conflict.
+
+- [ ] **Step 2: Run the tests to verify they fail**
+
+Run: `pytest tests/unit/test_report_stale_text.py -q`
+Expected: FAIL --
+
+```text
+FAILED tests/unit/test_report_stale_text.py::test_footer_lists_the_plots_in_the_format_they_were_written
+FAILED tests/unit/test_report_stale_text.py::test_footer_lists_only_files_that_exist
+FAILED tests/unit/test_report_stale_text.py::test_methods_paragraph_describes_the_default_ensemble
+FAILED tests/unit/test_report_stale_text.py::test_references_cite_every_default_caller
+FAILED tests/unit/test_report_stale_text.py::test_method_help_lists_every_caller_and_the_real_default
+5 failed in 3.87s
+```
+
+- [ ] **Step 3: Implement**
+
+In `src/tessera/recomb/report.py`, replace:
+
+```python
+ write_windows_tsv(result, per_window_winners, output_dir, logger)
+ write_stats_tsv(analysis, output_dir, logger)
+ write_winners_tsv(analysis, output_dir, logger)
+ write_regions_tsv(regions, output_dir, logger)
+ write_coverage_tsv(gaps, ctx.coverage_threshold, output_dir, logger)
+ if ctx.signal is not None:
+ write_profile_tsv(ctx.signal, output_dir, logger)
+ if len(ctx.methods_run) > 1 and ctx.method_breakdown is not None:
+ write_methods_tsv(
+ ctx.method_breakdown, ctx.methods_run, output_dir, logger, ctx.methods_not_run
+ )
+```
+
+with:
+
+```python
+ # The files written beside the report, in the order the footer lists them. Built as
+ # they are written, so the footer never names a file that does not exist or a plot
+ # in a format that was not asked for.
+ companions: list[str] = []
+
+ write_regions_tsv(regions, output_dir, logger)
+ companions.append("recombination_regions.tsv")
+ if len(ctx.methods_run) > 1 and ctx.method_breakdown is not None:
+ write_methods_tsv(
+ ctx.method_breakdown, ctx.methods_run, output_dir, logger, ctx.methods_not_run
+ )
+ companions.append("recombination_methods.tsv")
+ write_coverage_tsv(gaps, ctx.coverage_threshold, output_dir, logger)
+ companions.append("coverage_gaps.tsv")
+ if ctx.signal is not None:
+ write_profile_tsv(ctx.signal, output_dir, logger)
+ companions.append("recombination_profile.tsv")
+ write_winners_tsv(analysis, output_dir, logger)
+ write_stats_tsv(analysis, output_dir, logger)
+ write_windows_tsv(result, per_window_winners, output_dir, logger)
+ companions += ["window_winners.tsv", "similarity_stats.tsv", "similarity_windows.tsv"]
+```
+
+In `src/tessera/recomb/report.py`, replace:
+
+```python
+ plot_top_n(result, top_datasets, regions, output_dir, plot_format, logger)
+ if ctx.site_result is not None:
+ write_site_windows_tsv(
+ ctx.site_result, winners_per_window(ctx.site_result), output_dir, logger
+ )
+ plot_top_n(
+ ctx.site_result, top_datasets, regions, output_dir, plot_format, logger,
+ ylabel="Identity to query at informative sites",
+ title=f"Identity at informative sites, query {result.query}",
+ stem="informative_sites_top",
+ )
+
+ pair = pair_datasets(regions, rank_datasets(ranking, 2))
+ plot_pairwise(result, pair, regions, output_dir, plot_format, logger)
+
+ write_html_report(
+ result, analysis, regions, top_datasets, provenance, output_dir, logger, ctx
+ )
+```
+
+with:
+
+```python
+ plots = [plot_top_n(result, top_datasets, regions, output_dir, plot_format, logger)]
+ if ctx.site_result is not None:
+ write_site_windows_tsv(
+ ctx.site_result, winners_per_window(ctx.site_result), output_dir, logger
+ )
+ companions.append("informative_site_windows.tsv")
+ plots.append(plot_top_n(
+ ctx.site_result, top_datasets, regions, output_dir, plot_format, logger,
+ ylabel="Identity to query at informative sites",
+ title=f"Identity at informative sites, query {result.query}",
+ stem="informative_sites_top",
+ ))
+
+ pair = pair_datasets(regions, rank_datasets(ranking, 2))
+ plots.append(plot_pairwise(result, pair, regions, output_dir, plot_format, logger))
+ # A plot function returns None when it had nothing to draw and wrote no file.
+ companions += [path.name for path in plots if path is not None]
+ if (output_dir / "run_provenance.json").exists():
+ companions.append("run_provenance.json")
+
+ write_html_report(
+ result, analysis, regions, top_datasets, provenance, output_dir, logger, ctx,
+ companion_files=companions,
+ )
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+def _footer_html(provenance: dict[str, str]) -> str:
+ files = [
+ "recombination_regions.tsv", "recombination_methods.tsv", "coverage_gaps.tsv",
+ "recombination_profile.tsv", "window_winners.tsv", "similarity_stats.tsv",
+ "similarity_windows.tsv", "similarity_top*.pdf", "similarity_pair.pdf",
+ ]
+ flist = ", ".join(f"{f}" for f in files)
+ ver = html.escape(provenance.get("tessera version", ""))
+ date = html.escape(provenance.get("date (UTC)", ""))
+ return (
+ f''
+ )
+```
+
+with:
+
+```python
+def _footer_html(provenance: dict[str, str], companion_files: list[str] | None = None) -> str:
+ """The version line and the companion files that were written with this report.
+
+ ``companion_files`` comes from the writer that produced them. With none given the
+ sentence is left out: a list is only worth printing if it is the list of this run.
+ """
+ ver = html.escape(provenance.get("tessera version", ""))
+ date = html.escape(provenance.get("date (UTC)", ""))
+ companions = ""
+ if companion_files:
+ flist = ", ".join(f"{html.escape(f)}" for f in companion_files)
+ companions = f"
Tessera segments the query against the reference panel with an HMM '
+ '(jpHMM-style) and reports a region only when its donor beats the major parent on the '
+ 'sites that distinguish them (a sign test on discordant sites, immune to window '
+ 'overlap; Benjamini-Hochberg FDR across segments), with a posterior breakpoint '
+ 'interval. By default it runs an ensemble of callers (HMM and the 3SEQ triplet '
+ 'test) and merges their regions into one consensus, so a region found by more than '
+ 'one method is flagged as agreeing and treated as higher confidence (see the '
+ 'caller line under Run parameters). It remains an indicative screen, not a full '
+ 'phylogenetic test (e.g. GARD) -- confirm strong candidates.
'
+```
+
+with:
+
+```python
+ '
Tessera runs one or more region callers on the same alignment and '
+ 'merges their regions into one consensus; a region found by more than one caller '
+ 'is flagged as agreeing and treated as higher confidence. The callers that ran '
+ 'for this report are named in the caller line under Run parameters. The default '
+ 'ensemble is four callers: an HMM segmentation of the query against the reference '
+ 'panel (jpHMM-style), which reports a segment only when its donor beats the major '
+ 'parent on the sites that distinguish them (a one-sided sign test on discordant '
+ 'sites, with a posterior breakpoint interval); the 3SEQ triplet test; MaxChi; and '
+ 'Bootscan. Each caller applies Benjamini-Hochberg correction within its own '
+ 'candidates; nothing is corrected across callers or across the genome. This is an '
+ 'indicative screen, not a full phylogenetic test (e.g. GARD) -- confirm strong '
+ 'candidates.
'
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ ctx: ReportContext,
+) -> Path:
+ """Write a single self-contained ``report.html``."""
+```
+
+with:
+
+```python
+ ctx: ReportContext,
+ companion_files: list[str] | None = None,
+) -> Path:
+ """Write a single self-contained ``report.html``.
+
+ ``companion_files`` are the names of the files written beside it, listed in the
+ footer; ``write_reports`` supplies them.
+ """
+```
+
+In `src/tessera/recomb/report_html.py`, replace:
+
+```python
+ f"{_footer_html(provenance)}"
+```
+
+with:
+
+```python
+ f"{_footer_html(provenance, companion_files)}"
+```
+
+In `src/tessera/recomb/report_assets.py`, replace:
+
+```python
+ ("Support",
+ "The share of distinguishing (discordant) sites -- where the query matches one "
+ "candidate parent but not the other -- that favour the donor. 0.5 = no "
+ "preference, 1.0 = every distinguishing site favours the donor."),
+ ("q-value",
+ "The sign-test p-value after Benjamini-Hochberg correction across all candidate "
+ "segments (false-discovery-rate control). A region is reported when q <= alpha."),
+```
+
+with:
+
+```python
+ ("Support",
+ "The supporting statistic of the test behind the region; its meaning depends on "
+ "the caller, and the statistic column of recombination_regions.tsv names it. For "
+ "the HMM it is the share of distinguishing (discordant) sites -- where the query "
+ "matches one candidate parent but not the other -- that favour the donor "
+ "(0.5 = no preference, 1.0 = every distinguishing site favours the donor)."),
+ ("q-value",
+ "The p-value of the test behind the region after Benjamini-Hochberg correction "
+ "across that caller's own candidates (false-discovery-rate control within one "
+ "caller). A region is reported when q <= alpha. For a region several callers "
+ "found, the most significant caller's value is shown."),
+```
+
+In `src/tessera/recomb/report_assets.py`, replace:
+
+```python
+ ("3SEQ triplet test",
+ "Boni MF, Posada D, Feldman MW (2007). An exact nonparametric method for inferring "
+ "mosaic structure in sequence triplets. Genetics 176(2):1035-1047."),
+```
+
+with:
+
+```python
+ ("3SEQ triplet test",
+ "Boni MF, Posada D, Feldman MW (2007). An exact nonparametric method for inferring "
+ "mosaic structure in sequence triplets. Genetics 176(2):1035-1047."),
+ ("MaxChi",
+ "Maynard Smith J (1992). Analyzing the mosaic structure of genes. Journal of "
+ "Molecular Evolution 34(2):126-129."),
+ ("Bootscan",
+ "Salminen MO, Carr JK, Burke DS, McCutchan FE (1995). Identification of breakpoints "
+ "in intergenotypic recombinants of HIV type 1 by bootscanning. AIDS Research and "
+ "Human Retroviruses 11(11):1423-1425."),
+```
+
+In `src/tessera/cli/cmd_recomb.py`, replace:
+
+```python
+ "the default is hmm,3seq,maxchi,bootscan (all but the legacy heuristic). Callers: "
+ "hmm (HMM segmentation + a discordant-site "
+ "significance test), 3seq (scan-aware triplet max-drawdown test; strong at low "
+ "divergence), maxchi (chi-square triplet test, complementary to 3seq), bootscan "
+ "(distance + bootstrap support for the closest parent), barcode (clade-marker "
+ "lineage attribution; needs typed references), heuristic (legacy margin/merge). "
+ "Pass a single name (e.g. --method hmm) for one caller.",
+```
+
+with:
+
+```python
+ "the default is hmm,3seq,maxchi,bootscan; geneconv, barcode and heuristic are "
+ "opt-in, and 'all' runs every one. Callers: "
+ "hmm (HMM segmentation + a discordant-site "
+ "significance test), 3seq (scan-aware triplet max-drawdown test; strong at low "
+ "divergence), maxchi (chi-square triplet test, complementary to 3seq), bootscan "
+ "(distance + bootstrap support for the closest parent), geneconv (longest "
+ "uninterrupted donor-match run), barcode (clade-marker "
+ "lineage attribution; needs typed references), heuristic (legacy margin/merge). "
+ "Pass a single name (e.g. --method hmm) for one caller.",
+```
+
+In `src/tessera/cli/cmd_detect.py`, replace:
+
+```python
+ help="Region caller(s): a comma-separated list of hmm/3seq/maxchi/bootscan/"
+ "heuristic, or 'all'. Several run as an ensemble and their regions are merged "
+ "(default hmm,3seq,maxchi,bootscan).",
+```
+
+with:
+
+```python
+ help="Region caller(s): a comma-separated list of hmm/3seq/maxchi/bootscan/"
+ "geneconv/barcode/heuristic, or 'all'. Several run as an ensemble and their "
+ "regions are merged (default hmm,3seq,maxchi,bootscan).",
+```
+
+In `src/tessera/cli/cmd_fill_references.py`, replace:
+
+```python
+ help="Region caller(s) for the detection step: a comma-separated list of "
+ "hmm/3seq/maxchi/bootscan/heuristic, or 'all'. Several run as an ensemble "
+ "(default hmm,3seq,maxchi,bootscan).",
+```
+
+with:
+
+```python
+ help="Region caller(s) for the detection step: a comma-separated list of "
+ "hmm/3seq/maxchi/bootscan/geneconv/barcode/heuristic, or 'all'. Several run as "
+ "an ensemble (default hmm,3seq,maxchi,bootscan).",
+```
+
+In `docs/detection-methods.md`, replace:
+
+```markdown
+### Low-divergence panels (intra-species sets, DNA viruses)
+```
+
+with:
+
+```markdown
+### Near-duplicate references (lineage clustering)
+
+A recruited panel often holds several near-identical genomes of one lineage. Competed
+individually they tie in every window and fragment the call, so the HMM caller first
+pools them (`--cluster-lineages`, on by default; hmm only). Two references are pooled
+when their identity stays at or above 98.5 % in every window, with no region-sized run
+below it. The pooled lineage competes as one state under the label of its best-covering
+member, and a region names that member. Clustering is skipped for panels of fewer than 4
+or more than 200 references.
+
+One limitation follows from the absolute threshold. On a panel in which *every* pair of
+references is at least 98.5 % identical in every window -- mpox, VZV and other sets below
+roughly 1.5 % divergence -- all references pool into a single lineage and the HMM has
+nothing left to compete, so it calls no region whatever the data. The site-based callers
+(3SEQ, MaxChi) are unaffected. On such a panel, run with `--no-cluster-lineages` to keep
+the HMM's vote. `run_provenance.json` records whether clustering was on.
+
+### Low-divergence panels (intra-species sets, DNA viruses)
+```
+
+In `docs/reference-panels.md`, replace:
+
+```markdown
+lineage saturates `nt` and no parental lineage can be recruited by similarity. A broad
+fetch is capped (`--fetch-limit`, default 2000) and dereplicated; for a heavily
+sequenced taxon the capped sample may miss lineages, so a curated `--candidate-pool`
+is recommended (and the run says so). Whether the diversity panel actually contains the
+```
+
+with:
+
+```markdown
+lineage saturates `nt` and no parental lineage can be recruited by similarity. A broad
+fetch is not truncated: the whole set is downloaded and dereplicated locally, and the run
+logs a notice when it exceeds `--fetch-limit` (default 2000) because that step can take
+a few minutes. For a heavily sequenced taxon a curated `--candidate-pool` is the faster
+route. Whether the diversity panel actually contains the
+```
+
+In `example_data/README.md`, replace:
+
+```markdown
+The default ensemble also runs 3SEQ, which pools the discriminating sites into an exact
+triplet test and recovers the event (q-value ~1e-12, `methods` = 3seq):
+```
+
+with:
+
+```markdown
+The default ensemble also runs the site-based callers, which pool the discriminating
+sites into triplet tests and recover the event (q-value ~1e-12, `methods` = 3seq,maxchi):
+```
+
+- [ ] **Step 4: Run the tests to verify they pass**
+
+Run: `pytest tests/unit/test_report_stale_text.py tests/unit/test_cli_commands.py tests/integration -q`
+Expected: PASS -- `58 passed, 1 skipped`
+
+Run: `ruff check src tests validation && mypy src`
+Expected: `All checks passed!` and `Success: no issues found in 79 source files`
+
+- [ ] **Step 5: Commit**
+
+```bash
+git add tests/unit/test_report_stale_text.py src/tessera/recomb/report.py src/tessera/recomb/report_html.py src/tessera/recomb/report_assets.py src/tessera/cli/cmd_recomb.py src/tessera/cli/cmd_detect.py src/tessera/cli/cmd_fill_references.py docs/detection-methods.md docs/reference-panels.md example_data/README.md
+git commit -m "Bring report, help and documentation text up to date" -m "Co-Authored-By: Claude Fable 5.1 "
+```
+
+---
+
+### Task 11: Housekeeping: unused dependency, version probe, validation README (B10)
+
+`seaborn` is a declared runtime dependency that nothing imports. `sibeliaz -v` is rejected by the wrapper (`illegal option -- v`, exit 1) and that line is recorded as the aligner version. `validation/README.md` says the clonal control reports 0 regions and "7 PASS, 0 FAIL", and its "default agreement gate" is the harness default, not the CLI default.
+
+**Files:**
+- Create: `tests/unit/test_dependencies.py`
+- Modify: `src/tessera/core/binaries.py`
+- Modify: `src/tessera/aligners/sibeliaz.py`
+- Modify: `pyproject.toml`
+- Modify: `validation/README.md`
+- Test (extend): `tests/unit/test_binaries.py`
+
+**Interfaces:**
+- Consumes: nothing new.
+- Produces: `BinarySpec.version_args: tuple[str, ...] | None` (`None` = the tool has no version option; it is not executed and its version is `unknown`).
+
+- [ ] **Step 1: Write the failing tests**
+
+Append to the end of `tests/unit/test_binaries.py`:
+
+```python
+# --- a failed probe is not a version --------------------------------------
+
+def failing_binary(directory: Path, name: str, output: str) -> Path:
+ """An executable that prints ``output`` to stderr and exits 1, like a tool given an
+ option it does not have."""
+ path = directory / name
+ path.write_text(f'#!/bin/sh\nprintf %s "{output}" >&2\nexit 1\n')
+ path.chmod(path.stat().st_mode | stat.S_IEXEC | stat.S_IXGRP | stat.S_IXOTH)
+ return path
+
+
+def test_error_output_is_not_recorded_as_a_version(on_path) -> None:
+ """`sibeliaz -v` prints "illegal option -- v" and exits 1; that line went into the
+ provenance sidecar as the aligner version."""
+ failing_binary(on_path, "sibeliaz", "sibeliaz: illegal option -- v")
+ versions = check_binaries((BinarySpec("sibeliaz", version_args=("-v",)),))
+ assert versions["sibeliaz"] == "unknown"
+
+
+def test_failed_probe_that_still_prints_a_version_is_kept(on_path) -> None:
+ # Some tools print their version in a usage message and exit non-zero.
+ failing_binary(on_path, "usagey", "usagey 1.4.2 -- usage: usagey [options]")
+ assert check_binaries((BinarySpec("usagey"),))["usagey"] == "1.4.2"
+
+
+def test_tool_without_a_version_option_is_not_probed(on_path, tmp_path) -> None:
+ marker = tmp_path / "was_run"
+ path = on_path / "noversion"
+ path.write_text(f'#!/bin/sh\ntouch "{marker}"\n')
+ path.chmod(path.stat().st_mode | stat.S_IEXEC | stat.S_IXGRP | stat.S_IXOTH)
+ versions = check_binaries((BinarySpec("noversion", version_args=None),))
+ assert versions == {"noversion": "unknown"}
+ assert not marker.exists() # declared as having no version option: never executed
+
+
+def test_sibeliaz_declares_no_version_probe() -> None:
+ from tessera.aligners.sibeliaz import SibeliazAligner
+
+ (spec,) = SibeliazAligner.capabilities.required_binaries
+ assert spec.version_args is None
+```
+
+Create `tests/unit/test_dependencies.py`:
+
+```python
+"""Every declared runtime dependency is one the package imports.
+
+Tessera is dependency-light by design, so a declared dependency nothing imports is a
+cost with no benefit: it is installed for every user and audited in CI.
+"""
+
+from __future__ import annotations
+
+import re
+import tomllib
+from pathlib import Path
+
+REPO = Path(__file__).resolve().parents[2]
+
+# Distribution name -> import name, where they differ.
+IMPORT_NAME = {"biopython": "Bio"}
+
+
+def test_every_runtime_dependency_is_imported() -> None:
+ project = tomllib.loads((REPO / "pyproject.toml").read_text())["project"]
+ names = [re.split(r"[<>=!~\[ ;]", dep, maxsplit=1)[0] for dep in project["dependencies"]]
+ source = "\n".join(p.read_text() for p in (REPO / "src" / "tessera").rglob("*.py"))
+ unused = [
+ name for name in names
+ if not re.search(rf"^\s*(?:import|from)\s+{IMPORT_NAME.get(name, name)}\b",
+ source, flags=re.M)
+ ]
+ assert unused == []
+```
+
+`test_failed_probe_that_still_prints_a_version_is_kept` passes before and after: it pins what must not change (a tool that prints its version in a usage message and exits non-zero).
+
+- [ ] **Step 2: Run the tests to verify they fail**
+
+Run: `pytest tests/unit/test_binaries.py tests/unit/test_dependencies.py -q`
+Expected: FAIL --
+
+```text
+FAILED tests/unit/test_binaries.py::test_error_output_is_not_recorded_as_a_version
+FAILED tests/unit/test_binaries.py::test_tool_without_a_version_option_is_not_probed
+FAILED tests/unit/test_binaries.py::test_sibeliaz_declares_no_version_probe
+FAILED tests/unit/test_dependencies.py::test_every_runtime_dependency_is_imported
+4 failed, 20 passed in 1.76s
+```
+
+- [ ] **Step 3: Implement**
+
+In `src/tessera/core/binaries.py`, replace:
+
+```python
+ ``version_args`` is the argument vector that prints a version (e.g.
+ ``("--version",)``). ``min_version`` is an optional dotted-string requirement.
+ """
+
+ name: str
+ version_args: tuple[str, ...] = field(default=("--version",))
+```
+
+with:
+
+```python
+ ``version_args`` is the argument vector that prints a version (e.g.
+ ``("--version",)``), or ``None`` for a tool that has no version option: it is then
+ not executed at all and its version is recorded as ``unknown``. ``min_version`` is
+ an optional dotted-string requirement.
+ """
+
+ name: str
+ version_args: tuple[str, ...] | None = field(default=("--version",))
+```
+
+In `src/tessera/core/binaries.py`, replace:
+
+```python
+ blob = (proc.stdout or "") + (proc.stderr or "")
+ parsed = _parse_version(blob)
+ if parsed:
+ return ".".join(map(str, parsed))
+ return blob.strip().splitlines()[0] if blob.strip() else None
+```
+
+with:
+
+```python
+ blob = (proc.stdout or "") + (proc.stderr or "")
+ parsed = _parse_version(blob)
+ if parsed:
+ return ".".join(map(str, parsed))
+ # No dotted version in the output. A tool that exited cleanly said something about
+ # itself (a build date, say) and that is worth keeping; one that exited non-zero
+ # printed an error about the option, which is not a version.
+ if proc.returncode != 0:
+ return None
+ return blob.strip().splitlines()[0] if blob.strip() else None
+```
+
+In `src/tessera/core/binaries.py`, replace:
+
+```python
+ reported = _query_version(spec.name, spec.version_args)
+ versions[spec.name] = reported or "unknown"
+```
+
+with:
+
+```python
+ reported = (
+ None if spec.version_args is None
+ else _query_version(spec.name, spec.version_args)
+ )
+ versions[spec.name] = reported or "unknown"
+```
+
+In `src/tessera/aligners/sibeliaz.py`, replace:
+
+```python
+ required_binaries=(BinarySpec("sibeliaz", version_args=("-v",)),),
+```
+
+with:
+
+```python
+ # The sibeliaz wrapper has no version option ("-v" is rejected as an illegal
+ # option), so there is nothing to probe; its version is recorded as unknown.
+ required_binaries=(BinarySpec("sibeliaz", version_args=None),),
+```
+
+In `pyproject.toml`, delete this line:
+
+```toml
+ "seaborn>=0.13,<1.0",
+```
+
+In `validation/README.md`, replace:
+
+```markdown
+It is deliberately sensitive to the failure mode it was built for. With the default
+agreement gate the scan is clean; dropping the gate with `--min-methods 1` surfaces the
+single-caller regions again (measured at 4 replicates: 9/16 runs, 10 false regions,
+almost all from one caller). Treat a non-zero total as a regression to explain.
+```
+
+with:
+
+```markdown
+It is deliberately sensitive to the failure mode it was built for. **The harness's own
+default is `--min-methods 2`; the `tessera` CLI default is `--min-methods 1`**, so a clean
+default harness run does not describe the shipped default. Measured at 3 replicates on
+2026-10-01 (commit `457bdfb`):
+
+| gate | runs with a false region | false regions | source |
+|---|---|---|---|
+| `--min-methods 2` (harness default) | 0/12 (CI 0-24 %) | 0 | -- |
+| `--min-methods 1` (CLI default) | 7/12 (58 %, CI 32-81 %) | 8 | hmm = 8 |
+
+The positive control was detected 3/3 with the correct donor at both gates (median
+breakpoint error 55 bp). Treat a non-zero total at `--min-methods 2` as a regression to
+explain, and quote the `--min-methods 1` row when describing what a default run reports.
+```
+
+In `validation/README.md`, replace:
+
+```markdown
+| `hcv_clonal_1b` | HCV ~9.4 kb | pure genotype-1b (non-recombinant control) | mafft | resolves to 1b throughout; **0 regions** (real-data specificity) |
+```
+
+with:
+
+```markdown
+| `hcv_clonal_1b` | HCV ~9.4 kb | pure genotype-1b (non-recombinant control) | mafft | resolves to 1b throughout, but **currently FAILS**: one 12 bp MaxChi-only region (GT2a donor, q = 0.045) is reported where none is expected (real-data specificity) |
+```
+
+In `validation/README.md`, replace:
+
+```markdown
+Each reproduces its published event (or, for the clonal control, its *absence* of
+recombination) end-to-end; the current run is **7 PASS, 0 FAIL**
+(`orthopox_example` SKIPs until its 7-genome collection is built).
+```
+
+with:
+
+```markdown
+The six recombinant datasets that ran reproduce their published events end-to-end
+(`orthopox_example` SKIPs until its 7-genome collection is built). The clonal control does
+not currently pass: the run on 2026-10-01 was **6 PASS, 1 FAIL** (`hcv_clonal_1b`), **1
+SKIP**. The failing region rests on a single caller at the CLI default `--min-methods 1`;
+it is recorded here as an open specificity item rather than hidden.
+```
+
+- [ ] **Step 4: Run the tests to verify they pass**
+
+Run: `pytest tests/unit/test_binaries.py tests/unit/test_dependencies.py tests/unit/test_plugins.py tests/unit/test_aligner_params.py -q`
+Expected: PASS -- `32 passed`
+
+Run: `ruff check src tests validation && mypy src`
+Expected: `All checks passed!` and `Success: no issues found in 79 source files`
+
+**The numbers in `validation/README.md` were measured on 2026-10-01** with these changes applied: `python validation/run_validation.py` gave PASS for `sarscov2_xbb`, `hiv1_crf`, `norovirus_gii`, `enterovirus_e11`, `hiv_crf02ag`, `hcv_2k1b`; FAIL for `hcv_clonal_1b` (one 12 bp MaxChi-only region, donor `GT2a_JFH1`, q = 0.0451); SKIP for `orthopox_example`. If the fetched datasets are present under `validation/data/` and mafft and minimap2 are available, re-run it and correct the README where your result differs:
+
+```bash
+PY=$(which python)
+PATH="$PATH:$HOME/miniforge3/envs/recomfi-aln/bin" $PY validation/run_validation.py | tail -12
+```
+
+If the data are not present, keep the text as written: it states the date it was measured.
+
+- [ ] **Step 5: Commit**
+
+```bash
+git add tests/unit/test_binaries.py tests/unit/test_dependencies.py src/tessera/core/binaries.py src/tessera/aligners/sibeliaz.py pyproject.toml validation/README.md
+git commit -m "Drop the unused seaborn dependency and stop recording a failed version probe" -m "Co-Authored-By: Claude Fable 5.1 "
+```
+
+---
+
+### Task 12: Close the remaining CLI validation gaps (B11)
+
+Inputs that are accepted and fail later, or are silently ignored: `find-references --msa ` and `recomb -o ` give "Unexpected error"; `--max-rounds 0` exits 0 having built nothing; `reassort` accepts `--ani-floor 500`, `--margin -3` and a `--dataset` key that matches no segment; `type-lineages` rejects `.fasta.gz` / `.fas` collections the other commands accept; a multi-line failure note breaks `segment_scan.tsv`.
+
+**Files:**
+- Modify: `src/tessera/cli/main.py`
+- Modify: `src/tessera/cli/cmd_recomb.py`
+- Modify: `src/tessera/cli/cmd_find_references.py`
+- Modify: `src/tessera/cli/cmd_detect.py`
+- Modify: `src/tessera/cli/cmd_build_panel.py`
+- Modify: `src/tessera/cli/cmd_fill_references.py`
+- Modify: `src/tessera/cli/cmd_reassort.py`
+- Modify: `src/tessera/cli/cmd_type_lineages.py`
+- Test (extend): `tests/unit/test_cli_input_checks.py`
+
+**Interfaces:**
+- Consumes: `_require_lineage_map` and the import lists from Task 6; `core.io.read_fasta`, `core.io.collection_genomes`, `core.io._require_fasta`.
+- Produces: `_require_output_directory(path: Path) -> None` and `_require_scan_windows(window_size: int, window_step: int) -> None` in `cli/main.py`.
+
+- [ ] **Step 1: Write the failing tests**
+
+Append to the end of `tests/unit/test_cli_input_checks.py`:
+
+```python
+# --- paths ------------------------------------------------------------------
+
+def test_find_references_rejects_a_missing_msa(tmp_path: Path, no_pipeline) -> None:
+ result = runner.invoke(app, [
+ "find-references", "--msa", str(tmp_path / "nope.fasta"), "--query", "query",
+ "--output", str(tmp_path / "out"),
+ ])
+ _rejected(result, "MSA file not found")
+
+
+def test_recomb_rejects_an_output_path_that_is_a_file(tmp_path: Path) -> None:
+ occupied = tmp_path / "out"
+ occupied.write_text("not a directory\n")
+ result = runner.invoke(app, [
+ "recomb", "--msa", EXAMPLE, "--query", "query", "--output", str(occupied),
+ ])
+ _rejected(result, "is not a directory")
+ assert occupied.read_text() == "not a directory\n" # left untouched
+
+
+# --- numeric options on the panel-building commands ------------------------
+
+@pytest.mark.parametrize("command", ["detect", "fill-references", "build-panel"])
+@pytest.mark.parametrize(
+ ("option", "value"),
+ [("--max-rounds", "0"), ("--window-size", "0"), ("--window-step", "0")],
+)
+def test_panel_commands_reject_out_of_range_options(
+ tmp_path: Path, no_pipeline, command: str, option: str, value: str
+) -> None:
+ """`fill-references --max-rounds 0` exited 0 having built no alignment and no
+ report; a bad window reached the scan only after seeding and alignment."""
+ result = runner.invoke(app, [
+ command, "--query", str(_query(tmp_path)), "--output", str(tmp_path / "out"),
+ option, value,
+ ])
+ _rejected(result, f"Invalid {option}")
+
+
+@pytest.mark.parametrize(("option", "value"), [("--window-size", "0"), ("--window-step", "0")])
+def test_find_references_rejects_out_of_range_windows(
+ tmp_path: Path, no_pipeline, option: str, value: str
+) -> None:
+ result = runner.invoke(app, [
+ "find-references", "--msa", EXAMPLE, "--query", "query",
+ "--output", str(tmp_path / "out"), option, value,
+ ])
+ _rejected(result, f"Invalid {option}")
+
+
+# --- reassort ---------------------------------------------------------------
+
+def _segments(tmp_path: Path) -> Path:
+ path = tmp_path / "segments.fasta"
+ path.write_text(">HA\nACGTACGT\n>NA\nACGTACGT\n")
+ return path
+
+
+@pytest.fixture
+def no_reassort(monkeypatch):
+ monkeypatch.setattr("tessera.cli.cmd_reassort.assign_segments", _must_not_run)
+
+
+@pytest.mark.parametrize(
+ ("option", "value", "needle"),
+ [("--ani-floor", "500", "Invalid --ani-floor"),
+ ("--ani-floor", "-1", "Invalid --ani-floor"),
+ ("--margin", "-3", "Invalid --margin")],
+)
+def test_reassort_rejects_out_of_range_options(
+ tmp_path: Path, no_reassort, option: str, value: str, needle: str
+) -> None:
+ """A negative margin makes every near-best set empty (a clonal pair then reads as
+ undetermined); an ANI floor above 100 leaves every segment unassigned. Both exited 0."""
+ result = runner.invoke(app, [
+ "reassort", "--query", str(_segments(tmp_path)), "--output", str(tmp_path / "out"),
+ option, value,
+ ])
+ _rejected(result, needle)
+
+
+def test_reassort_rejects_a_dataset_override_for_an_unknown_segment(
+ tmp_path: Path, no_reassort
+) -> None:
+ """A typo in the segment name silently fell back to dataset auto-detection."""
+ result = runner.invoke(app, [
+ "reassort", "--query", str(_segments(tmp_path)), "--output", str(tmp_path / "out"),
+ "--dataset", "HAA=nextstrain/flu/h3n2/ha",
+ ])
+ _rejected(result, "HAA")
+ assert "HA, NA" in " ".join(result.output.split()) # names the segments that do exist
+
+
+def test_segment_scan_tsv_keeps_one_row_per_segment(tmp_path: Path, monkeypatch) -> None:
+ """An aligner error message spans several lines and may hold tabs; written verbatim
+ into the note column it broke the table."""
+ from tessera.reassort.assign import ReassortmentResult, SegmentAssignment
+ from tessera.reassort.scan import SegmentScan
+
+ def fake_assign(query, **kwargs):
+ return ReassortmentResult(
+ segments=[SegmentAssignment("HA", "dataset", None, None, 0.0, "unassigned")],
+ verdict="undetermined", groups=[], pair_notes=[],
+ scans=[SegmentScan("HA", False, False, 0,
+ "scan failed: mafft failed:\nline1\tx\nline2")],
+ )
+
+ monkeypatch.setattr("tessera.cli.cmd_reassort.assign_segments", fake_assign)
+ out = tmp_path / "out"
+ result = runner.invoke(app, [
+ "reassort", "--query", str(_segments(tmp_path)), "--output", str(out),
+ ])
+ assert result.exit_code == 0, result.output
+ lines = (out / "segment_scan.tsv").read_text().splitlines()
+ assert len(lines) == 2 # header + one segment
+ assert all(len(line.split("\t")) == 4 for line in lines)
+ assert lines[1].split("\t")[3] == "scan failed: mafft failed: line1 x line2"
+
+
+# --- type-lineages reads the same collection the other commands do ---------
+
+def test_type_lineages_accepts_what_a_collection_may_hold(tmp_path: Path, monkeypatch) -> None:
+ """Every other command treats each file in the collection as a genome, whatever its
+ extension and gzip-compressed or not; type-lineages filtered on three suffixes and
+ reported "No FASTA genomes found"."""
+ import gzip
+
+ collection = tmp_path / "collection"
+ collection.mkdir()
+ with gzip.open(collection / "a.fasta.gz", "wt") as fo:
+ fo.write(">a\nACGTACGT\n")
+ (collection / "b.fas").write_text(">b\nACGTACGT\n")
+ (collection / ".DS_Store").write_text("not a genome")
+ seen: list[str] = []
+
+ def fake_assign(genomes, **kwargs):
+ seen.extend(p.name for p in genomes)
+ return [("a", "L1", "denovo"), ("b.fas", "L1", "denovo")]
+
+ monkeypatch.setattr("tessera.discover.lineage_assign.assign_lineages", fake_assign)
+ out = tmp_path / "out"
+ result = runner.invoke(app, [
+ "type-lineages", "--collection", str(collection), "--output", str(out),
+ ])
+ assert result.exit_code == 0, result.output
+ assert seen == ["a.fasta.gz", "b.fas"] # hidden files are not genomes
+ assert (out / "lineages.tsv").exists()
+
+
+def test_type_lineages_rejects_a_file_that_is_not_fasta(tmp_path: Path, no_pipeline) -> None:
+ collection = _collection(tmp_path)
+ (collection / "notes.txt").write_text("these are my notes\n")
+ result = runner.invoke(app, [
+ "type-lineages", "--collection", str(collection), "--output", str(tmp_path / "out"),
+ ])
+ _rejected(result, "does not look like a FASTA file")
+```
+
+`type-lineages` now reads a collection the way every other command does: each non-hidden file is a genome, and a file that is not FASTA is an error rather than something to skip. A collection holding a stray `notes.txt` was silently accepted before and is now rejected by name.
+
+- [ ] **Step 2: Run the tests to verify they fail**
+
+Run: `pytest tests/unit/test_cli_input_checks.py -q`
+Expected: FAIL --
+
+```text
+FAILED tests/unit/test_cli_input_checks.py::test_find_references_rejects_a_missing_msa
+FAILED tests/unit/test_cli_input_checks.py::test_recomb_rejects_an_output_path_that_is_a_file
+FAILED tests/unit/test_cli_input_checks.py::test_panel_commands_reject_out_of_range_options[--max-rounds-0-detect]
+FAILED tests/unit/test_cli_input_checks.py::test_panel_commands_reject_out_of_range_options[--max-rounds-0-fill-references]
+FAILED tests/unit/test_cli_input_checks.py::test_panel_commands_reject_out_of_range_options[--max-rounds-0-build-panel]
+FAILED tests/unit/test_cli_input_checks.py::test_panel_commands_reject_out_of_range_options[--window-size-0-detect]
+FAILED tests/unit/test_cli_input_checks.py::test_panel_commands_reject_out_of_range_options[--window-size-0-fill-references]
+FAILED tests/unit/test_cli_input_checks.py::test_panel_commands_reject_out_of_range_options[--window-size-0-build-panel]
+FAILED tests/unit/test_cli_input_checks.py::test_panel_commands_reject_out_of_range_options[--window-step-0-detect]
+FAILED tests/unit/test_cli_input_checks.py::test_panel_commands_reject_out_of_range_options[--window-step-0-fill-references]
+FAILED tests/unit/test_cli_input_checks.py::test_panel_commands_reject_out_of_range_options[--window-step-0-build-panel]
+FAILED tests/unit/test_cli_input_checks.py::test_find_references_rejects_out_of_range_windows[--window-size-0]
+FAILED tests/unit/test_cli_input_checks.py::test_find_references_rejects_out_of_range_windows[--window-step-0]
+FAILED tests/unit/test_cli_input_checks.py::test_reassort_rejects_out_of_range_options[--ani-floor-500-Invalid --ani-floor]
+FAILED tests/unit/test_cli_input_checks.py::test_reassort_rejects_out_of_range_options[--ani-floor--1-Invalid --ani-floor]
+FAILED tests/unit/test_cli_input_checks.py::test_reassort_rejects_out_of_range_options[--margin--3-Invalid --margin]
+FAILED tests/unit/test_cli_input_checks.py::test_reassort_rejects_a_dataset_override_for_an_unknown_segment
+FAILED tests/unit/test_cli_input_checks.py::test_segment_scan_tsv_keeps_one_row_per_segment
+FAILED tests/unit/test_cli_input_checks.py::test_type_lineages_accepts_what_a_collection_may_hold
+FAILED tests/unit/test_cli_input_checks.py::test_type_lineages_rejects_a_file_that_is_not_fasta
+20 failed, 6 passed in 1.97s
+```
+
+- [ ] **Step 3: Implement**
+
+In `src/tessera/cli/main.py`, replace:
+
+```python
+def _parse_key_values(items: list[str], label: str) -> dict[str, str]:
+```
+
+with:
+
+```python
+def _require_output_directory(path: Path) -> None:
+ """Reject an output path that exists and is not a directory.
+
+ The writers create the directory if it is missing; given an existing file they fail
+ with ``[Errno 17] File exists`` from wherever the first output is written.
+ """
+ if Path(path).exists() and not Path(path).is_dir():
+ raise UserInputError(f"Output path exists and is not a directory: {path}")
+
+
+def _require_scan_windows(window_size: int, window_step: int) -> None:
+ """The window options every scanning command shares; see :func:`_require_range`."""
+ _require_range(window_size, "--window-size", lo=1)
+ _require_range(window_step, "--window-step", lo=1)
+
+
+def _parse_key_values(items: list[str], label: str) -> dict[str, str]:
+```
+
+In `src/tessera/cli/cmd_recomb.py`, replace:
+
+```python
+ _require_lineage_map,
+ _require_range,
+```
+
+with:
+
+```python
+ _require_lineage_map,
+ _require_output_directory,
+ _require_range,
+```
+
+In `src/tessera/cli/cmd_recomb.py`, replace:
+
+```python
+ _require_lineage_map(lineage_map)
+```
+
+with:
+
+```python
+ _require_lineage_map(lineage_map)
+ _require_output_directory(output)
+```
+
+In `src/tessera/cli/cmd_find_references.py`, replace:
+
+```python
+from .main import app, get_logger, stage_errors
+```
+
+with:
+
+```python
+from .main import _require_file, _require_scan_windows, app, get_logger, stage_errors
+```
+
+In `src/tessera/cli/cmd_find_references.py`, replace:
+
+```python
+ with stage_errors(logger):
+ params = FindRefParams(
+```
+
+with:
+
+```python
+ with stage_errors(logger):
+ _require_file(msa, "MSA file")
+ _require_scan_windows(window_size, window_step)
+ params = FindRefParams(
+```
+
+In `src/tessera/cli/cmd_detect.py`, replace:
+
+```python
+ _require_lineage_map,
+ app,
+```
+
+with:
+
+```python
+ _require_lineage_map,
+ _require_range,
+ _require_scan_windows,
+ app,
+```
+
+In `src/tessera/cli/cmd_detect.py`, replace:
+
+```python
+ _require_lineage_map(lineage_map)
+```
+
+with:
+
+```python
+ _require_lineage_map(lineage_map)
+ # Checked here, before any network or aligner work: a round count of zero
+ # builds nothing and still exits 0, and a bad window otherwise surfaces only
+ # after the panel has been recruited and aligned.
+ _require_range(max_rounds, "--max-rounds", lo=1)
+ _require_scan_windows(window_size, window_step)
+```
+
+In `src/tessera/cli/cmd_build_panel.py`, replace:
+
+```python
+ _require_lineage_map,
+ app,
+```
+
+with:
+
+```python
+ _require_lineage_map,
+ _require_range,
+ _require_scan_windows,
+ app,
+```
+
+In `src/tessera/cli/cmd_build_panel.py`, replace:
+
+```python
+ _require_lineage_map(lineage_map)
+```
+
+with:
+
+```python
+ _require_lineage_map(lineage_map)
+ # Checked here, before any network or aligner work: a round count of zero
+ # builds nothing and still exits 0, and a bad window otherwise surfaces only
+ # after the panel has been recruited and aligned.
+ _require_range(max_rounds, "--max-rounds", lo=1)
+ _require_scan_windows(window_size, window_step)
+```
+
+In `src/tessera/cli/cmd_fill_references.py`, replace:
+
+```python
+ _require_lineage_map,
+ app,
+```
+
+with:
+
+```python
+ _require_lineage_map,
+ _require_range,
+ _require_scan_windows,
+ app,
+```
+
+In `src/tessera/cli/cmd_fill_references.py`, replace:
+
+```python
+ _require_lineage_map(lineage_map)
+```
+
+with:
+
+```python
+ _require_lineage_map(lineage_map)
+ # Checked here, before any network or aligner work: a round count of zero
+ # builds nothing and still exits 0, and a bad window otherwise surfaces only
+ # after the panel has been recruited and aligned.
+ _require_range(max_rounds, "--max-rounds", lo=1)
+ _require_scan_windows(window_size, window_step)
+```
+
+In `src/tessera/cli/cmd_reassort.py`, replace:
+
+```python
+from ..core.errors import UserInputError
+from ..reassort import assign_segments
+```
+
+with:
+
+```python
+from ..core.errors import UserInputError
+from ..core.io import read_fasta
+from ..reassort import assign_segments
+```
+
+In `src/tessera/cli/cmd_reassort.py`, replace:
+
+```python
+from .main import _require_file, app, get_logger, stage_errors
+```
+
+with:
+
+```python
+from .main import _require_file, _require_range, app, get_logger, stage_errors
+```
+
+In `src/tessera/cli/cmd_reassort.py`, replace:
+
+```python
+ _require_file(query, "Query file")
+ overrides: dict[str, str] = {}
+ for item in dataset or []:
+ if "=" not in item:
+ raise UserInputError(f"--dataset must be SEGMENT=path, got {item!r}")
+ seg, path = item.split("=", 1)
+ overrides[seg.strip()] = path.strip()
+```
+
+with:
+
+```python
+ _require_file(query, "Query file")
+ # ANI is a percentage. Above 100 nothing can be assigned; a negative margin
+ # leaves every near-best set empty, so a clonal pair reads as undetermined.
+ _require_range(ani_floor, "--ani-floor", lo=0.0, hi=100.0)
+ _require_range(margin, "--margin", lo=0.0)
+ overrides: dict[str, str] = {}
+ for item in dataset or []:
+ if "=" not in item:
+ raise UserInputError(f"--dataset must be SEGMENT=path, got {item!r}")
+ seg, path = item.split("=", 1)
+ overrides[seg.strip()] = path.strip()
+ if overrides:
+ # An override is looked up by segment name; one that matches no record would
+ # be ignored and that segment's dataset auto-detected instead.
+ segments = [name for name, _seq in read_fasta(query) if name]
+ unknown = sorted(set(overrides) - set(segments))
+ if unknown:
+ raise UserInputError(
+ f"--dataset names segment(s) not in the query: {', '.join(unknown)}. "
+ f"Segments in {query.name}: {', '.join(segments) or '(none)'}."
+ )
+```
+
+In `src/tessera/cli/cmd_reassort.py`, replace:
+
+```python
+ fo.write(f"{sc.segment}\t{flag}\t{sc.n_regions}\t{sc.note}\n")
+```
+
+with:
+
+```python
+ # A failure note quotes the aligner's own message, which can span
+ # lines and hold tabs; keep it to one cell.
+ note = " ".join(sc.note.split())
+ fo.write(f"{sc.segment}\t{flag}\t{sc.n_regions}\t{note}\n")
+```
+
+In `src/tessera/cli/cmd_type_lineages.py`, replace:
+
+```python
+from ..core.errors import UserInputError
+```
+
+with:
+
+```python
+from ..core.errors import UserInputError
+from ..core.io import _require_fasta, collection_genomes
+```
+
+In `src/tessera/cli/cmd_type_lineages.py`, replace:
+
+```python
+ genomes = sorted(
+ p for p in collection.iterdir()
+ if p.is_file() and p.suffix.lower() in (".fasta", ".fa", ".fna")
+ )
+ if not genomes:
+ raise UserInputError(f"No FASTA genomes found in {collection}")
+```
+
+with:
+
+```python
+ # The same reading of a collection as every other command: each non-hidden
+ # file is a genome, whatever its extension and gzip-compressed or not, and a
+ # file that is not FASTA is an error rather than something to skip.
+ genomes = collection_genomes(collection)
+ if not genomes:
+ raise UserInputError(f"No genome files found in {collection}")
+ for genome in genomes:
+ _require_fasta(genome)
+```
+
+- [ ] **Step 4: Run the tests to verify they pass**
+
+Run: `pytest tests/unit/test_cli_input_checks.py tests/unit/test_cli_commands.py tests/unit/test_cli_validation.py tests/unit/test_cli_reassort.py -q`
+Expected: PASS -- `79 passed`
+
+Run: `ruff check src tests validation && mypy src`
+Expected: `All checks passed!` and `Success: no issues found in 79 source files`
+
+- [ ] **Step 5: Commit**
+
+```bash
+git add tests/unit/test_cli_input_checks.py src/tessera/cli/main.py src/tessera/cli/cmd_recomb.py src/tessera/cli/cmd_find_references.py src/tessera/cli/cmd_detect.py src/tessera/cli/cmd_build_panel.py src/tessera/cli/cmd_fill_references.py src/tessera/cli/cmd_reassort.py src/tessera/cli/cmd_type_lineages.py
+git commit -m "Validate the remaining CLI inputs before any work starts" -m "Co-Authored-By: Claude Fable 5.1 "
+```
+
+---
+
+### Task 13: Changelog, full gate and harness comparison
+
+**Files:**
+- Modify: `CHANGELOG.md`
+
+**Interfaces:**
+- Consumes: the "before" harness outputs saved in Task 1 (`/tmp/tessera-plan-b/before_gate2.txt`, `before_gate1.txt`).
+- Produces: the pull request.
+
+- [ ] **Step 1: Add the changelog entry**
+
+In `CHANGELOG.md`, replace:
+
+```markdown
+## [Unreleased]
+```
+
+with:
+
+```markdown
+## [Unreleased]
+
+Reporting fixes from the post-1.2.0 audit
+(`docs/superpowers/specs/2026-10-01-post-1.2.0-audit-design.md`, items B1-B11). No region
+call changes: which regions are found, and their coordinates and p-values, are as before.
+
+### Fixed
+
+- **The run provenance named the wrong callers.** MaxChi, Bootscan, GENECONV and the barcode
+ caller were each described as `heuristic (min ... / margin ... / merge ...)` in
+ `run_provenance.json` and the report, so a default run recorded
+ `hmm + 3seq + heuristic + heuristic`. Each caller is now described under its own name, and
+ the record gains the settings that change what is reported: the agreement gate
+ (`--min-methods`), sibling exclusion, lineage clustering and donor re-attribution.
+- **A clean recombinant between divergent parents was reported as a possible missing
+ reference.** A window straddling a breakpoint matches neither parent well on its own, so
+ its best similarity fell below the coverage threshold, the stretch was called a `divergent`
+ coverage gap, and the region was marked `donor_undercovered` -- on the shipped
+ `divergent` example the headline read "low confidence" for a donor identical to the query.
+ A gap within one window of a called region boundary that the region's two parents together
+ explain is now labelled `breakpoint` in `coverage_gaps.tsv` and the report; it does not
+ caveat the region, is not turned into a donor-absent region, and is left out of the
+ headline. **`donor_undercovered` and the confidence wording change for such regions.**
+ Reference recruitment (`fill-references`, `find-references`) is unaffected.
+- The report's "covering N kb (P %) of the query" added the lengths of overlapping regions
+ that name different donors; it now reports their union.
+- After `--reattribute-donors`, `recombination_methods.tsv` and the report's method table
+ kept the donor from before re-attribution.
+- `--lineage-map` pointing at a file that does not exist was ignored (exit 0, untyped
+ report) by `recomb`, `type-lineages`, `detect`, `fill-references` and `build-panel`. It
+ is now an error.
+- **The barcode caller on an untyped panel read as a negative.** It named the first record
+ of the alignment as major parent and its column in the method table said `no`. It now
+ reports no major parent; in an ensemble it is logged and shown as `not run`, the
+ agreement gate counts only callers that ran, and a run that selected only `barcode` is
+ refused.
+- The report judged the PHI p-value at alpha 0.05 whatever `--alpha` was.
+- **A PHI test that could not reject was reported as "no signal".** With no more
+ informative sites than the window can hold (`--phi-window`, default 100) the permutation
+ p-value is 1 for any data. This is now reported as `not testable` (`NA` in the
+ `recombination_profile.tsv` header, `phi_p = None` in the API).
+- Plots labelled a donor-absent region "recombinant: "; `similarity_pair` showed
+ the two leading window winners rather than the major parent and the leading donor; with
+ `--top-n 1` the donor was drawn grey.
+- The report footer listed `.pdf` plots under `--plot-format png` and files that were not
+ written; the methods text and references described a two-caller ensemble; `--method`
+ help omitted `geneconv`.
+- `sibeliaz` was probed with `-v`, which it rejects, and the error line was recorded as the
+ aligner version. A failed probe is no longer recorded as a version.
+- Input checks: `find-references --msa ` and `recomb -o ` failed
+ with "Unexpected error"; `--max-rounds 0` exited 0 having built nothing; `reassort`
+ accepted `--ani-floor 500`, `--margin -3` and a `--dataset` key matching no segment;
+ `type-lineages` rejected `.fasta.gz` / `.fas` collections; a multi-line aligner error
+ broke `segment_scan.tsv`.
+
+### Changed
+
+- `RecombinationSignal.phi_p` is `float | None`.
+- `seaborn` is no longer a dependency; nothing imported it.
+
+### Documentation
+
+- `docs/detection-methods.md` describes lineage clustering (including its limit on panels
+ below about 1.5 % divergence), the `breakpoint` coverage kind and the untestable-PHI case.
+- `validation/README.md` states that the specificity harness defaults to `--min-methods 2`
+ while the CLI defaults to 1, gives the measured rate at each, and records that
+ `hcv_clonal_1b` currently fails.
+```
+
+- [ ] **Step 2: Run the full gate**
+
+Run: `ruff check src tests validation`
+Expected: `All checks passed!`
+
+Run: `pytest -m "not requires_binary" -q`
+Expected: `671 passed, 1 deselected`
+
+Run: `mypy src`
+Expected: `Success: no issues found in 79 source files`
+
+Then repeat the suite the way CI sees it -- a narrow terminal and no aligner or NCBI tools on `PATH` -- so a test that only passes because `skani` or `efetch` is installed locally is caught here:
+
+```bash
+PY=$(which python)
+PATH="$(dirname "$PY"):/usr/bin:/bin" COLUMNS=80 "$PY" -m pytest -m "not requires_binary" -q
+```
+
+Expected: `671 passed, 1 deselected`
+
+- [ ] **Step 3: Run the specificity harness again and compare**
+
+```bash
+python validation/run_specificity.py --reps 3 | tee /tmp/tessera-plan-b/after_gate2.txt
+python validation/run_specificity.py --reps 3 --min-methods 1 | tee /tmp/tessera-plan-b/after_gate1.txt
+diff /tmp/tessera-plan-b/before_gate2.txt /tmp/tessera-plan-b/after_gate2.txt
+diff /tmp/tessera-plan-b/before_gate1.txt /tmp/tessera-plan-b/after_gate1.txt
+```
+
+Expected: both `diff` commands print nothing. This plan changes no region call, so the false-region counts and the positive control must be identical to the baseline (`0/12` and `7/12, hmm=8`; `detected 3/3 | correct donor 3/3 | median breakpoint error 55 bp`). Any difference means a caller's behaviour changed: stop and find out why before going further.
+
+- [ ] **Step 4: Run the hybrids harness if its data and aligners are present**
+
+Only if Task 1 Step 4 produced `/tmp/tessera-plan-b/hybrids_before.txt`. The harness scores detection, donor and backbone from `recombination_regions.tsv`; this plan changes the `donor_undercovered` flag only, so no case's verdict should move.
+
+```bash
+PY=$(which python)
+PATH="$PATH:$HOME/miniforge3/envs/recomfi-aln/bin" $PY validation/run_hybrids.py \
+ | tee /tmp/tessera-plan-b/hybrids_after.txt | tail -5
+grep -E "passed \(" /tmp/tessera-plan-b/hybrids_before.txt /tmp/tessera-plan-b/hybrids_after.txt
+```
+
+Expected: the same `N/M passed (S skipped, E error)` line in both files. Read the per-case lines above it in the two files as well (run times differ between runs, so compare the verdicts, not the whole file) and list any case whose PASS/FAIL/SKIP changed. If the harness could not be run, say so in the pull request -- do not report it as run.
+
+- [ ] **Step 5: Commit and open the pull request**
+
+```bash
+git add CHANGELOG.md
+git commit -m "Record the reporting fixes in the changelog" -m "Co-Authored-By: Claude Fable 5.1 "
+git push -u origin fix-audit-report-faithfulness
+gh pr create --base main --title "Reporting faithfulness: audit items B1-B11" --body "$(cat <<'BODY'
+Implements plan B of the post-1.2.0 audit (docs/superpowers/specs/2026-10-01-post-1.2.0-audit-design.md, items B1-B11): outputs that misstated what a run did. No region call changes.
+
+Verification
+- ruff, mypy clean; pytest -m "not requires_binary": 671 passed (586 before).
+- run_specificity.py --reps 3:
+- run_hybrids.py:
+
+Behaviour a user will notice
+- A region whose only coverage gaps are the windows straddling its breakpoints is no longer flagged donor_undercovered / "low confidence".
+- --method barcode on an untyped panel is refused; in an ensemble it is shown as "not run".
+- The PHI test is reported as "not testable" when there are too few informative sites for the window.
+- A missing --lineage-map file is an error.
+
+🤖 Generated with [Claude Code](https://claude.com/claude-code)
+BODY
+)"
+```
+
+Replace the two `` placeholders in the body with the measured lines before running the command. Do not merge: hand the pull request to the maintainer.
diff --git a/docs/superpowers/plans/2026-10-01-audit-c-caller-behaviour.md b/docs/superpowers/plans/2026-10-01-audit-c-caller-behaviour.md
new file mode 100644
index 0000000..6350464
--- /dev/null
+++ b/docs/superpowers/plans/2026-10-01-audit-c-caller-behaviour.md
@@ -0,0 +1,2983 @@
+# Audit Plan C: Caller Behaviour Implementation Plan
+
+> **For agentic workers:** REQUIRED SUB-SKILL: Use superpowers:subagent-driven-development (recommended) or superpowers:executing-plans to implement this plan task-by-task. Steps use checkbox (`- [ ]`) syntax for tracking.
+
+**Goal:** Fix the caller-behaviour defects C1-C7 of the post-1.2.0 audit, each behind a measured harness gate, and produce the numbers the maintainer needs for the `--min-methods` decision (C8).
+
+**Architecture:** Every task changes what the scan reports, so every task is evaluate-first: a failing test, one candidate implementation, then a harness gate with numeric acceptance criteria and a stated action when the gate fails. Independent fixes come first (3SEQ/MaxChi tract start, p-value underflow, ensemble merge, pool typing, PHI window), then the two that interact (HMM region span, then lineage clustering, which only makes sense once HMM spans are tight), and last a characterisation of the false-positive rate that ends in a decision record rather than a code change. A new self-contained harness (`validation/run_regimes.py`) simulates the two regimes the audit used and no existing harness samples.
+
+**Tech Stack:** Python 3.11+, numpy, pytest, ruff, mypy. No new dependency. Harnesses: `validation/run_specificity.py`, `run_regimes.py` (new), `run_validation.py`, `run_benchmark.py`, `run_hybrids.py`.
+
+**Spec:** `docs/superpowers/specs/2026-10-01-post-1.2.0-audit-design.md` (section "Plan C"). Read it before starting; the "Deviations from the spec" section below lists where this plan departs from it and why.
+
+## Global Constraints
+
+- No new runtime dependency.
+- Modest scientific language in code, docs and messages; reported numbers must be faithful (state what passes, what fails, what was skipped).
+- A new behaviour needs a test that fails without the change. Where a test in this plan is a guard that passes before and after, the plan says so.
+- Anything touching a caller, a default or region calling is validated on `validation/run_specificity.py` at **both** `--min-methods 1` and `--min-methods 2` and on `validation/run_hybrids.py`, before and after, and the numbers go in the PR description.
+- "Could not test" must never be reported as "tested, found nothing".
+- Work on branch `fix-audit-caller-behaviour`. Commit messages explain the problem, the evidence and the trade-off, and end with `Co-Authored-By: Claude Fable 5.1 `.
+- Append the aligner env to `PATH`; never prepend it (prepending swaps the Python interpreter): `export PATH="$PATH:$HOME/miniforge3/envs/recomfi-aln/bin"`.
+- **A failed gate is a result, not an obstacle.** If a task's gate fails: `git revert` that task's commit, write the measured numbers under "Gate results" in the PR description, and move on to the next task. Do not adjust thresholds, constants or the test until the gate passes.
+- The `--min-methods` default is not changed in this branch. Task 10 presents it to the maintainer as a decision.
+- Do not touch plan A or plan B items. In particular the PHI "not testable" wording and the coverage-gap relabelling belong to plan B.
+
+## What was measured while writing this plan
+
+The candidates below were applied in a scratch worktree of `main` at `457bdfb` and measured. These numbers are the reference the gates compare against; the executor re-measures at 10 replicates in Task 1. Replicate counts are small, so the intervals are wide -- read them as "does the failure mode exist", not as rates.
+
+| measurement | main (`457bdfb`) | with C3, C4, C7, C2 | with all candidates |
+|---|---|---|---|
+| `run_specificity.py --reps 3 --min-methods 2`: runs with a false region | 0/12 | 0/12 | 0/12 |
+| `run_specificity.py --reps 3 --min-methods 1`: runs with a false region | 7/12 (8 regions, hmm = 8) | 7/12 (8, hmm = 8) | 7/12 (8, hmm = 8) |
+| positive control (both gates): detected / donor / median breakpoint error | 3/3, 3/3, 55 bp | 3/3, 3/3, 13 bp | 3/3, 3/3, 13 bp |
+| `run_validation.py` | 6 PASS, 1 FAIL (`hcv_clonal_1b`), 1 SKIP | same verdicts, same regions, coordinates change | same |
+| `run_regimes.py --reps 10`, near-identical: detected / HMM among callers / donor / error | 10/10, 0/10, 10/10, 465 bp | 6/6, 0/6, 6/6, 263 bp (6 replicates; `main` gives the same at 6) | 10/10, 10/10, 10/10, 465 bp |
+| near-identical clonal, 20 panels, `--min-methods 1`: runs with a false region | 0/20 | not run at 20 (0/6 in the regime harness) | 1/20 (hmm) |
+| near-identical clonal, 20 panels, `--min-methods 2` | 0/20 | not run | 0/20 |
+| sibling tract (true difference 600 bp), 10 reps: median reported length | 11700 bp | 1001 bp (6 replicates) | 1001 bp |
+| `run_benchmark.py` (PHI): power / specificity | 8/185, 36/36 | -- | 13/185, 36/36 |
+
+**Not measured here:** `run_hybrids.py`. It downloads the Nextclade index, reference and tree for every case, and this session ran without network. It is an executor gate for every task that names it. C3 and C4 were measured together, not separately. C7 was measured only inside the cumulative stack.
+
+## Deviations from the spec
+
+1. **C2 -- the sign test stays on the HMM segment; only the reported span is trimmed.** The spec says to recompute the sign test on the trimmed span. Trimming removes flanking sites that favour the major, so recomputing would lower p-values on a span chosen by looking at the data, in the caller that already produces every false region at `--min-methods 1`. Keeping the test on the segment means the set of called regions cannot change through this step; the measurements above confirm it (same eight false regions, same validation verdicts). The similarities are recomputed on the reported span.
+2. **C2 -- a segment touching the first or last window is searched to the alignment end.** Trimming can only shrink a span, so it cannot fix a tract that starts at column 0 and is reported from the first window centre. This small extension does.
+3. **C2 -- one alternative was tried and not chosen.** Taking the 3SEQ maximum-descent interval over the full extent of the segment's windows gave a slightly lower breakpoint error on the positive control (9 bp against 13 bp) but changed rows on real data: on `hiv_crf02ag` one HMM region grew by 61 bp, absorbed a coverage gap and removed a donor-absent row. The chosen rule changes coordinates only.
+4. **C5 -- the rule is `min(--phi-window, sites // 10)`.** The spec left the rule open. Six rules were compared (Task 7). The benchmark does not separate them strongly: PHI power on it is low under every rule (see Task 7's note on a finding outside this plan's scope).
+5. **C6 -- pool labels are passed explicitly, not inferred from the header shape.** The spec says "when a header comes from a pool, read the second token". Cached pools hold multi-word clades (`clade 2 wild-type`, `Clade VI`), so the label is everything after the accession; and a user's `--candidate-pool` holds free-text titles, so whether a header is structured is decided by where the genome came from, carried in a small sidecar.
+6. **C1 -- "specificity no worse" holds on `run_specificity.py` but not strictly on the new near-identical clonal panels** (1/20 against 0/20 at `--min-methods 1`). That is the cost of the HMM having a vote there at all; Task 9 states it and Task 10 carries it into the decision.
+
+## File Structure
+
+| File | Change | Responsibility |
+|---|---|---|
+| `validation/run_regimes.py` | create | simulate and score the near-identical and sibling-tract regimes (opt-in harness) |
+| `tests/unit/test_regime_scoring.py` | create | unit tests for that harness's simulation and scoring |
+| `tests/integration/test_regimes.py` | create | one-seed pipeline regressions for C1 and C2 |
+| `src/tessera/recomb/threeseq.py` | modify | C3 tract start, C4 exact p-value |
+| `src/tessera/recomb/maxchi.py` | modify | C3: the permutation null uses the same tract rule |
+| `src/tessera/recomb/ensemble.py` | modify | C7 merge edge cases |
+| `src/tessera/recomb/typing.py` | modify | C6: read a pool label verbatim |
+| `src/tessera/discover/iterate.py` | modify | C6: pass pool labels to selection and panel typing |
+| `src/tessera/recomb/diagnostics.py` | modify | C5: cap the PHI window |
+| `src/tessera/cli/cmd_recomb.py` | modify | C5: `--phi-window` help text |
+| `src/tessera/recomb/regions.py` | modify | C2: reported span of an HMM region |
+| `src/tessera/recomb/clusters.py` | modify | C1: pairwise agreement on informative columns |
+| `validation/characterise_false_regions.py` | create | C8: one row per false region |
+| `docs/superpowers/specs/2026-10-01-min-methods-default-decision.md` | create | C8: decision record |
+| `docs/detection-methods.md`, `docs/reference-panels.md`, `validation/README.md`, `CHANGELOG.md` | modify | describe each change where it lands |
+
+## Review Focus
+
+Inputs the spec implies but its own examples do not exercise. Each has a test in the task that owns the code.
+
+1. **A query with gaps upstream of a region.** After the span is trimmed, `query_start` / `query_end` must still be the query coordinates of the reported columns. *Task 8, `test_trimmed_region_reports_query_coordinates_across_a_query_gap`.*
+2. **`--min-methods 2` after trimming.** Agreement is counted on overlapping regions, so a trimmed HMM region must still overlap the 3SEQ / MaxChi tract or a real event is dropped. *Task 8, `test_trimmed_hmm_region_still_merges_with_the_site_callers`.*
+3. **`--informative-sites` forced on a divergent panel that holds true duplicates.** Clustering then compares on informative columns too; duplicates must still pool. *Task 9, `test_duplicates_still_merge_when_informative_windowing_is_forced`.*
+4. **An alignment with exactly ten informative sites.** The capped PHI window is then one rank; the signal must still be computed. *Task 7, `test_signal_at_the_minimum_number_of_informative_sites`.*
+5. **A user's `--candidate-pool` whose FASTA headers end in a single word.** That word must not be taken for a clade. *Task 6, `test_select_from_still_mines_a_user_pool`.*
+
+---
+
+### Task 1: Branch and baseline
+
+Nothing is committed in this task. Its output is a directory of harness results that every later gate compares against. `validation/data/` is git-ignored, so the results live there.
+
+**Files:**
+- Create (untracked, git-ignored): `validation/data/audit-c/baseline/*.txt`
+
+**Interfaces:**
+- Produces: `validation/data/audit-c/baseline/` holding `spec_mm1.txt`, `spec_mm2.txt`, `validation.txt`, `validation_regions.tsv`, `benchmark.txt`, `hybrids.txt`. Later tasks write the same file names under `validation/data/audit-c//` and diff against these.
+
+- [ ] **Step 1: Branch**
+
+```bash
+git switch main && git pull --ff-only
+git switch -c fix-audit-caller-behaviour
+pip install -e ".[dev]"
+```
+
+- [ ] **Step 2: Confirm the starting point is green**
+
+Run: `ruff check src tests validation && pytest -m "not requires_binary" -q && mypy src`
+Expected: `All checks passed!`, `586 passed, 1 deselected`, `Success: no issues found in 79 source files`. If any of these differs, stop: the baseline is not the one this plan was written against.
+
+- [ ] **Step 3: Write the gate script**
+
+The same commands run after every task, so put them in one untracked script. Create `validation/data/audit-c/gate.sh`:
+
+```bash
+#!/bin/bash
+# gate.sh : run the offline-capable harnesses and save their output under
+# validation/data/audit-c//. Run from the repository root.
+set -u
+out="validation/data/audit-c/$1"
+mkdir -p "$out"
+export PATH="$PATH:$HOME/miniforge3/envs/recomfi-aln/bin" # appended, never prepended
+python validation/run_specificity.py --reps 10 --min-methods 2 > "$out/spec_mm2.txt" 2>&1
+python validation/run_specificity.py --reps 10 --min-methods 1 > "$out/spec_mm1.txt" 2>&1
+python validation/run_validation.py > "$out/validation.txt" 2>&1
+python - > "$out/validation_regions.tsv" <<'EOF'
+import csv
+from pathlib import Path
+for d in sorted(Path("validation/data").iterdir()):
+ f = d / "_run" / "recomb_out" / "recombination_regions.tsv"
+ if not f.exists() or d.name == "orthopox_example":
+ continue
+ for r in csv.DictReader(open(f), delimiter="\t"):
+ print(d.name, r["minor_parent"], r["major_parent"], r["query_start"], r["query_end"],
+ r.get("methods") or "-", r.get("donor_absent", ""), r.get("donor_undercovered", ""),
+ sep="\t")
+EOF
+python validation/run_benchmark.py > "$out/benchmark.txt" 2>&1
+grep -E "TOTAL|detected" "$out/spec_mm2.txt" "$out/spec_mm1.txt"
+tail -9 "$out/validation.txt"
+tail -2 "$out/benchmark.txt"
+```
+
+```bash
+mkdir -p validation/data/audit-c && chmod +x validation/data/audit-c/gate.sh
+```
+
+- [ ] **Step 4: Capture the baseline**
+
+Run: `validation/data/audit-c/gate.sh baseline` (about 15 minutes).
+
+Expected, at 10 replicates (the 3-replicate values measured for this plan are in the table above):
+- `spec_mm2.txt`: `TOTAL 0/40` or close to it; positive control `detected 10/10`.
+- `spec_mm1.txt`: roughly half the runs carry a false region, almost all from `hmm`.
+- `validation.txt`: `PASS` for `sarscov2_xbb`, `hiv1_crf`, `norovirus_gii`, `enterovirus_e11`, `hiv_crf02ag`, `hcv_2k1b`; `FAIL` for `hcv_clonal_1b` (a known open item: a 12 bp MaxChi-only region); `SKIP` for `orthopox_example`.
+- `benchmark.txt`: `power 0.04 (300 recombining) | specificity 1.00 (60 clonal) | 139 untestable`.
+
+What to do when something is absent:
+- **No aligner on `PATH`** (`mafft`, `minimap2`): `run_validation.py` prints `SKIP` with ` not on PATH` for every dataset. Create the env (`conda env create -f environment.yml`) and append its `bin` to `PATH`. Do not continue past Task 7 without it.
+- **Datasets not fetched**: `run_validation.py` prints `SKIP ... sequences not present`. Run `python validation/fetch.py` (needs `efetch` and network).
+- **No benchmark alignments**: `run_benchmark.py` prints `[SKIP] no benchmark alignments`. Download `performance.tar.gz` from Dryad (doi:10.5061/dryad.d7wm37q6f) in a browser and extract the `msa_*.fasta` files into `validation/data/benchmark/`. Without it, Task 7's gate cannot be evaluated; record that in the PR.
+
+- [ ] **Step 5: Capture the hybrids baseline**
+
+```bash
+export PATH="$PATH:$HOME/miniforge3/envs/recomfi-aln/bin"
+python validation/run_hybrids.py > validation/data/audit-c/baseline/hybrids.txt 2>&1
+grep -A60 "^case " validation/data/audit-c/baseline/hybrids.txt | tail -60
+```
+
+This needs network (Nextclade index, references and trees) and `mafft`, `skani`, `skDER`; allow 30-60 minutes. Expected: a table with one row per case and a final line `N/M passed (S skipped, E error)` with `E = 0`.
+
+If every row reads `ERROR: Could not fetch the Nextclade dataset index`, there is no network. Then the hybrids gate cannot be run for any task: say so in the PR description under "Skipped", and do not merge Tasks 6, 8 or 9 until someone has run it. `CONTRIBUTING.md` requires it for those three.
+
+- [ ] **Step 6: Record the baseline**
+
+Copy the summary lines of the five files into a scratch note; they open the PR description in Task 11. No commit.
+
+---
+
+### Task 2: Regime harness
+
+The audit found two defects with simulations no harness covers. This task commits those simulations so the later gates are reproducible.
+
+**Where it lives, and why both places.** `run_specificity.py` set the pattern: the simulation and scoring sit in an opt-in `validation/` script because rates need replicates and replicates take minutes, and CI unit-tests that script's pure logic by loading it by path. The same split is used here. The harness (`validation/run_regimes.py`) reports rates; the one-seed regressions that must hold on every commit go in the fast suite (`tests/integration/test_regimes.py`, added in Tasks 8 and 9) and import the simulators from the harness, so there is one copy of the simulation.
+
+**Files:**
+- Create: `validation/run_regimes.py`
+- Create: `tests/unit/test_regime_scoring.py`
+- Modify: `validation/README.md` (layout block, and a new section after the specificity section)
+
+**Interfaces:**
+- Produces, in `validation/run_regimes.py`:
+ - `QUERY = "QUERY"`, `NEAR_REFS = ("R0", "R1", "R2", "R3", "R4")`
+ - `simulate_near_identical(seed: int, *, recombinant: bool, length: int = 120_000, distance: float = 0.001, tract: tuple[int, int] = (40_000, 48_000)) -> tuple[dict[str, str], tuple[int, int] | None]`
+ - `simulate_sibling_tract(seed: int, *, length: int = 30_000, distance: float = 0.05, tract: tuple[int, int] = (12_000, 18_000), extra: int = 600) -> tuple[dict[str, str], tuple[int, int]]`
+ - `score_tract(rows, tract, *, donor) -> dict` with keys `detected`, `hmm`, `donor_ok`, `breakpoint_error`
+ - `score_clonal(rows) -> tuple[int, dict[str, int]]`
+ - `score_span(rows, span) -> dict` with keys `detected`, `reported_bp`, `excess_bp`
+ - `scan(seqs, logger, *, min_methods: int = 1, cluster_lineages: bool = True) -> list[dict]`
+ - `_evolve(seq: np.ndarray, distance: float, rng) -> np.ndarray`, `_as_text(seqs) -> dict[str, str]`
+- Tests load the module by path as `rg`, exactly as `tests/unit/test_specificity_scoring.py` loads `run_specificity.py`.
+
+- [ ] **Step 1: Write the failing tests**
+
+Create `tests/unit/test_regime_scoring.py`:
+
+```python
+"""Unit tests for the regime harness's simulation and scoring (no binaries, no network)."""
+
+from __future__ import annotations
+
+import importlib.util
+import sys
+from pathlib import Path
+
+_PATH = Path(__file__).resolve().parents[2] / "validation" / "run_regimes.py"
+_SPEC = importlib.util.spec_from_file_location("run_regimes", _PATH)
+rg = importlib.util.module_from_spec(_SPEC)
+sys.modules["run_regimes"] = rg
+_SPEC.loader.exec_module(rg)
+
+SMALL = {"length": 6000, "distance": 0.004, "tract": (2000, 3000)}
+
+
+def _identity(a: str, b: str, lo: int, hi: int) -> float:
+ return sum(1 for i in range(lo, hi) if a[i] == b[i]) / (hi - lo)
+
+
+# --- simulation ------------------------------------------------------------
+
+def test_near_identical_simulation_is_deterministic_and_aligned():
+ a, tract = rg.simulate_near_identical(3, recombinant=True, **SMALL)
+ b, _ = rg.simulate_near_identical(3, recombinant=True, **SMALL)
+ assert a == b and tract == SMALL["tract"]
+ assert rg.simulate_near_identical(4, recombinant=True, **SMALL)[0] != a
+ assert rg.QUERY in a and set(rg.NEAR_REFS) <= set(a)
+ assert len({len(s) for s in a.values()}) == 1
+
+
+def test_near_identical_panel_is_near_identical():
+ seqs, _ = rg.simulate_near_identical(1, recombinant=False, **SMALL)
+ # two references each 0.4 % from the root are ~0.8 % apart
+ assert 0.985 < _identity(seqs["R0"], seqs["R1"], 0, SMALL["length"]) < 0.998
+
+
+def test_near_identical_recombinant_carries_the_donor_over_the_tract_only():
+ seqs, tract = rg.simulate_near_identical(1, recombinant=True, **SMALL)
+ lo, hi = tract
+ query = seqs[rg.QUERY]
+ assert query[lo:hi] == seqs["R1"][lo:hi]
+ assert _identity(query, seqs["R0"], 0, lo) > _identity(query, seqs["R1"], 0, lo)
+
+
+def test_near_identical_clonal_has_no_tract():
+ seqs, tract = rg.simulate_near_identical(1, recombinant=False, **SMALL)
+ assert tract is None
+ assert _identity(seqs[rg.QUERY], seqs["R0"], 0, SMALL["length"]) > 0.998
+
+
+def test_sibling_tract_differs_from_the_query_only_over_the_span():
+ seqs, (lo, hi) = rg.simulate_sibling_tract(
+ 2, length=6000, distance=0.05, tract=(2000, 3000), extra=300
+ )
+ query, sibling = seqs[rg.QUERY], seqs["sibling"]
+ assert (lo, hi) == (3000, 3300)
+ assert query[:lo] == sibling[:lo] and query[hi:] == sibling[hi:]
+ assert query[lo:hi] != sibling[lo:hi]
+ assert query[lo:hi] == seqs["parent_A"][lo:hi]
+ assert sibling[hi:] == seqs["parent_A"][hi:] # identical downstream of the span
+
+
+# --- scoring ---------------------------------------------------------------
+
+def _row(**kw):
+ row = {"minor_parent": "R1", "query_start": "100", "query_end": "200",
+ "donor_absent": "no", "methods": "3seq,maxchi"}
+ row.update(kw)
+ return row
+
+
+def test_score_tract_reports_detection_donor_and_error():
+ got = rg.score_tract([_row(query_start="90", query_end="230")], (100, 200), donor="R1")
+ assert got == {"detected": True, "hmm": False, "donor_ok": True, "breakpoint_error": 20}
+
+
+def test_score_tract_notes_when_the_hmm_was_among_the_callers():
+ rows = [_row(methods="hmm,3seq")]
+ assert rg.score_tract(rows, (100, 200), donor="R1")["hmm"] is True
+
+
+def test_score_tract_miss_and_wrong_donor():
+ assert rg.score_tract([_row(query_start="500", query_end="600")], (100, 200),
+ donor="R1")["detected"] is False
+ assert rg.score_tract([_row(minor_parent="R3")], (100, 200),
+ donor="R1")["donor_ok"] is False
+
+
+def test_score_tract_ignores_donor_absent_rows():
+ rows = [_row(donor_absent="yes", methods="")]
+ assert rg.score_tract(rows, (100, 200), donor="R1")["detected"] is False
+
+
+def test_score_clonal_counts_present_regions_per_caller():
+ rows = [_row(methods="hmm"), _row(methods="hmm,3seq"), _row(donor_absent="yes")]
+ assert rg.score_clonal(rows) == (2, {"hmm": 2, "3seq": 1})
+ assert rg.score_clonal([]) == (0, {})
+
+
+def test_score_span_reports_the_excess_over_the_true_difference():
+ rows = [_row(query_start="17900", query_end="29600"),
+ _row(query_start="18007", query_end="18591")]
+ got = rg.score_span(rows, (18000, 18600))
+ assert got == {"detected": True, "reported_bp": 11700, "excess_bp": 11100}
+ tight = rg.score_span(rows[1:], (18000, 18600))
+ assert tight["excess_bp"] == 0
+ assert rg.score_span([], (18000, 18600))["detected"] is False
+```
+
+- [ ] **Step 2: Run them to see them fail**
+
+Run: `pytest tests/unit/test_regime_scoring.py -q`
+Expected: collection error, `FileNotFoundError: ... validation/run_regimes.py`.
+
+- [ ] **Step 3: Write the harness**
+
+Create `validation/run_regimes.py`:
+
+```python
+#!/usr/bin/env python
+"""Opt-in harness: two regimes the other harnesses do not sample.
+
+``run_specificity.py`` simulates clades ~16 % apart, so it always runs under base-pair
+windowing, and its positive control has parents that differ everywhere. Two situations
+found by the post-1.2.0 audit fall outside it:
+
+ near_identical a panel whose references are ~0.2 % apart (the mpox / VZV /
+ within-lineage SARS-CoV-2 regime, analysed under informative-site
+ windowing). Run twice: with a donor tract spliced into the query
+ (is it found, and does the HMM contribute?) and clonal (what is
+ reported when there is nothing to find?).
+ sibling_tract the panel holds a sibling recombinant whose donor tract runs 600 bp
+ further than the query's. Against that sibling the query differs
+ only over those 600 bp; downstream of them the donor and the
+ backbone are identical, so nothing marks where the region ends.
+
+Like ``run_specificity.py`` this needs no aligner, no network and no downloaded data:
+the simulated sequences are already aligned. It is opt-in because a full run is minutes
+of wall clock; the simulation and scoring logic is unit-tested in CI
+(``tests/unit/test_regime_scoring.py``).
+
+ python validation/run_regimes.py # 10 replicates, --min-methods 1
+ python validation/run_regimes.py --reps 3 # quick look
+ python validation/run_regimes.py --min-methods 2 # with the agreement gate
+ python validation/run_regimes.py --no-cluster-lineages
+
+``--min-methods`` defaults to 1 here because that is the CLI default; the harness
+reports what a user gets.
+
+Caveat: a star tree under JC69 is simpler than real viral evolution, and a handful of
+replicates carries real sampling error. These numbers show whether a failure mode
+exists and roughly how large it is, not its magnitude on real panels.
+"""
+
+from __future__ import annotations
+
+import logging
+import sys
+import tempfile
+from collections import Counter
+from pathlib import Path
+
+import numpy as np
+
+from tessera.recomb.run import RecombParams, run_recomb
+
+QUERY = "QUERY"
+_BASES = np.frombuffer(b"ACGT", dtype=np.uint8)
+
+# near_identical: five references on a star tree, each this far from the root.
+NEAR_LENGTH = 120_000
+NEAR_DISTANCE = 0.001
+NEAR_REFS = ("R0", "R1", "R2", "R3", "R4")
+NEAR_TRACT = (40_000, 48_000) # donor tract (from R1) in the R0-derived query
+
+# sibling_tract: parents 10 % apart; the sibling's tract runs 600 bp past the query's.
+SIB_LENGTH = 30_000
+SIB_DISTANCE = 0.05
+SIB_QUERY_TRACT = (12_000, 18_000)
+SIB_EXTRA = 600
+
+
+# --- simulation ------------------------------------------------------------------
+
+def _evolve(seq: np.ndarray, distance: float, rng) -> np.ndarray:
+ """JC69: mutate each site with probability 3/4(1 - exp(-4/3 * d))."""
+ p = 0.75 * (1.0 - np.exp(-4.0 / 3.0 * distance))
+ hit = rng.random(seq.size) < p
+ out = seq.copy()
+ n = int(hit.sum())
+ if n: # a mutation always changes the base
+ out[hit] = (seq[hit] + rng.integers(1, 4, size=n)) % 4
+ return out
+
+
+def _as_text(seqs: dict[str, np.ndarray]) -> dict[str, str]:
+ return {k: _BASES[v].tobytes().decode("ascii") for k, v in seqs.items()}
+
+
+def simulate_near_identical(
+ seed: int, *, recombinant: bool, length: int = NEAR_LENGTH,
+ distance: float = NEAR_DISTANCE, tract: tuple[int, int] = NEAR_TRACT,
+) -> tuple[dict[str, str], tuple[int, int] | None]:
+ """A near-identical panel and a query descended from ``R0``.
+
+ With ``recombinant`` the query carries ``R1`` over ``tract``; otherwise it is a
+ plain descendant of ``R0`` and nothing is recombined. Returns ``(sequences, tract)``
+ with ``tract`` ``None`` for the clonal case.
+ """
+ rng = np.random.default_rng(seed)
+ root = rng.integers(0, 4, size=length)
+ seqs = {label: _evolve(root, distance, rng) for label in NEAR_REFS}
+ query = _evolve(seqs["R0"], distance / 10.0, rng)
+ if recombinant:
+ lo, hi = tract
+ query[lo:hi] = seqs["R1"][lo:hi]
+ seqs[QUERY] = query
+ return _as_text(seqs), (tract if recombinant else None)
+
+
+def simulate_sibling_tract(
+ seed: int, *, length: int = SIB_LENGTH, distance: float = SIB_DISTANCE,
+ tract: tuple[int, int] = SIB_QUERY_TRACT, extra: int = SIB_EXTRA,
+) -> tuple[dict[str, str], tuple[int, int]]:
+ """A query and a sibling recombinant whose donor tract runs ``extra`` bp further.
+
+ Both are a ``parent_A`` backbone carrying ``parent_B``; the query over ``tract``, the
+ sibling over ``tract`` extended by ``extra``. Returns ``(sequences, span)`` where
+ ``span`` is the only stretch on which the query and the sibling differ -- there the
+ query matches ``parent_A``. Everywhere downstream ``parent_A`` and the sibling are
+ identical.
+ """
+ rng = np.random.default_rng(seed)
+ root = rng.integers(0, 4, size=length)
+ parent_a, parent_b, other = (_evolve(root, distance, rng) for _ in range(3))
+ lo, hi = tract
+ query = parent_a.copy()
+ query[lo:hi] = parent_b[lo:hi]
+ sibling = parent_a.copy()
+ sibling[lo:hi + extra] = parent_b[lo:hi + extra]
+ seqs = {QUERY: query, "sibling": sibling, "parent_A": parent_a,
+ "parent_B": parent_b, "other": other}
+ return _as_text(seqs), (hi, hi + extra)
+
+
+# --- scoring ---------------------------------------------------------------------
+
+def _present(rows: list[dict]) -> list[dict]:
+ """Rows that claim a donor; a ``donor_absent`` row is a coverage statement."""
+ return [r for r in rows if r.get("donor_absent") != "yes"]
+
+
+def _methods(row: dict) -> list[str]:
+ return [m.strip() for m in (row.get("methods") or "").split(",") if m.strip()]
+
+
+def score_tract(rows: list[dict], tract: tuple[int, int], *, donor: str) -> dict:
+ """Detection of a known tract, whether the HMM was among the callers, donor
+ attribution and breakpoint error."""
+ lo, hi = tract
+ overlapping = [
+ r for r in _present(rows)
+ if int(r["query_start"]) < hi and int(r["query_end"]) > lo
+ ]
+ if not overlapping:
+ return {"detected": False, "hmm": False, "donor_ok": False,
+ "breakpoint_error": None}
+ best = max(
+ overlapping,
+ key=lambda r: min(int(r["query_end"]), hi) - max(int(r["query_start"]), lo),
+ )
+ return {
+ "detected": True,
+ "hmm": any("hmm" in _methods(r) for r in overlapping),
+ "donor_ok": best["minor_parent"] == donor,
+ "breakpoint_error": (
+ abs(int(best["query_start"]) - lo) + abs(int(best["query_end"]) - hi)
+ ) // 2,
+ }
+
+
+def score_clonal(rows: list[dict]) -> tuple[int, dict[str, int]]:
+ """``(false regions, per-caller counts)`` for a run that must report nothing."""
+ present = _present(rows)
+ per_caller: Counter[str] = Counter()
+ for row in present:
+ per_caller.update(_methods(row))
+ return len(present), dict(per_caller)
+
+
+def score_span(rows: list[dict], span: tuple[int, int]) -> dict:
+ """How much longer than the true difference the reported region is.
+
+ ``reported_bp`` is the length of the longest reported region overlapping ``span``
+ (``None`` when none does); ``excess_bp`` is that length minus the span's.
+ """
+ lo, hi = span
+ overlapping = [
+ r for r in _present(rows)
+ if int(r["query_start"]) < hi and int(r["query_end"]) > lo
+ ]
+ if not overlapping:
+ return {"detected": False, "reported_bp": None, "excess_bp": None}
+ reported = max(int(r["query_end"]) - int(r["query_start"]) for r in overlapping)
+ return {"detected": True, "reported_bp": reported,
+ "excess_bp": max(0, reported - (hi - lo))}
+
+
+# --- running ---------------------------------------------------------------------
+
+def _write_fasta(path: Path, seqs: dict[str, str]) -> None:
+ with path.open("w") as fh:
+ for label, seq in seqs.items():
+ fh.write(f">{label}\n")
+ for i in range(0, len(seq), 70):
+ fh.write(seq[i : i + 70] + "\n")
+
+
+def _read_regions(path: Path) -> list[dict]:
+ if not path.exists():
+ return []
+ lines = [x for x in path.read_text().splitlines() if x.strip()]
+ if len(lines) < 2:
+ return []
+ header = lines[0].split("\t")
+ return [dict(zip(header, x.split("\t"), strict=False)) for x in lines[1:]]
+
+
+def scan(
+ seqs: dict[str, str], logger: logging.Logger, *, min_methods: int = 1,
+ cluster_lineages: bool = True,
+) -> list[dict]:
+ """Run the shipped pipeline on one simulated alignment; return its region rows."""
+ with tempfile.TemporaryDirectory() as td:
+ msa = Path(td) / "aln.fasta"
+ _write_fasta(msa, seqs)
+ out = Path(td) / "out"
+ run_recomb(
+ RecombParams(msa=msa, output=out, query=QUERY, plot_format="png",
+ min_methods=min_methods, cluster_lineages=cluster_lineages),
+ logger,
+ )
+ return _read_regions(out / "recombination_regions.tsv")
+
+
+def _median(values: list[int]) -> str:
+ return f"{int(np.median(values))} bp" if values else "n/a"
+
+
+def main(argv: list[str]) -> int:
+ reps, min_methods = 10, 1
+ if "--reps" in argv:
+ reps = int(argv[argv.index("--reps") + 1])
+ if "--min-methods" in argv:
+ min_methods = int(argv[argv.index("--min-methods") + 1])
+ cluster = "--no-cluster-lineages" not in argv
+
+ logger = logging.getLogger("tessera.regimes")
+ logger.addHandler(logging.NullHandler())
+ logger.propagate = False # the scan is chatty; the table below is the output
+
+ def run(seqs: dict[str, str]) -> list[dict]:
+ return scan(seqs, logger, min_methods=min_methods, cluster_lineages=cluster)
+
+ print(f"Tessera regime harness -- {reps} replicate(s), --min-methods {min_methods}, "
+ f"lineage clustering {'on' if cluster else 'off'}\n")
+
+ detected = hmm = donor_ok = 0
+ errors: list[int] = []
+ for rep in range(reps):
+ seqs, tract = simulate_near_identical(3000 + rep, recombinant=True)
+ assert tract is not None
+ got = score_tract(run(seqs), tract, donor="R1")
+ detected += got["detected"]
+ hmm += got["hmm"]
+ donor_ok += got["donor_ok"]
+ if got["breakpoint_error"] is not None:
+ errors.append(got["breakpoint_error"])
+ print(f"near_identical, tract {NEAR_TRACT[0]}-{NEAR_TRACT[1]} from R1:")
+ print(f" detected {detected}/{reps} | called by the HMM {hmm}/{reps} | "
+ f"correct donor {donor_ok}/{reps} | median breakpoint error {_median(errors)}")
+
+ bad = regions = 0
+ callers: Counter[str] = Counter()
+ for rep in range(reps):
+ seqs, _ = simulate_near_identical(4000 + rep, recombinant=False)
+ n_false, per_caller = score_clonal(run(seqs))
+ bad += bool(n_false)
+ regions += n_false
+ callers.update(per_caller)
+ print("near_identical, clonal (every region is a false positive):")
+ print(f" runs with a false region {bad}/{reps} | false regions {regions} | "
+ + (", ".join(f"{k}={v}" for k, v in sorted(callers.items())) or "--"))
+
+ found = 0
+ reported: list[int] = []
+ excess: list[int] = []
+ for rep in range(reps):
+ seqs, span = simulate_sibling_tract(5000 + rep)
+ got = score_span(run(seqs), span)
+ found += got["detected"]
+ if got["reported_bp"] is not None:
+ reported.append(got["reported_bp"])
+ excess.append(got["excess_bp"])
+ print(f"sibling_tract, true difference {SIB_EXTRA} bp:")
+ print(f" detected {found}/{reps} | median reported length {_median(reported)} | "
+ f"median excess {_median(excess)}")
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main(sys.argv[1:]))
+```
+
+- [ ] **Step 4: Run the tests**
+
+Run: `pytest tests/unit/test_regime_scoring.py -q && ruff check validation tests`
+Expected: `11 passed`, `All checks passed!`
+
+- [ ] **Step 5: Run the harness on the unchanged callers**
+
+```bash
+python validation/run_regimes.py --reps 10 | tee validation/data/audit-c/baseline/regimes_mm1.txt
+python validation/run_regimes.py --reps 10 --min-methods 2 | tee validation/data/audit-c/baseline/regimes_mm2.txt
+```
+
+Expected (measured for this plan, both gates):
+
+```
+near_identical, tract 40000-48000 from R1:
+ detected 10/10 | called by the HMM 0/10 | correct donor 10/10 | median breakpoint error 465 bp
+near_identical, clonal (every region is a false positive):
+ runs with a false region 0/10 | false regions 0 | --
+sibling_tract, true difference 600 bp:
+ detected 10/10 | median reported length 11700 bp | median excess 11100 bp
+```
+
+`called by the HMM 0/10` is defect C1 and `median reported length 11700 bp` is defect C2. If these two lines do not reproduce, stop and report: the defects this plan fixes are not present on your checkout.
+
+- [ ] **Step 6: Document the harness**
+
+In `validation/README.md`, in the layout block, add after the `run_specificity.py` line:
+
+```
+ run_regimes.py near-identical panels and sibling tracts, simulated (no aligner needed)
+```
+
+Then add this section immediately before the `## Prerequisites` heading:
+
+````markdown
+## Regimes the other harnesses do not sample
+
+`run_regimes.py` simulates two situations that fall outside `run_specificity.py`, whose
+clades are about 16 % apart and so always run under base-pair windowing:
+
+| regime | what it asks |
+|---|---|
+| `near_identical` with a tract | references about 0.2 % apart (the mpox / VZV regime, analysed under informative-site windowing): is the tract found, and is the HMM among the callers? |
+| `near_identical` clonal | the same panel with nothing recombined: every region is a false positive |
+| `sibling_tract` | the panel holds a sibling recombinant whose tract runs 600 bp past the query's; downstream of those 600 bp the donor and the backbone are identical, so nothing marks where the region ends |
+
+It defaults to `--min-methods 1`, the CLI default, so it reports what a user gets; pass
+`--min-methods 2` for the agreement gate and `--no-cluster-lineages` to switch lineage
+clustering off. Like the specificity harness it needs no aligner, network or data.
+
+```
+python validation/run_regimes.py --reps 3 # quick look
+```
+
+The same caveat applies: a star tree under JC69 and a handful of replicates show whether a
+failure mode exists, not how large it is on real panels.
+````
+
+- [ ] **Step 7: Commit**
+
+```bash
+git add validation/run_regimes.py tests/unit/test_regime_scoring.py validation/README.md
+git commit -m "$(cat <<'EOF'
+Add a harness for near-identical panels and sibling tracts
+
+The post-1.2.0 audit found two defects with simulations that no harness
+covers: on a panel ~0.2 % divergent the HMM never contributes a call, and
+when a sibling recombinant is in the panel an HMM region runs on through
+columns where donor and backbone are identical. run_specificity.py cannot
+see either -- its clades are ~16 % apart, so it always runs under base-pair
+windowing, and its positive control's parents differ everywhere.
+
+run_regimes.py simulates both, plus a clonal near-identical panel so that
+any fix which gives the HMM a vote there is measured against its
+false-positive cost. It follows run_specificity.py: self-contained, opt-in,
+with the simulation and scoring unit-tested in CI.
+
+On main: HMM among the callers 0/10; sibling tract of 600 bp reported as
+11700 bp (median, 10 replicates).
+
+Co-Authored-By: Claude Fable 5.1
+EOF
+)"
+```
+
+---
+
+### Task 3: 3SEQ and MaxChi tract start on ties (C3)
+
+`threeseq.max_descent` finds the trough of the walk and then takes the **first** maximum before it. If the walk reaches its maximum, dips, returns to the same height and only then descends, the reported tract starts at the first maximum and includes the balanced excursion. MaxChi scores the same tract, and its permutation null (`maxchi._exceedances`) applies the same rule to every permuted walk, so the two must change together or observed and null statistics stop being comparable.
+
+**Files:**
+- Modify: `src/tessera/recomb/threeseq.py` (`max_descent`, the `peak = ...` line)
+- Modify: `src/tessera/recomb/maxchi.py` (`_exceedances`, the `peak = ...` line)
+- Test: `tests/unit/test_threeseq.py`, `tests/unit/test_maxchi.py`
+
+**Interfaces:**
+- Consumes: nothing from earlier tasks.
+- Produces: `max_descent(steps).start_site` is the last index at which the walk is at its maximum before the trough. Signature unchanged. Task 8 calls `triplet_steps` from this module; it does not depend on this change.
+
+- [ ] **Step 1: Write the failing test**
+
+Append to `tests/unit/test_threeseq.py`:
+
+```python
+
+
+def test_max_descent_starts_at_the_last_maximum_before_the_trough() -> None:
+ # The walk reaches its maximum (2) at site 2, dips, returns to 2 at site 4 and only
+ # then descends. The donor run is sites [4, 8); sites 2-3 are a balanced excursion.
+ steps = np.array([1, 1, -1, 1, -1, -1, -1, -1])
+ d = max_descent(steps)
+ assert d.depth == 4
+ assert (d.start_site, d.end_site) == (4, 8)
+ assert set(steps[d.start_site:d.end_site]) == {-1}
+```
+
+- [ ] **Step 2: Add the consistency guard for MaxChi**
+
+In `tests/unit/test_maxchi.py`, change the import block to include `_exceedances`:
+
+```python
+from tessera.recomb.maxchi import (
+ _chi2_2x2,
+ _exceedances,
+ maxchi_pvalue,
+ maxchi_statistic,
+)
+```
+
+and append:
+
+```python
+
+
+def test_permutation_null_uses_the_same_tract_rule_as_the_observed_walk() -> None:
+ """The vectorised null must score each permuted walk on the tract ``max_descent``
+ would report for it; otherwise observed and null statistics are not comparable.
+
+ A consistency guard: it holds before and after the tie rule changes, and fails if
+ only one of ``threeseq.max_descent`` / ``maxchi._exceedances`` is changed.
+ """
+ steps = np.array([1] * 30 + [-1] * 14)
+ np.random.default_rng(5).shuffle(steps)
+ k, seed = 400, 11
+ order = np.argsort(np.random.default_rng(seed).random((k, steps.size)), axis=1)
+ stats = []
+ for perm in steps[order]:
+ d = max_descent(perm)
+ stats.append(maxchi_statistic(perm, d.start_site, d.end_site))
+ observed = float(np.median(stats))
+ expected = sum(s >= observed for s in stats)
+ assert 0 < expected < k # the threshold splits the permutations
+ assert _exceedances(steps, observed, k, np.random.default_rng(seed)) == expected
+```
+
+- [ ] **Step 3: Run the tests**
+
+Run: `pytest tests/unit/test_threeseq.py tests/unit/test_maxchi.py -q`
+Expected: 1 failed -- `test_max_descent_starts_at_the_last_maximum_before_the_trough` with `assert (2, 8) == (4, 8)`. The MaxChi guard passes (both sides still use the first maximum).
+
+- [ ] **Step 4: Change the rule in `max_descent`**
+
+In `src/tessera/recomb/threeseq.py`, replace
+
+```python
+ peak = int(np.argmax(cumulative[: trough + 1])) if trough > 0 else 0
+```
+
+with
+
+```python
+ # The tract starts at the LAST maximum before the trough. The first one (plain
+ # argmax) would pull in any balanced excursion that returns to the same height,
+ # reporting the tract as starting earlier than the donor run does.
+ peak = trough - int(np.argmax(cumulative[trough::-1])) if trough > 0 else 0
+```
+
+- [ ] **Step 5: Run the tests again**
+
+Run: `pytest tests/unit/test_threeseq.py tests/unit/test_maxchi.py -q`
+Expected: 1 failed -- now the MaxChi guard, because the null still uses the first maximum. This is the guard doing its job.
+
+- [ ] **Step 6: Change the rule in the MaxChi null**
+
+In `src/tessera/recomb/maxchi.py`, in `_exceedances`, replace
+
+```python
+ # peak = where the running max was reached, on or before the trough
+ masked = np.where(cols[None, :] <= trough[:, None], cum, np.iinfo(np.int64).min)
+ peak = np.argmax(masked, axis=1)
+```
+
+with
+
+```python
+ # peak = the LAST place the running max was reached, on or before the trough --
+ # the same rule threeseq.max_descent applies to the observed walk, so the null
+ # and the observed statistic are computed on the same kind of tract.
+ masked = np.where(cols[None, :] <= trough[:, None], cum, np.iinfo(np.int64).min)
+ peak = length - np.argmax(masked[:, ::-1], axis=1)
+```
+
+- [ ] **Step 7: Run the tests and the fast suite**
+
+Run: `pytest tests/unit/test_threeseq.py tests/unit/test_maxchi.py -q && pytest -m "not requires_binary" -q && ruff check src tests`
+Expected: all pass.
+
+- [ ] **Step 8: Harness gate**
+
+Run: `validation/data/audit-c/gate.sh task3` and, if Task 1 Step 5 produced a hybrids baseline, `python validation/run_hybrids.py > validation/data/audit-c/task3/hybrids.txt 2>&1`.
+
+Acceptance (all must hold):
+- `spec_mm2.txt` and `spec_mm1.txt`: the `TOTAL` line (runs with a false region, false regions, per-caller counts) is identical to the baseline.
+- Positive control: `detected` and `correct donor` identical to the baseline; median breakpoint error not higher than the baseline. (Measured at 3 replicates: 55 bp -> 50 bp.)
+- `validation.txt`: the same PASS / FAIL / SKIP per dataset as the baseline.
+- `diff validation/data/audit-c/baseline/validation_regions.tsv validation/data/audit-c/task3/validation_regions.tsv`: the same rows; only 3SEQ / MaxChi start coordinates may move, and only later. (Measured: one row, `hiv1_crf` donor `C`, 8433 -> 8439.)
+- Hybrids: no case changes from PASS to FAIL.
+
+The 3SEQ p-value depends only on the depth of the descent, which this does not change. The MaxChi p-value can change, because the tract's chi-square is computed on a different interval; that is why the false-region counts are part of the gate.
+
+If the gate fails: `git checkout -- src/tessera/recomb/threeseq.py src/tessera/recomb/maxchi.py`, keep the two tests out of the commit, record the numbers, go to Task 4.
+
+- [ ] **Step 9: Commit**
+
+```bash
+git add src/tessera/recomb/threeseq.py src/tessera/recomb/maxchi.py tests/unit/test_threeseq.py tests/unit/test_maxchi.py
+git commit -m "$(cat <<'EOF'
+Start a 3SEQ/MaxChi tract at the last maximum before the trough
+
+max_descent took the first maximum of the walk before the trough. When the
+walk reaches its maximum, dips, and returns to the same height before
+descending, that reports the tract from the earlier point and includes a
+balanced stretch that is not donor tract: for steps +1 +1 -1 +1 -1 -1 -1 -1
+it gave sites 2-8 (support 5/6) where the donor run is 4-8 (4/4).
+
+MaxChi scores the same tract and its permutation null applied the same rule
+to each permuted walk, so both change together; a guard test now fails if
+only one of them does.
+
+The 3SEQ p-value is a function of the descent depth and is unchanged.
+Harness: .
+
+Co-Authored-By: Claude Fable 5.1
+EOF
+)"
+```
+
+Replace the `` line with the measured lines before committing.
+
+---
+
+### Task 4: Exact 3SEQ p-values keep their magnitude (C4)
+
+`descent_pvalue_exact` propagates the probability that the drawdown never reaches the depth and returns one minus it. Below about 1e-16 that subtraction is exactly `0.0`, which is what `recombination_regions.tsv` then holds as `pvalue` and `qvalue` for the strongest regions. Propagating the complementary probability directly keeps every term a sum of non-negative products.
+
+**Files:**
+- Modify: `src/tessera/recomb/threeseq.py` (`descent_pvalue_exact`, whole function)
+- Test: `tests/unit/test_threeseq.py`
+
+**Interfaces:**
+- Consumes: nothing from earlier tasks.
+- Produces: `descent_pvalue_exact(m: int, n: int, depth: int) -> float`, same signature and same values to within floating-point error, except that values below ~1e-16 are no longer `0.0`. Still returns exactly `0.0` when `n < depth` (the event is impossible) and `1.0` when `depth <= 0`.
+
+- [ ] **Step 1: Write the failing test**
+
+In `tests/unit/test_threeseq.py`, add `import math` after `import itertools`, and append:
+
+```python
+
+
+def test_descent_pvalue_exact_keeps_small_values() -> None:
+ # depth == n: every down-step must be consecutive, so the count of favourable
+ # arrangements is the m + 1 positions of that block among the up-steps.
+ for m, n in ((200, 100), (50, 30), (12, 9)):
+ expected = (m + 1) / math.comb(m + n, n)
+ got = descent_pvalue_exact(m, n, n)
+ assert got > 0.0
+ assert math.isclose(got, expected, rel_tol=1e-9)
+```
+
+- [ ] **Step 2: Run it to see it fail**
+
+Run: `pytest tests/unit/test_threeseq.py::test_descent_pvalue_exact_keeps_small_values -q`
+Expected: FAIL with `assert 0.0 > 0.0` (the true value for `(200, 100)` is about 4.8e-80).
+
+- [ ] **Step 3: Rewrite the recursion**
+
+In `src/tessera/recomb/threeseq.py`, replace the whole of `descent_pvalue_exact` with:
+
+```python
+def descent_pvalue_exact(m: int, n: int, depth: int) -> float:
+ """Exact ``P(max drawdown >= depth)`` over the uniform arrangements of ``m`` +1 and
+ ``n`` -1 steps.
+
+ Dynamic program over the walk's current drawdown ``delta`` in ``[0, depth)``: an
+ up-step moves ``delta -> max(0, delta - 1)``, a down-step ``delta -> delta + 1``.
+ ``g[delta]`` is the probability that the drawdown *reaches* ``depth`` from that
+ state; a down-step from ``delta == depth - 1`` reaches it with probability 1.
+
+ The reaching probability is propagated directly rather than as one minus the
+ probability of never reaching it: every term is a sum of non-negative products, so
+ a very small p-value keeps its magnitude instead of cancelling to exactly ``0.0``
+ below about 1e-16. Probabilities (not path counts) are propagated, so there is no
+ big-integer overflow.
+ """
+ if depth <= 0:
+ return 1.0
+ if n < depth: # need at least ``depth`` consecutive down-steps to reach it
+ return 0.0
+
+ def after_down(row_below: np.ndarray) -> np.ndarray:
+ down = np.empty(depth)
+ down[: depth - 1] = row_below[1:depth] # delta -> delta + 1
+ down[depth - 1] = 1.0 # ... and from depth-1 the walk reaches ``depth``
+ return down
+
+ # prev[j] is the length-``depth`` vector g(i-1, j, .). Build row i from row i-1.
+ # i == 0 row: only down-steps remain, a single forced order; nothing left -> 0.
+ prev = [np.zeros(depth)]
+ for j in range(1, n + 1):
+ prev.append(after_down(prev[j - 1]))
+ for i in range(1, m + 1):
+ cur: list[np.ndarray] = [np.zeros(depth)] # j == 0: only up-steps remain
+ for j in range(1, n + 1):
+ total = i + j
+ up = np.empty(depth)
+ up[0] = prev[j][0]
+ up[1:] = prev[j][: depth - 1] # delta -> max(0, delta - 1)
+ cur.append((i / total) * up + (j / total) * after_down(cur[j - 1]))
+ prev = cur
+ return float(prev[n][0])
+```
+
+- [ ] **Step 4: Run the 3SEQ tests**
+
+Run: `pytest tests/unit/test_threeseq.py -q`
+Expected: all pass, including `test_descent_pvalue_exact_matches_brute_force` (every `m, n < 7` against enumeration) and `test_descent_pvalue_edges`.
+
+- [ ] **Step 5: Confirm the region table no longer holds a zero**
+
+```bash
+tessera recomb --msa example_data/divergent.msa.fasta --query query --output /tmp/tessera_c4 \
+ --window-size 300 --window-step 30 >/dev/null 2>&1
+cut -f14,15,16 /tmp/tessera_c4/recombination_regions.tsv
+```
+
+Expected: a `pvalue` and `qvalue` that are small and non-zero (before this task both read `0.0`), with `test` = `3SEQ max-descent (exact)`.
+
+- [ ] **Step 6: Harness gate**
+
+Run: `validation/data/audit-c/gate.sh task4`
+
+Acceptance: every line of the gate identical to `task3` (or to `baseline`, if Task 3 was reverted). The value of an exact p-value changes only below 1e-16; no comparison against `--alpha` can flip. `diff validation/data/audit-c/task3/validation_regions.tsv validation/data/audit-c/task4/validation_regions.tsv` must be empty.
+
+If it is not empty, something other than underflow changed: revert and record.
+
+- [ ] **Step 7: Commit**
+
+```bash
+git add src/tessera/recomb/threeseq.py tests/unit/test_threeseq.py
+git commit -m "$(cat <<'EOF'
+Keep the magnitude of very small exact 3SEQ p-values
+
+descent_pvalue_exact propagated the probability of never reaching the
+descent depth and returned one minus it. Below about 1e-16 that subtraction
+is exactly 0.0, so the strongest regions were written to
+recombination_regions.tsv with pvalue = 0.0 and qvalue = 0.0 -- a value no
+permutation or exact test can produce.
+
+The recursion now propagates the probability of reaching the depth. Every
+term is a sum of non-negative products, so nothing cancels: m=200, n=100,
+depth=100 gives 4.8e-80, equal to the closed form 201 / C(300, 100). Values
+above 1e-16 are unchanged (the brute-force test over all m, n < 7 still
+passes), so no call can change.
+
+Co-Authored-By: Claude Fable 5.1
+EOF
+)"
+```
+
+---
+
+### Task 5: Ensemble merge edge cases (C7)
+
+Two defects in `recomb/ensemble.py`, both shown at unit level only; neither was reached end to end in the audit, so this task has unit tests and a regression gate, not a detection gate.
+
+1. `_merge_group` relabels every region's major parent to the canonical backbone. When the callers disagree on the backbone, a site caller's region can name the canonical backbone as its *donor*, and after relabelling the row reads "H donated into H".
+2. `_group` sorts by `(minor_parent, query_start)` and puts each region into the first group it overlaps. On a typed panel two labels of one lineage arrive out of coordinate order, so a region can overlap two existing groups; it joins one, and the same donor lineage is reported as two overlapping regions.
+
+**Files:**
+- Modify: `src/tessera/recomb/ensemble.py` (`_merge_group`, `_group`)
+- Test: `tests/unit/test_ensemble.py`
+
+**Interfaces:**
+- Consumes: nothing from earlier tasks.
+- Produces: `consensus_regions(...)` unchanged in signature. A merged `Region` never has `minor_parent == major_parent`; regions of one donor (genome, or lineage when typed) that overlap transitively are one region.
+
+- [ ] **Step 1: Write the failing tests**
+
+Append to `tests/unit/test_ensemble.py`:
+
+```python
+
+
+def test_region_is_not_relabelled_as_a_donor_into_itself() -> None:
+ # The callers disagree on the backbone: the HMM reports H (it set the sibling S
+ # aside), 3SEQ kept S and found an H tract in it. H is the canonical backbone, so
+ # relabelling the 3SEQ region's major to H would read "H donated into H".
+ hmm = mk("X", 100, 200, "hmm", major="H", qvalue=1e-6)
+ seq = mk("H", 500, 600, "3seq", major="S", qvalue=1e-9)
+ merged, _ = consensus_regions({"hmm": [hmm], "3seq": [seq]}, major="H")
+ assert all(r.minor_parent != r.major_parent for r in merged)
+ by_minor = {r.minor_parent: r.major_parent for r in merged}
+ assert by_minor == {"X": "H", "H": "S"} # the tested pair is kept for the odd one out
+
+
+def test_same_lineage_regions_merge_transitively_across_labels() -> None:
+ # A1 and A2 are one lineage. A1 has two separate regions; the A2 region overlaps
+ # both. All three are one event; sorting by label must not leave two groups.
+ lmap = {"A1": "L", "A2": "L"}
+ first = mk("A1", 0, 10, "hmm", qvalue=1e-6)
+ second = mk("A1", 20, 30, "hmm", qvalue=1e-6)
+ bridge = mk("A2", 5, 25, "3seq", qvalue=1e-9)
+ merged, breakdown = consensus_regions(
+ {"hmm": [first, second], "3seq": [bridge]}, major="M", lineage_map=lmap
+ )
+ assert len(merged) == 1
+ assert (merged[0].query_start, merged[0].query_end) == (0, 30)
+ assert merged[0].methods == ("hmm", "3seq")
+ assert len(breakdown) == 1
+```
+
+- [ ] **Step 2: Run them to see them fail**
+
+Run: `pytest tests/unit/test_ensemble.py -q`
+Expected: 2 failed. The first fails on `assert all(r.minor_parent != r.major_parent ...)` (the 3SEQ region is relabelled `H` -> `H`); the second on `assert 2 == 1`.
+
+- [ ] **Step 3: Keep the tested pair when the canonical backbone is the donor**
+
+In `src/tessera/recomb/ensemble.py`, in `_merge_group`, replace
+
+```python
+ return Region(
+ minor_parent=best.minor_parent,
+ major_parent=major or best.major_parent,
+```
+
+with
+
+```python
+ # The canonical backbone replaces each caller's own, so the table reads against one
+ # backbone -- except when it IS this region's donor. That happens when the callers
+ # disagree on the backbone (the HMM set a sibling aside, a site caller did not): the
+ # region was called against the other backbone, and relabelling it would report a
+ # genome as a donor into itself. Keep the pair the caller actually tested.
+ relabel = bool(major) and major != best.minor_parent
+ return Region(
+ minor_parent=best.minor_parent,
+ major_parent=major if relabel and major else best.major_parent,
+```
+
+- [ ] **Step 4: Fold every group a region touches into one**
+
+In the same file, in `_group`, replace
+
+```python
+ for region in sorted(regions, key=lambda r: (r.minor_parent, r.query_start)):
+ placed = False
+ for group in groups:
+ if _same_donor(group[0], region, lineage_map) and any(
+ _overlap(region, member) for member in group
+ ):
+ group.append(region)
+ placed = True
+ break
+ if not placed:
+ groups.append([region])
+ return groups
+```
+
+with
+
+```python
+ for region in sorted(regions, key=lambda r: (r.minor_parent, r.query_start)):
+ hits = [
+ group for group in groups
+ if _same_donor(group[0], region, lineage_map)
+ and any(_overlap(region, member) for member in group)
+ ]
+ if not hits:
+ groups.append([region])
+ continue
+ # A region can bridge groups that did not overlap each other -- on a typed panel
+ # the sort is by genome label, so two labels of one lineage arrive out of
+ # coordinate order. Fold every group it touches into the first, or the same
+ # donor would be reported as two overlapping regions.
+ target = hits[0]
+ target.append(region)
+ for other in hits[1:]:
+ target.extend(other)
+ groups = [g for g in groups if all(g is not other for other in hits[1:])]
+ return groups
+```
+
+The groups are compared by identity (`is not`), not equality: two groups can hold equal regions.
+
+- [ ] **Step 5: Run the tests**
+
+Run: `pytest tests/unit/test_ensemble.py -q && pytest -m "not requires_binary" -q && ruff check src tests && mypy src`
+Expected: all pass, no new mypy error.
+
+- [ ] **Step 6: Regression gate**
+
+Run: `validation/data/audit-c/gate.sh task5`
+
+Acceptance: identical to the previous task's output in every file, including an empty `diff` of `validation_regions.tsv`. Neither edge case occurs in these harnesses (untyped panels; callers agree on the backbone), so any difference is a regression in the ordinary path: revert and record.
+
+- [ ] **Step 7: Commit**
+
+```bash
+git add src/tessera/recomb/ensemble.py tests/unit/test_ensemble.py
+git commit -m "$(cat <<'EOF'
+Close two edge cases in the ensemble merge
+
+The merge relabels every region's major parent to the canonical backbone.
+When the callers disagree on the backbone -- the HMM set a sibling aside, a
+site caller kept it -- a site caller's region can name the canonical
+backbone as its donor, and the relabelled row then read as a genome donating
+into itself. Such a region now keeps the pair its caller tested.
+
+Grouping put each region into the first group it overlapped. On a typed
+panel the sort is by genome label, so a region of one label can overlap two
+groups of another label of the same lineage; it joined one and the lineage
+was reported as two overlapping regions. A region now folds every group it
+touches into one.
+
+Both were found by reading and reproduced with constructed regions only;
+neither was reached end to end. The harness outputs are unchanged.
+
+Co-Authored-By: Claude Fable 5.1
+EOF
+)"
+```
+
+---
+
+### Task 6: Read pool clade labels as written (C6)
+
+Tessera writes pool genomes itself with the defline `>{accession} {label}` (`discover/nextclade.py`, `_write_genome`; `discover/pool.py`, the NCBI Virus split). `_select_from` and `_type_panel` then pass that defline to `genotype_from_title`, a free-text miner that wants a digit-bearing token of at least four characters. Short and multi-word clades are lost, and with them lineage-aware selection and the recombinant-lineage exclusion for those genomes.
+
+Measured on the 33 pools cached on the audit machine (genomes carrying a label, typed correctly by mining, typed when the label is read as written):
+
+| pool | labelled | mined | read |
+|---|---|---|---|
+| measles | 846 | 0 | 846 |
+| mpox (all clades) | 1556 | 2 | 1556 |
+| VZV | 256 | 0 | 256 |
+| HIV-1 (hxb2) | 1052 | 639 | 1052 |
+| RSV-A | 1677 | 1170 | 1677 |
+| SARS-CoV-2 (XBB dataset) | 2803 | 2672 (421 recombinant-named) | 2803 (549 recombinant-named) |
+
+And the effect on selection from three of those pools (one tip held out as the query, local skani only): measles 1 genome selected -> 4 (4 clades); HIV-1 9 genomes in 5 clades -> 9 genomes in 9 clades; mpox 5 -> 8.
+
+**This changes which references a panel contains for most pathogens.** `run_hybrids.py` does not exercise `_select_from`: it types its pools from the Nextclade tree and calls `select_regional` directly, which is the behaviour this task gives the product. So the hybrids harness has been validating lineage-typed selection that the CLI did not perform, and it cannot serve as this task's gate. The gate below is an end-to-end panel build.
+
+**Files:**
+- Modify: `src/tessera/recomb/typing.py` (new `label_from_pool_header`, `pool_labels`; `build_lineage_map` gains `pool_labels`)
+- Modify: `src/tessera/discover/iterate.py` (`_seed_from_pool`, `_select_from`, `_type_panel`; new `_record_pool_labels`, `_read_pool_labels`, `POOL_LABELS_TSV`)
+- Modify: `docs/reference-panels.md`
+- Test: `tests/unit/test_typing.py`, `tests/unit/test_select_from_lineage.py`, `tests/unit/test_iterate.py`
+
+**Interfaces:**
+- Consumes: nothing from earlier tasks.
+- Produces:
+ - `typing.label_from_pool_header(title: str) -> str | None`
+ - `typing.pool_labels(files: Iterable[Path]) -> dict[str, str]` (file label -> clade)
+ - `typing.build_lineage_map(*, user_tsv=None, datasets_rows=None, title_by_label=None, organism=None, pool_labels: Mapping[str, str] | None = None)`; rows from pool labels carry the source string `"pool-header"`
+ - `iterate._select_from(params, genomes, logger, *, pool_headers: bool = False)`
+ - `iterate.POOL_LABELS_TSV = "pool_labels.tsv"`; `/pool_labels.tsv` is `labelclade`, one row per pool-seeded genome
+
+- [ ] **Step 1: Write the failing tests for the typing helpers**
+
+In `tests/unit/test_typing.py`, replace the import block with:
+
+```python
+from tessera.recomb.typing import (
+ build_lineage_map,
+ dominant_lineage_token,
+ genotype_from_title,
+ label_from_pool_header,
+ lineage_map_from_rows,
+ lineage_of,
+ load_lineage_map,
+ pool_labels,
+ read_lineage_rows,
+ titles_from_collection,
+ typed,
+ write_lineage_map,
+)
+```
+
+and append:
+
+```python
+
+
+# --- structured pool labels -------------------------------------------------
+
+def test_label_from_pool_header_takes_the_label_verbatim():
+ # the short and multi-word clades that title mining drops
+ assert label_from_pool_header("MN908947.3 B.1") == "B.1"
+ assert label_from_pool_header("OX1 XBB") == "XBB"
+ assert label_from_pool_header("K03455 B") == "B"
+ assert label_from_pool_header("MV1 D8") == "D8"
+ assert label_from_pool_header("NC_063383 IIb") == "IIb"
+ assert label_from_pool_header("VZV1 clade 2 wild-type") == "clade 2 wild-type"
+ assert genotype_from_title("MN908947.3 B.1") is None # why mining is not enough
+
+
+def test_label_from_pool_header_none_for_unlabelled_genomes():
+ for header in ("ACC1", "ACC1 ", "ACC1 NA", "ACC1 example", ""):
+ assert label_from_pool_header(header) is None
+
+
+def test_pool_labels_reads_each_file_header(tmp_path: Path):
+ (tmp_path / "ACC1.fasta").write_text(">ACC1 B\nACGT\n")
+ (tmp_path / "ACC2.fasta").write_text(">ACC2 NA\nACGT\n")
+ (tmp_path / "ACC3.2.fasta").write_text(">ACC3.2 clade 5\nACGT\n")
+ files = sorted(tmp_path.iterdir())
+ assert pool_labels(files) == {"ACC1": "B", "ACC3.2": "clade 5"}
+
+
+def test_build_lineage_map_pool_label_beats_title_and_yields_to_datasets_and_user(
+ tmp_path: Path,
+):
+ user = tmp_path / "user.tsv"
+ user.write_text("U1\tUSER\n")
+ rows = build_lineage_map(
+ user_tsv=user,
+ datasets_rows=[("D1", "DATASETS")],
+ title_by_label={"P1": "P1 CRF01_AE", "T1": "T1 Norovirus GII.4 isolate x"},
+ pool_labels={"P1": "A1", "D1": "pool", "U1": "pool"},
+ )
+ assert dict((label, (g, src)) for label, g, src in rows) == {
+ "P1": ("A1", "pool-header"), # not the token mined from its title
+ "T1": ("GII.4", "title"), # no pool label: mined as before
+ "D1": ("DATASETS", "ncbi-datasets"),
+ "U1": ("USER", "user"),
+ }
+```
+
+- [ ] **Step 2: Write the failing tests for selection**
+
+Append to `tests/unit/test_select_from_lineage.py`:
+
+```python
+
+
+def _captured_lineage_map(monkeypatch, tmp_path, *, pool_headers: bool):
+ pool = tmp_path / "pool"
+ pool.mkdir()
+ genomes = [_write(pool, "ACC1", "B"), _write(pool, "ACC2", "XBB"),
+ _write(pool, "ACC3", "NA")]
+ captured = {}
+
+ def fake_select_regional(query, genomes, **kwargs):
+ captured.update(kwargs)
+ return PoolSelection(selected=list(genomes), table=[])
+
+ monkeypatch.setattr("tessera.discover.pool.select_regional", fake_select_regional)
+ params = FillParams(query=tmp_path / "q.fasta", collection=None, output=tmp_path / "out")
+ it._select_from(params, genomes, _LOG, pool_headers=pool_headers)
+ return captured["lineage_of"]
+
+
+def test_select_from_reads_structured_pool_labels(monkeypatch, tmp_path):
+ # Short clade labels as a Nextclade pool writes them. Mined as free text they are all
+ # dropped, so lineage selection and the recombinant exclusion never see them.
+ lineage_of = _captured_lineage_map(monkeypatch, tmp_path, pool_headers=True)
+ assert lineage_of == {"ACC1": "B", "ACC2": "XBB"} # "NA" is not a clade
+
+
+def test_select_from_still_mines_a_user_pool(monkeypatch, tmp_path):
+ # A --candidate-pool holds the user's own files; their headers are free text and a
+ # trailing word must not be taken for a clade.
+ assert _captured_lineage_map(monkeypatch, tmp_path, pool_headers=False) is None
+```
+
+- [ ] **Step 3: Extend the existing seeding test**
+
+In `tests/unit/test_iterate.py`, in `test_seed_source_nextclade_routes_through_pool_selection`, replace
+
+```python
+ def fake_select(params, genomes, logger):
+ from tessera.discover.pool import PoolSelection
+ return PoolSelection(selected=list(genomes))
+```
+
+with
+
+```python
+ def fake_select(params, genomes, logger, *, pool_headers=False):
+ from tessera.discover.pool import PoolSelection
+ captured["pool_headers"] = pool_headers
+ return PoolSelection(selected=list(genomes))
+```
+
+and after the line `assert captured["dataset"] == "nextstrain/sars-cov-2/XBB"` add:
+
+```python
+ # a Nextclade pool's headers hold structured clade labels; they are read as such
+ assert captured["pool_headers"] is True
+ # ... and carried to the panel's lineage table, although "A1" is too short to mine
+ assert (out / "pool_labels.tsv").read_text() == "REF1\tA1\n"
+ assert "REF1\tA1\tpool-header" in (out / "lineages.tsv").read_text()
+```
+
+(The test's pool genome is written as `>REF1 A1`.)
+
+- [ ] **Step 4: Run them to see them fail**
+
+Run: `pytest tests/unit/test_typing.py tests/unit/test_select_from_lineage.py tests/unit/test_iterate.py -q`
+Expected: `test_typing.py` fails at import (`cannot import name 'label_from_pool_header'`); the two new selection tests fail with `TypeError: _select_from() got an unexpected keyword argument 'pool_headers'`; the seeding test fails on `assert captured["pool_headers"] is True` (the flag is never passed).
+
+- [ ] **Step 5: Add the typing helpers**
+
+In `src/tessera/recomb/typing.py`, insert immediately before `def first_header(`:
+
+```python
+# What the pool builders write in the label slot when a genome carries no clade: an
+# empty note, Nextclade's "NA", and the marker given to a dataset's example sequences.
+_NO_POOL_LABEL = frozenset({"", "na", "example"})
+
+
+def label_from_pool_header(title: str) -> str | None:
+ """The clade/lineage label of a pool genome's defline, verbatim.
+
+ Tessera writes pool genomes itself -- Nextclade tree tips and NCBI Virus genomes --
+ with the defline ``{accession} {label}``, so the label is everything after the
+ accession and needs no mining. :func:`genotype_from_title` is for free-text GenBank
+ titles: it wants a digit-bearing token of four characters or more, which drops
+ ``B.1``, ``XBB``, ``D8``, ``IIb``, the HIV pure subtypes and any multi-word clade
+ (``clade 2 wild-type``). Returns ``None`` for a genome that carries no label.
+ """
+ parts = title.split(None, 1)
+ label = parts[1].strip() if len(parts) == 2 else ""
+ return None if label.lower() in _NO_POOL_LABEL else label
+
+
+def pool_labels(files: Iterable[Path]) -> dict[str, str]:
+ """``label -> clade`` for pool genome files, read from their deflines."""
+ out: dict[str, str] = {}
+ for path in files:
+ label = label_from_pool_header(first_header(path))
+ if label:
+ out[strip_sequence_extension(path.name)] = label
+ return out
+
+
+```
+
+- [ ] **Step 6: Give `build_lineage_map` a pool-label tier**
+
+In the same file, replace the signature, docstring and first loop of `build_lineage_map`:
+
+```python
+ title_by_label: Mapping[str, str] | None = None,
+ organism: str | None = None,
+) -> list[tuple[str, str, str]]:
+ """Merge typed names from all sources by priority into sorted sidecar rows.
+
+ Priority (highest wins): user map > NCBI datasets lineage > mined title token.
+ Keys are normalized with ``strip_sequence_extension`` so they match MSA leaves.
+ """
+ merged: dict[str, tuple[str, str]] = {} # label -> (genotype, source)
+ # Source 3 (lowest): a token mined from each reference's title/header note.
+ for label, title in (title_by_label or {}).items():
+ genotype = genotype_from_title(title, organism)
+ if genotype:
+ merged[strip_sequence_extension(label)] = (genotype, "title")
+```
+
+with
+
+```python
+ title_by_label: Mapping[str, str] | None = None,
+ organism: str | None = None,
+ pool_labels: Mapping[str, str] | None = None,
+) -> list[tuple[str, str, str]]:
+ """Merge typed names from all sources by priority into sorted sidecar rows.
+
+ Priority (highest wins): user map > NCBI datasets lineage > pool label > mined
+ title token. ``pool_labels`` are the structured labels of genomes that came from a
+ pool Tessera built (see :func:`label_from_pool_header`); a genome that has one is
+ never typed from a mined token. Keys are normalized with
+ ``strip_sequence_extension`` so they match MSA leaves.
+ """
+ merged: dict[str, tuple[str, str]] = {} # label -> (genotype, source)
+ # Source 4 (lowest): a token mined from each reference's title/header note.
+ for label, title in (title_by_label or {}).items():
+ genotype = genotype_from_title(title, organism)
+ if genotype:
+ merged[strip_sequence_extension(label)] = (genotype, "title")
+ # Source 3: the label a pool builder wrote into the header, taken verbatim.
+ for label, clade in (pool_labels or {}).items():
+ if clade:
+ merged[strip_sequence_extension(label)] = (clade, "pool-header")
+```
+
+The parameter shadows the module-level function `pool_labels` inside `build_lineage_map` only; the function body does not call it.
+
+- [ ] **Step 7: Pass pool labels through recruitment**
+
+In `src/tessera/discover/iterate.py`:
+
+(a) add `pool_labels,` to the `from ..recomb.typing import (...)` block, between `organism_from_title,` and `titles_from_collection,`;
+
+(b) after the `from .run import (...)` block and before `@dataclass class FillParams`, add:
+
+```python
+POOL_LABELS_TSV = "pool_labels.tsv" # sidecar: labelclade of pool-seeded genomes
+```
+
+(c) insert immediately before `def _type_panel(`:
+
+```python
+def _record_pool_labels(output: Path, genomes: list[Path]) -> None:
+ """Add the pool labels of ``genomes`` to ``/pool_labels.tsv``.
+
+ Seeded genomes are copied into the collection next to downloaded ones whose headers
+ are GenBank titles; once there, nothing says which header holds a structured label.
+ The sidecar carries that across to :func:`_type_panel`.
+ """
+ merged = _read_pool_labels(output)
+ merged.update(pool_labels(genomes))
+ if not merged:
+ return
+ output.mkdir(parents=True, exist_ok=True)
+ with open(output / POOL_LABELS_TSV, "w") as fo:
+ for label, clade in sorted(merged.items()):
+ fo.write(f"{label}\t{clade}\n")
+
+
+def _read_pool_labels(output: Path) -> dict[str, str]:
+ """``label -> clade`` from ``/pool_labels.tsv`` (empty when absent)."""
+ path = output / POOL_LABELS_TSV
+ if not path.exists():
+ return {}
+ labels: dict[str, str] = {}
+ for line in path.read_text().splitlines():
+ label, _, clade = line.partition("\t")
+ if label and clade:
+ labels[label] = clade
+ return labels
+
+
+```
+
+(d) in `_type_panel`, replace
+
+```python
+ lineage_rows = build_lineage_map(
+ user_tsv=params.lineage_map,
+ title_by_label=titles_from_collection(coll_files),
+ organism=params.taxon,
+ )
+```
+
+with
+
+```python
+ panel_labels = {strip_sequence_extension(p.name) for p in coll_files}
+ lineage_rows = build_lineage_map(
+ user_tsv=params.lineage_map,
+ title_by_label=titles_from_collection(coll_files),
+ organism=params.taxon,
+ pool_labels={
+ label: clade
+ for label, clade in _read_pool_labels(params.output).items()
+ if label in panel_labels
+ },
+ )
+```
+
+(e) in `_seed_from_pool`, replace the last line
+
+```python
+ _copy_into(_select_from(params, genomes, logger).selected, collection, logger)
+```
+
+with
+
+```python
+ # A pool Tessera built carries each genome's clade in its header; a user's own
+ # --candidate-pool carries whatever titles its files happen to have.
+ structured = force_ncbi or params.seed_source != "local"
+ selected = _select_from(params, genomes, logger, pool_headers=structured).selected
+ if structured:
+ _record_pool_labels(params.output, selected)
+ _copy_into(selected, collection, logger)
+```
+
+(f) replace the head of `_select_from`
+
+```python
+def _select_from(params: FillParams, genomes: list[Path], logger: logging.Logger):
+ from .pool import select_regional
+
+ # Type the pool from its headers (user map > NCBI datasets lineage > mined title
+ # token) so the panel is reduced by lineage rather than clade-blind ANI. An empty
+ # map (untyped pool) yields the pre-lineage global behaviour.
+ rows = build_lineage_map(
+ user_tsv=params.lineage_map,
+ title_by_label=titles_from_collection(genomes),
+ organism=params.taxon,
+ )
+```
+
+with
+
+```python
+def _select_from(
+ params: FillParams, genomes: list[Path], logger: logging.Logger,
+ *, pool_headers: bool = False,
+):
+ from .pool import select_regional
+
+ # Type the pool from its headers (user map > pool label > mined title token) so the
+ # panel is reduced by lineage rather than clade-blind ANI. With ``pool_headers`` the
+ # header's label is read as written; mining it as free text loses short clades
+ # (B.1, XBB, D8, IIb, the HIV pure subtypes), which then skip lineage selection and
+ # the recombinant-lineage exclusion. An empty map (untyped pool) yields the
+ # pre-lineage global behaviour.
+ rows = build_lineage_map(
+ user_tsv=params.lineage_map,
+ title_by_label=titles_from_collection(genomes),
+ organism=params.taxon,
+ pool_labels=pool_labels(genomes) if pool_headers else None,
+ )
+```
+
+The `--deep-typing` branch of `_type_panel` is left as it is: there, genomes the miner leaves untyped are typed against the Nextclade tips by nearest reference.
+
+- [ ] **Step 8: Run the tests**
+
+Run: `pytest tests/unit/test_typing.py tests/unit/test_select_from_lineage.py tests/unit/test_iterate.py tests/unit/test_pool.py tests/unit/test_lineage_select.py -q && ruff check src tests && mypy src`
+Expected: all pass.
+
+- [ ] **Step 9: Document it**
+
+In `docs/reference-panels.md`, in the `nextclade` bullet, replace
+
+```
+dataset reference plus its mutations and labelled by clade, so the report names
+parents by clade. Fetched pools are cached per dataset version.
+```
+
+with
+
+```
+dataset reference plus its mutations and labelled by clade, so the report names
+parents by clade. The label is read from the pool as written (short clades such as
+`B.1`, `D8` or `IIb` included), drives lineage-aware selection, and is recorded for the
+seeded genomes in `/pool_labels.tsv`. Fetched pools are cached per dataset
+version.
+```
+
+(If the line breaks in the file differ, keep the wording and re-wrap to 100 columns.)
+
+- [ ] **Step 10: Offline measurement of the selection change**
+
+This needs a populated Nextclade cache (`~/.cache/tessera/nextclade`) and `skani`; skip it and say so if either is absent.
+
+```bash
+export PATH="$PATH:$HOME/miniforge3/envs/recomfi-aln/bin"
+python - <<'EOF'
+import logging, tempfile
+from collections import Counter
+from pathlib import Path
+from tessera.core.cache import cached_genomes
+from tessera.core.io import strip_sequence_extension
+from tessera.discover import iterate as it
+from tessera.discover.iterate import FillParams
+from tessera.recomb.typing import pool_labels
+
+log = logging.getLogger("sel"); log.addHandler(logging.NullHandler()); log.propagate = False
+root = Path.home() / ".cache/tessera/nextclade"
+for pool_dir in sorted(p for p in root.iterdir() if p.is_dir() and not p.name.endswith("_consensus")):
+ genomes = cached_genomes(pool_dir)
+ if len(genomes) < 20:
+ continue
+ truth = pool_labels(genomes)
+ query = genomes[len(genomes) // 2] # hold one tip out as the query
+ pool = [g for g in genomes if g != query]
+ for mode in (False, True):
+ with tempfile.TemporaryDirectory() as td:
+ params = FillParams(query=query, collection=None, output=Path(td))
+ try:
+ selected = it._select_from(params, pool, log, pool_headers=mode).selected
+ except Exception as exc: # skani rejects very short gene datasets
+ print(pool_dir.name[:44], mode, f"skipped: {type(exc).__name__}", sep="\t")
+ continue
+ clades = Counter(truth.get(strip_sequence_extension(g.name), "-") for g in selected)
+ print(pool_dir.name[:44], f"pool_headers={mode}", f"selected={len(selected)}",
+ f"clades={len(clades)}", sep="\t")
+EOF
+```
+
+Save the output as `validation/data/audit-c/task6/selection.txt`. Expected direction: with `pool_headers=True` the number of clades represented in the selection is equal or higher for every pool. Measured for this plan: measles 1 -> 4, HIV-1 5 -> 9, mpox 5 -> 8 clades.
+
+- [ ] **Step 11: End-to-end gate (needs network, mafft, skani, skDER, nextclade data)**
+
+For measles, mpox and HIV-1, take one genome from the cached pool as a stand-in query and run the shipped command on `main` and on this branch:
+
+```bash
+export PATH="$PATH:$HOME/miniforge3/envs/recomfi-aln/bin"
+for ds in nextstrain/measles/genome/WHO-2012 nextstrain/mpox/all-clades community/neherlab/hiv-1/hxb2; do
+ name=$(echo "$ds" | tr '/-' '__')
+ query=$(ls ~/.cache/tessera/nextclade/${name}_*/*.fasta | head -200 | tail -1)
+ mkdir -p validation/data/audit-c/task6
+ tessera detect --query "$query" --nextclade --nextclade-dataset "$ds" \
+ --output "validation/data/audit-c/task6/$name" 2>&1 | tee "validation/data/audit-c/task6/$name.log" \
+ | grep -E "Lineage selection|Seeding|Caller\(s\)"
+done
+```
+
+Run the same loop against `main` first, writing to `validation/data/audit-c/baseline-task6/` instead: `git stash`, run the loop with that output directory, `git stash pop`. (The package is installed editable, so the checked-out source is what runs.)
+
+Acceptance:
+- Each query is an unrecombined tree tip. This branch must not report a present-donor region (a row with `donor_absent = no` in `recombination_regions.tsv`) that `main` does not report. A new region here is a false positive introduced by the panel change.
+- The `Lineage selection:` log line shows at least as many lineages represented as on `main`.
+- `lineages.tsv` names the seeded references by clade with source `pool-header`.
+
+If a query gains a region: revert the commit, record the three logs, and report it. Do not adjust `select_regional`.
+
+If there is no network, this step cannot run. Record "Task 6 end-to-end gate: not run" in the PR and leave the commit in place only if the maintainer agrees; the unit tests and Step 10 alone do not validate a panel-composition change.
+
+- [ ] **Step 12: Commit**
+
+```bash
+git add src/tessera/recomb/typing.py src/tessera/discover/iterate.py docs/reference-panels.md \
+ tests/unit/test_typing.py tests/unit/test_select_from_lineage.py tests/unit/test_iterate.py
+git commit -m "$(cat <<'EOF'
+Read pool clade labels as written instead of mining them
+
+Tessera writes pool genomes with the defline "accession label", then passed
+that defline to genotype_from_title, which is built for free-text GenBank
+titles and wants a digit-bearing token of four or more characters. Short and
+multi-word clades did not survive: on cached pools every measles tip (846),
+all but two mpox tips (1556) and 413 of 1052 HIV-1 tips -- all the pure
+subtypes -- came back untyped, and 128 recombinant-named SARS-CoV-2 tips
+were never excluded because they were never typed.
+
+For those genomes lineage-aware selection fell back to ANI dereplication.
+run_hybrids.py did not see it: it types its pools from the tree and calls
+select_regional itself, so the harness was validating a selection the CLI
+did not perform.
+
+The label is now read verbatim for pools Tessera built (Nextclade, NCBI
+Virus) and carried to the panel's lineage table through a pool_labels.tsv
+sidecar. A user's --candidate-pool is still mined: its headers are free
+text.
+
+This changes panel composition. Selection from cached pools:
+.
+End-to-end: .
+
+Co-Authored-By: Claude Fable 5.1
+EOF
+)"
+```
+
+---
+
+### Task 7: Cap the PHI window (C5)
+
+The PHI window is counted in informative-site ranks and defaults to 100. PHI compares the incompatibility of nearby site pairs with a random reordering; once the window spans most of the informative sites nearly every pair is "nearby" and the test cannot reject. On `example_data/divergent.msa.fasta` (159 informative sites; four callers agree on the tract) the default gives p = 1.0, and a window of 20 gives p = 0.001.
+
+Six rules were compared while writing this plan (significant / testable; "testable" excludes alignments where the window covers every pair):
+
+| rule | benchmark power (rec > 0) | benchmark specificity (rec = 0) | clonal simulations, false | `divergent` example | `cryptic_insert` example | near-identical recombinant |
+|---|---|---|---|---|---|---|
+| 100 (current) | 8/73 | 16/16 | 0/20 | 1.0 | not testable | 0/5 |
+| `sites // 2` | 8/185 | 36/36 | 0/20 | 1.0 | 1.0 | 0/5 |
+| `sites // 4` | 11/185 | 36/36 | 0/20 | 0.001 | 0.81 | 0/5 |
+| **`sites // 10`** | **13/185** | **36/36** | **0/20** | **0.001** | **0.026** | **4/5** |
+| fixed 20 | 21/150 | 29/30 | 3/20 | 0.001 | 1.0 | 2/5 |
+| fixed 10 | 16/164 | 30/30 | 2/20 | 0.001 | 0.48 | 4/5 |
+
+The candidate is `min(--phi-window, sites // 10)`: it has no false positive on any clonal set, it is the only rule that recovers both shipped examples, and it leaves the window at 100 for any alignment with 1000 or more informative sites, so divergent panels are unaffected.
+
+**A finding outside this plan, to be reported and not fixed here.** PHI power on the Jaya 2023 benchmark is low under every rule (at most 14 %). The benchmark's most divergent class (`mut = 0.1`) yields zero biallelic informative columns, and `mut = 0.01` about 65: `biallelic_columns` discards every column with a third allele, which at high divergence is most of them. The window is not what limits PHI there. Record this in the PR under "Found, not fixed".
+
+**Interaction with plan B.** With this cap the effective window is always below `sites - 1`, so the "PHI not testable (window covers every pair)" case that plan B item B7 words can no longer occur through the default path. Plan B's wording stays correct but becomes unreachable; tell whoever executes plan B.
+
+**Files:**
+- Modify: `src/tessera/recomb/diagnostics.py` (new `effective_phi_window`, `_WINDOW_FRACTION`; `recombination_signal`)
+- Modify: `src/tessera/cli/cmd_recomb.py` (`--phi-window` help)
+- Modify: `docs/detection-methods.md`
+- Test: `tests/unit/test_diagnostics.py`
+
+**Interfaces:**
+- Consumes: nothing from earlier tasks.
+- Produces: `diagnostics.effective_phi_window(window: int, n_informative: int) -> int`. `RecombinationSignal.phi_window` holds the window actually used.
+
+- [ ] **Step 1: Write the failing tests**
+
+In `tests/unit/test_diagnostics.py`, add `effective_phi_window,` to the `from tessera.recomb.diagnostics import (...)` block (after `corroborating_intervals,`), add the line `from tessera.recomb.similarity import _read_alignment` after that block, and insert before the `# --- corroborating intervals` comment:
+
+```python
+def test_effective_phi_window_is_capped_at_a_tenth_of_the_sites() -> None:
+ assert effective_phi_window(100, 159) == 15
+ assert effective_phi_window(100, 54) == 5
+ assert effective_phi_window(100, 10) == 1 # never below one rank
+ # unchanged once the alignment holds ten windows' worth of sites
+ assert effective_phi_window(100, 1000) == 100
+ assert effective_phi_window(100, 5000) == 100
+ assert effective_phi_window(8, 5000) == 8 # a smaller request is honoured
+
+
+def test_signal_at_the_minimum_number_of_informative_sites() -> None:
+ # Ten informative sites is the floor for reporting a signal at all; the window is
+ # then a single rank and the test must still return a valid p-value.
+ rows = _block_alignment(recombinant=True, per_block=5)
+ signal = recombination_signal(rows, "s0", lambda c: c)
+ assert signal is not None and signal.n_informative == 10
+ assert signal.phi_window == 1
+ assert 0.0 < signal.phi_p <= 1.0
+
+
+def test_default_window_has_power_on_the_shipped_example(example_data) -> None:
+ # 159 informative sites: a 100-rank window spans most pairs, and PHI was 1.0 on an
+ # alignment whose recombinant tract four callers agree on.
+ rows = _read_alignment(str(example_data / "divergent.msa.fasta"))
+ signal = recombination_signal(rows, "query", lambda c: c)
+ assert signal is not None
+ assert signal.n_informative == 159
+ assert signal.phi_window == 15 # the window used, not the one asked for
+ assert signal.phi_p < 0.05
+
+
+```
+
+- [ ] **Step 2: Run them to see them fail**
+
+Run: `pytest tests/unit/test_diagnostics.py -q`
+Expected: collection error, `ImportError: cannot import name 'effective_phi_window'`.
+
+- [ ] **Step 3: Implement the cap**
+
+In `src/tessera/recomb/diagnostics.py`:
+
+(a) after `_PERMUTATIONS = 1000` add:
+
+```python
+# The PHI window is capped at 1/_WINDOW_FRACTION of the informative sites; see
+# effective_phi_window.
+_WINDOW_FRACTION = 10
+```
+
+(b) insert immediately before `def recombination_signal(`:
+
+```python
+def effective_phi_window(window: int, n_informative: int) -> int:
+ """The PHI window actually used: ``window``, capped at a tenth of the sites.
+
+ PHI compares the incompatibility of *nearby* site pairs with that of a random
+ reordering. "Nearby" only means something while the window is small against the
+ number of informative sites: once it spans most of them nearly every pair is
+ "nearby", the statistic barely changes under permutation, and the test cannot
+ reject whatever the data hold (with ``window >= n - 1`` it is exactly invariant and
+ p is always 1). Capping at ``n // 10`` keeps at least ten windows' worth of sites,
+ and leaves the window untouched on any alignment with ``10 * window`` sites or more.
+ """
+ return max(1, min(int(window), n_informative // _WINDOW_FRACTION))
+
+
+```
+
+(c) in `recombination_signal`, replace
+
+```python
+ incompatible = incompatibility_matrix(allele1, allele0)
+ p, observed = phi_pvalue(incompatible, window, seed=seed)
+```
+
+with
+
+```python
+ incompatible = incompatibility_matrix(allele1, allele0)
+ # `window` is an upper bound. The reported `phi_window` is the one actually used,
+ # so the report never states a window the test did not run with.
+ window = effective_phi_window(window, int(z))
+ p, observed = phi_pvalue(incompatible, window, seed=seed)
+```
+
+The rest of the function already passes `window` to `phi_profile` and stores it as `phi_window`.
+
+- [ ] **Step 4: Run the tests**
+
+Run: `pytest tests/unit/test_diagnostics.py -q`
+Expected: 12 passed (the nine existing tests use a 24-site alignment with `window=8`, which the cap turns into 2; they still pass).
+
+- [ ] **Step 5: Say so in the CLI and the docs**
+
+In `src/tessera/cli/cmd_recomb.py`, replace
+
+```python
+ help="PHI test window width, in informative-site ranks.",
+```
+
+with
+
+```python
+ help="PHI test window width, in informative-site ranks. An upper bound: the "
+ "window is capped at a tenth of the informative sites, and the report states "
+ "the one used.",
+```
+
+In `docs/detection-methods.md`, replace the sentence
+
+```
+parent-free outputs. The diagnostic runs for every `--method`; disable with
+`--no-phi`, or widen its window with `--phi-window`.
+```
+
+with
+
+```
+parent-free outputs. The diagnostic runs for every `--method`; disable with
+`--no-phi`.
+
+The PHI window (`--phi-window`, default 100) is counted in informative-site ranks and
+is an **upper bound**: the window used is capped at a tenth of the informative sites.
+PHI asks whether *nearby* sites are more compatible than a random reordering, and
+"nearby" stops meaning anything once the window spans most of the sites -- at 159
+informative sites a 100-rank window gave p = 1 on an alignment whose tract every caller
+found, where a 15-rank window gives p = 0.001. Alignments with a thousand informative
+sites or more are unaffected. The report and `recombination_profile.tsv` state the
+window that was used.
+```
+
+- [ ] **Step 6: Harness gate**
+
+Run: `validation/data/audit-c/gate.sh task7`
+
+Acceptance:
+- `benchmark.txt`: specificity `1.00`; the power count not lower than the baseline's. (Measured: 8 -> 13 significant of 185 testable recombining alignments; specificity 36/36 before and after.)
+- `spec_mm1.txt`, `spec_mm2.txt`, `validation.txt`, `validation_regions.tsv`: identical to the previous task. PHI does not gate any region; it only sets the `parent_free_support` flag, which `validation_regions.tsv` does not list.
+- Additionally count how many clonal runs now carry a significant PHI, which the gate script does not print:
+
+```bash
+python - <<'EOF'
+import importlib.util, sys
+import numpy as np
+from pathlib import Path
+from tessera.recomb.diagnostics import recombination_signal
+spec = importlib.util.spec_from_file_location("run_specificity", Path("validation/run_specificity.py"))
+rs = importlib.util.module_from_spec(spec); sys.modules["run_specificity"] = rs; spec.loader.exec_module(rs)
+sig = n = 0
+for scenario in rs.SCENARIOS:
+ for rep in range(10):
+ seqs = rs.simulate_clonal(scenario, seed=1000 + rep)
+ rows = {k: np.frombuffer(v.encode(), dtype=np.uint8) for k, v in seqs.items()}
+ s = recombination_signal(rows, rs.QUERY, lambda c: c)
+ n += 1
+ sig += s is not None and s.phi_p < 0.05
+print(f"clonal alignments with PHI p < 0.05: {sig}/{n}")
+EOF
+```
+
+ Acceptance: at most 4 of 40 (10 %; the documented rate on clonal data is 2.5-7.5 % against a nominal 5 %). Measured for this plan: 2/40. These panels have about 2600 informative sites, so the window is 100 before and after and the count should equal the one on `main`.
+
+If the benchmark specificity drops below 1.00 or the clonal count exceeds 4/40: revert and record.
+
+- [ ] **Step 7: Commit**
+
+```bash
+git add src/tessera/recomb/diagnostics.py src/tessera/cli/cmd_recomb.py docs/detection-methods.md tests/unit/test_diagnostics.py
+git commit -m "$(cat <<'EOF'
+Cap the PHI window at a tenth of the informative sites
+
+The PHI window is counted in informative-site ranks and defaulted to 100
+whatever the alignment held. PHI asks whether nearby sites are more
+compatible than a random reordering; once the window spans most of the
+sites nearly every pair is nearby and the test cannot reject. With 101 sites
+or fewer it is exactly invariant under permutation (p = 1 always). On the
+shipped divergent example, 159 sites, it gave p = 1.0 for a tract all four
+callers find.
+
+The window is now min(--phi-window, sites // 10). Of six rules compared,
+this is the only one with no false positive on clonal simulations that also
+recovers both shipped examples (p = 0.001 and 0.026). Alignments with 1000+
+informative sites keep a window of 100.
+
+Jaya 2023 benchmark: power -> of 185, specificity 36/36
+before and after. Power there stays low for a reason the window does not
+touch -- the biallelic-column filter leaves few sites at high divergence --
+which is recorded in the PR, not addressed here.
+
+Co-Authored-By: Claude Fable 5.1
+EOF
+)"
+```
+
+---
+
+### Task 8: Report the span an HMM region's donor actually explains (C2)
+
+An HMM segment is where the path sat in the donor state. Two things make that wider than the donor tract. Once in the donor state the path has no emission reason to leave while donor and major are identical, and the jump penalty keeps it there: a 600 bp difference was reported as 11.7 kb. And a segment's ends sit on the window grid (`positions[lo] - step` to `positions[hi] + step`), so a tract starting at column 0 is reported from the first window centre.
+
+**Design.** The reported span is from the first to the last site inside the segment at which the query matches the donor and not the major. Sites matching both parents or neither cannot hold a region open. A segment that includes the first (last) window is searched from the alignment start (to its end). **The sign test is still computed on the segment**, so the set of called regions does not depend on this step; `mean_sim_minor` / `mean_sim_major` are identity over the reported columns (the quantity the site callers already report), under either windowing.
+
+**Tried and not chosen:** the 3SEQ maximum-descent interval over the full extent of the segment's windows. Positive-control breakpoint error 9 bp against 13 bp, but on `hiv_crf02ag` an HMM region grew from 3744-4584 to 3834-4645, overlapped the coverage gap at 4624-7309 and removed that donor-absent row. A change that only tightens coordinates is easier to defend.
+
+**Files:**
+- Modify: `src/tessera/recomb/regions.py` (imports; `_call_regions_hmm` pass 2; new `_tract_span`)
+- Create: `tests/integration/test_regimes.py`
+- Modify: `docs/detection-methods.md`
+
+**Interfaces:**
+- Consumes: `validation/run_regimes.py` from Task 2 (`simulate_sibling_tract`, `scan`, `_evolve`, `_as_text`, `QUERY`); `threeseq.triplet_steps(rows, query, major, minor) -> (steps, columns)` where `steps` is `+1` for a site matching the major and `-1` for a site matching the minor.
+- Produces: `regions._tract_span(work: WindowSimilarity, seg: Segment, major: str) -> tuple[int, int]`. HMM `Region.msa_start/msa_end/query_start/query_end` are the trimmed span; `pvalue`, `qvalue`, `support`, `n_windows`, `breakpoint_lo/hi` are unchanged. Task 9 appends tests to `tests/integration/test_regimes.py`.
+
+- [ ] **Step 1: Write the failing tests**
+
+Create `tests/integration/test_regimes.py`:
+
+```python
+"""Pipeline-level regressions for the regimes simulated by validation/run_regimes.py.
+
+One seed each at reduced size, so they run in the fast suite; the harness itself
+reports rates over replicates.
+"""
+
+from __future__ import annotations
+
+import importlib.util
+import logging
+import sys
+from pathlib import Path
+
+import numpy as np
+
+from tessera.recomb.analyze import analyze
+from tessera.recomb.regions import RegionParams, call_regions
+from tessera.recomb.similarity import compute_similarity
+
+from ..conftest import write_fasta
+
+_PATH = Path(__file__).resolve().parents[2] / "validation" / "run_regimes.py"
+_SPEC = importlib.util.spec_from_file_location("run_regimes", _PATH)
+rg = importlib.util.module_from_spec(_SPEC)
+sys.modules["run_regimes"] = rg
+_SPEC.loader.exec_module(rg)
+
+WINDOW, STEP = 1000, 100
+_LOG = logging.getLogger("tessera.test")
+
+
+def _hmm_regions(result, window: int = WINDOW):
+ params = RegionParams.with_defaults(window, method="hmm")
+ regions, major, _ = call_regions(result, analyze(result), window, params)
+ return regions, major
+
+
+# --- HMM region span -------------------------------------------------------
+
+def test_hmm_region_stops_where_donor_and_major_become_identical(tmp_path) -> None:
+ # The query differs from the sibling only over `span`; downstream of it parent_A
+ # (the donor relative to the sibling) and the sibling are identical.
+ seqs, (lo, hi) = rg.simulate_sibling_tract(5000)
+ msa = write_fasta(tmp_path / "sib.fasta", seqs)
+ result = compute_similarity(str(msa), rg.QUERY, WINDOW, STEP)
+ regions, major = _hmm_regions(result)
+ assert major == "sibling"
+ assert [r.minor_parent for r in regions] == ["parent_A"]
+ region = regions[0]
+ assert abs(region.msa_start - lo) <= STEP and abs(region.msa_end - hi) <= STEP
+ assert region.length_bp <= (hi - lo) + STEP # was 11.7 kb for a 600 bp difference
+
+
+def test_hmm_region_reaches_the_alignment_start_for_a_terminal_tract(tmp_path) -> None:
+ rng = np.random.default_rng(1)
+ root = rng.integers(0, 4, size=3000)
+ parent_a, parent_b, other = (rg._evolve(root, 0.05, rng) for _ in range(3))
+ query = parent_a.copy()
+ query[:800] = parent_b[:800] # the donor tract starts at column 0
+ seqs = rg._as_text({rg.QUERY: query, "parent_A": parent_a, "parent_B": parent_b,
+ "other": other})
+ msa = write_fasta(tmp_path / "term.fasta", seqs)
+ result = compute_similarity(str(msa), rg.QUERY, 300, 30)
+ regions, _ = _hmm_regions(result, window=300)
+ assert [r.minor_parent for r in regions] == ["parent_B"]
+ # the first window centre is column 150; the tract must not be reported from there
+ assert regions[0].msa_start <= 30
+ assert abs(regions[0].msa_end - 800) <= 30
+
+
+def test_trimmed_region_reports_query_coordinates_across_a_query_gap(tmp_path) -> None:
+ # 300 alignment columns the query does not have (a backbone insertion), upstream of
+ # the region: MSA and query coordinates then differ by exactly those 300 columns.
+ seqs, (lo, hi) = rg.simulate_sibling_tract(5000)
+ gap_at, gap = 3000, 300
+ padded = {}
+ for label, seq in seqs.items():
+ insert = "-" * gap if label == rg.QUERY else seqs["parent_A"][gap_at:gap_at + gap]
+ padded[label] = seq[:gap_at] + insert + seq[gap_at:]
+ msa = write_fasta(tmp_path / "gapped.fasta", padded)
+ result = compute_similarity(str(msa), rg.QUERY, WINDOW, STEP)
+ regions, _ = _hmm_regions(result)
+ assert [r.minor_parent for r in regions] == ["parent_A"]
+ region = regions[0]
+ assert abs(region.msa_start - (lo + gap)) <= STEP
+ assert region.query_start == region.msa_start - gap
+ assert region.query_end == region.msa_end - gap
+ assert region.length_bp == region.length_msa
+
+
+def test_trimmed_hmm_region_still_merges_with_the_site_callers(tmp_path) -> None:
+ # Agreement is counted on overlapping regions. A trimmed HMM region must still
+ # overlap the 3SEQ / MaxChi tract, or --min-methods 2 would drop a real event.
+ seqs, (lo, hi) = rg.simulate_sibling_tract(5000)
+ rows = rg.scan(seqs, _LOG, min_methods=2)
+ hits = [r for r in rows if int(r["query_start"]) < hi and int(r["query_end"]) > lo]
+ assert len(hits) == 1
+ assert {"hmm", "3seq", "maxchi"} <= set(hits[0]["methods"].split(","))
+```
+
+The last two tests are guards for the review-focus items: they pass before and after this task and pin behaviour the trimming must not break.
+
+- [ ] **Step 2: Run them to see them fail**
+
+Run: `pytest tests/integration/test_regimes.py -q`
+Expected: 2 failed, 2 passed.
+- `test_hmm_region_stops_where_donor_and_major_become_identical`: `assert (100 <= 100 and 11000 <= 100)` -- the region is 17900-29600.
+- `test_hmm_region_reaches_the_alignment_start_for_a_terminal_tract`: `assert 120 <= 30`.
+
+- [ ] **Step 3: Implement `_tract_span`**
+
+In `src/tessera/recomb/regions.py`:
+
+(a) change two import lines:
+
+```python
+from .hmm import DEFAULT_JUMP_RATE, Segment, segment_query
+```
+
+and add after `from .stats import benjamini_hochberg, sign_test_pvalue`:
+
+```python
+from .threeseq import triplet_steps
+```
+
+(`threeseq` imports `regions` only inside its caller function, so there is no import cycle.)
+
+(b) insert immediately before `def _call_regions_heuristic(`:
+
+```python
+def _tract_span(work: WindowSimilarity, seg: Segment, major: str) -> tuple[int, int]:
+ """MSA columns ``[start, end)`` of an HMM segment that its donor actually explains.
+
+ The span from the first to the last site where the query matches the donor and not
+ the major. Columns where the query matches both parents or neither carry no
+ information about which one it was copied from, so they cannot hold a region open:
+ a run of windows in which donor and major are identical is left out, and so is the
+ padding the window grid adds at either end.
+
+ A segment that includes the first (last) window is searched from the alignment
+ start (to its end): window centres stop half a window short of the ends, so a tract
+ reaching an end would otherwise be reported from the first centre.
+
+ Falls back to the segment's own span when no site in it favours the donor.
+ """
+ lo = 0 if seg.start_window == 0 else seg.msa_start
+ hi = work.width if seg.end_window == len(work.positions) - 1 else seg.msa_end
+ steps, cols = triplet_steps(work.rows, work.query, major, seg.state)
+ donor_cols = cols[(steps == -1) & (cols >= lo) & (cols < hi)]
+ if donor_cols.size == 0:
+ return seg.msa_start, seg.msa_end
+ return int(donor_cols[0]), int(donor_cols[-1]) + 1
+
+
+```
+
+(c) in `_call_regions_hmm`, replace everything from ` support = favor_minor / (favor_minor + favor_major)` down to and including the line ` query_start=seg.query_start, query_end=seg.query_end,` -- that is, this block:
+
+```python
+ support = favor_minor / (favor_minor + favor_major)
+ if work.window_spans:
+ # Informative-site windowing: the per-window values are identity at
+ # polymorphic columns only, far below the identity over all columns (a
+ # 99 %-identical donor can sit near 0.8). Reporting their mean as the
+ # region's similarity would misstate it, and the coverage check compares
+ # this value against a base-pair threshold. Use identity over the region's
+ # columns, the same quantity the site-based callers report.
+ mean_minor = region_identity(
+ work.rows, work.query, seg.state, seg.msa_start, seg.msa_end
+ )
+ mean_major = region_identity(
+ work.rows, work.query, major, seg.msa_start, seg.msa_end
+ )
+ else:
+ idx = range(seg.start_window, seg.end_window + 1)
+ minor_sims = [work.similarities[seg.state][i] for i in idx
+ if not isnan(work.similarities[seg.state][i])]
+ major_sims = [work.similarities[major][i] for i in idx
+ if not isnan(work.similarities[major][i])]
+ mean_minor = mean(minor_sims) if minor_sims else float("nan")
+ mean_major = mean(major_sims) if major_sims else float("nan")
+ regions.append(
+ Region(
+ minor_parent=seg.state, major_parent=major,
+ msa_start=seg.msa_start, msa_end=seg.msa_end,
+ query_start=seg.query_start, query_end=seg.query_end,
+```
+
+with
+
+```python
+ support = favor_minor / (favor_minor + favor_major)
+ # The segment is where the HMM path sat in the donor state, which is not the
+ # same as where the donor is the better match: once in the donor state the path
+ # has no reason to leave while donor and major are identical, and its ends are
+ # only as fine as the window grid. The reported span is the part of the segment
+ # that the distinguishing sites support (see _tract_span). The sign test above
+ # stays on the segment, so which regions are called does not depend on this
+ # step -- only their coordinates do.
+ lo, hi = _tract_span(work, seg, major)
+ # Identity over the reported columns, the same quantity the site-based callers
+ # report, on the same scale under either windowing (the per-window values of
+ # informative-site windowing are identity at polymorphic columns only).
+ mean_minor = region_identity(work.rows, work.query, seg.state, lo, hi)
+ mean_major = region_identity(work.rows, work.query, major, lo, hi)
+ regions.append(
+ Region(
+ minor_parent=seg.state, major_parent=major,
+ msa_start=lo, msa_end=hi,
+ query_start=work.column_to_query(lo), query_end=work.column_to_query(hi),
+```
+
+`isnan` and `mean` are still used by `_call_regions_heuristic`; leave their imports.
+
+- [ ] **Step 4: Run the tests**
+
+Run: `pytest tests/integration/test_regimes.py tests/unit/test_hmm.py tests/unit/test_clusters.py tests/integration -q && pytest -m "not requires_binary" -q && ruff check src tests && mypy src`
+Expected: all pass. No existing test pins an HMM region's coordinates to the window grid.
+
+- [ ] **Step 5: Document it**
+
+In `docs/detection-methods.md`, in the "HMM caller" section, add after the paragraph that ends "...is flagged as marginal rather than dropped. The legacy `--method heuristic` (margin / merge-gap / min-region) is kept for comparison.":
+
+```markdown
+**The reported span is not the segment.** A segment is where the HMM path sat in the
+donor state. Once there, the path has no reason to leave while donor and major are
+identical, and its ends are only as fine as the window grid. The coordinates reported
+for an HMM region therefore run from the first to the last site inside the segment at
+which the query matches the donor and not the major; columns matching both parents or
+neither cannot hold a region open. A segment that includes the first or last window is
+searched to the end of the alignment, so a tract reaching a genome end is reported from
+that end. The sign test, its p-value and `support` are computed on the whole segment, so
+this step changes where a region is drawn and never whether it is called.
+`mean_sim_minor` / `mean_sim_major` are identity over the reported columns.
+```
+
+- [ ] **Step 6: Harness gate**
+
+```bash
+validation/data/audit-c/gate.sh task8
+python validation/run_regimes.py --reps 10 | tee validation/data/audit-c/task8/regimes_mm1.txt
+python validation/run_regimes.py --reps 10 --min-methods 2 | tee validation/data/audit-c/task8/regimes_mm2.txt
+python validation/run_hybrids.py > validation/data/audit-c/task8/hybrids.txt 2>&1
+```
+
+Acceptance (the spec's criteria for C2, made numeric):
+- `sibling_tract`: median reported length **at most 1100 bp** at both gates (baseline 11700 bp; measured 1001 bp). The remaining excess over 600 bp is the Bootscan member of the merged region, whose span runs between window centres; it is not addressed here.
+- `spec_mm1.txt`, `spec_mm2.txt`: the `TOTAL` lines identical to the previous task (measured: identical -- the same eight false regions, with tighter coordinates).
+- Positive control: detection and donor unchanged; **median breakpoint error not higher** than the previous task's (measured 50 bp -> 13 bp).
+- `validation.txt`: same verdict per dataset. `validation_regions.tsv`: **the same rows in the same order**; only coordinates of rows whose `methods` include `hmm` may differ. Measured differences, for comparison:
+
+ | dataset | donor | before | after |
+ |---|---|---|---|
+ | `hcv_2k1b` | GT2a_JFH1 | 33-3127 | 33-3113 |
+ | `hiv_crf02ag` | G | 1587-2624 | 1606-2614 |
+ | `hiv_crf02ag` | G | 3744-4584 | 3744-4539 |
+ | `hiv_crf02ag` | G | 7644-8989 | 7644-9058 (segment includes the last window) |
+ | `norovirus_gii` | GII.1_Hawaii | 5040-7447 | 5070-7484 (last window) |
+ | `sarscov2_xbb` | BA.2.75 | 21883-26888 | 22007-26112 |
+
+ `hcv_2k1b`'s published breakpoint is near nt 3187: the reported end moves 14 bp further from it (3127 -> 3113). State that in the PR.
+- Hybrids: no case changes from PASS to FAIL; where the table prints a breakpoint or donor-agreement column, no value worse than the baseline's.
+
+If a row appears or disappears in `validation_regions.tsv`, or a hybrids case flips to FAIL: revert, record the diff, go to Task 9 -- and note that Task 9's gate was written assuming this task is in place (see its Step 6).
+
+- [ ] **Step 7: Commit**
+
+```bash
+git add src/tessera/recomb/regions.py tests/integration/test_regimes.py docs/detection-methods.md
+git commit -m "$(cat <<'EOF'
+Report the span an HMM region's donor explains, not the whole segment
+
+An HMM segment is where the path sat in the donor state. After a tract the
+path has no emission reason to return to the major while the two are
+identical downstream, and the jump penalty keeps it where it is: with a
+sibling recombinant in the panel a 600 bp difference was reported as 11.7 kb
+(39 % of the query), and the ensemble's union span carried it into the
+merged region. Segment ends also sit on the window grid, so a tract starting
+at column 0 was reported from the first window centre.
+
+The reported span now runs from the first to the last site in the segment
+where the query matches the donor and not the major, and a segment that
+includes the first or last window is searched to the alignment end. Sites
+matching both parents or neither cannot hold a region open.
+
+The sign test stays on the segment. Recomputing it on the trimmed span would
+lower p-values on a span chosen from the data, in the caller that already
+produces every false region at --min-methods 1; keeping it means the set of
+called regions cannot change here. mean_sim_minor / mean_sim_major are now
+identity over the reported columns under either windowing.
+
+Not chosen: the 3SEQ max-descent interval over the segment's full window
+extent. 9 bp against 13 bp on the positive control, but it changed rows on
+hiv_crf02ag (a region grew into a coverage gap and removed a donor-absent
+row).
+
+Harness: .
+
+Co-Authored-By: Claude Fable 5.1
+EOF
+)"
+```
+
+---
+
+### Task 9: Lineage clustering on informative columns under informative-site windowing (C1)
+
+`cluster_references` merges two references when their per-window identity over **all** columns stays at or above 0.985 with no region-sized run below it. On a panel less than about 1.5 % divergent every pair qualifies, the whole panel pools into one lineage, and the HMM has one state. That is the regime informative-site windowing exists for, so in that regime the HMM never contributes a call when there are 4-200 references.
+
+**Design.** When the scan is under informative-site windowing (`result.window_spans` is non-empty), compare each pair on the informative columns only. There the 0.985 floor again separates true duplicates, which agree, from distinct lineages, which differ at a large share of the columns where the panel varies. Under base-pair windowing nothing changes.
+
+**The cost, measured.** With clustering no longer pooling the panel, the HMM competes the references individually, as it does with `--no-cluster-lineages`. On 20 clonal near-identical panels that produced one HMM-only false region at `--min-methods 1` (1/20, against 0/20 on `main`) and none at `--min-methods 2`. The spec's criterion "specificity no worse" holds on `run_specificity.py` (which never enters informative-site windowing) and does not strictly hold on these panels. This is a trade the maintainer should see: it is carried into Task 10.
+
+**Depends on Task 8.** Without the span trimming, an HMM region on such a panel is wide (median breakpoint error of the merged region 1087 bp with clustering off on `main`, against 263 bp without the HMM, at 6 replicates). With Task 8 in place the merged error is the same with and without the HMM (465 bp at 10 replicates). If Task 8 was reverted, do not start this task; record "C1 not attempted: depends on C2".
+
+**Files:**
+- Modify: `src/tessera/recomb/clusters.py` (imports; `cluster_references`)
+- Modify: `tests/integration/test_regimes.py` (imports; three appended tests)
+- Modify: `docs/detection-methods.md`
+
+**Interfaces:**
+- Consumes: `tests/integration/test_regimes.py` and its `rg`, `_hmm_regions`, `WINDOW` from Task 8; `rg.simulate_near_identical` from Task 2; `similarity._informative_column_mask(rows, ref_labels) -> np.ndarray` (boolean mask of columns polymorphic among the references).
+- Produces: `cluster_references(result, window_size, params)` unchanged in signature.
+
+- [ ] **Step 1: Write the failing tests**
+
+In `tests/integration/test_regimes.py`, add to the imports
+
+```python
+from tessera.recomb.clusters import cluster_references
+```
+
+(after the `analyze` import) and change the similarity import to
+
+```python
+from tessera.recomb.similarity import compute_similarity, compute_similarity_informative
+```
+
+then append to the file:
+
+```python
+# --- lineage clustering on a near-identical panel ---------------------------
+
+NEAR = {"length": 30_000, "distance": 0.004, "tract": (12_000, 18_000)}
+
+
+def test_near_identical_references_are_not_pooled_into_one_lineage(tmp_path) -> None:
+ seqs, _ = rg.simulate_near_identical(3, recombinant=True, **NEAR)
+ seqs["R0_copy"] = seqs["R0"] # a true duplicate: this is what clustering is for
+ msa = write_fasta(tmp_path / "near.fasta", seqs)
+ result = compute_similarity_informative(str(msa), rg.QUERY)
+ params = RegionParams.with_defaults(WINDOW, method="hmm")
+ clusters = {frozenset(c) for c in cluster_references(result, WINDOW, params)}
+ assert frozenset({"R0", "R0_copy"}) in clusters
+ assert len(clusters) == len(rg.NEAR_REFS) # every other reference on its own
+
+
+def test_duplicates_still_merge_when_informative_windowing_is_forced(tmp_path) -> None:
+ # --informative-sites on a divergent panel: the comparison is on informative columns
+ # there too, and true duplicates must still pool while distinct parents stay apart.
+ rng = np.random.default_rng(4)
+ root = rng.integers(0, 4, size=6000)
+ parent_a, parent_b = rg._evolve(root, 0.05, rng), rg._evolve(root, 0.05, rng)
+ seqs = rg._as_text({rg.QUERY: parent_a, "pa0": parent_a, "pa1": parent_a,
+ "pb0": parent_b, "pb1": parent_b})
+ msa = write_fasta(tmp_path / "dups.fasta", seqs)
+ result = compute_similarity_informative(str(msa), rg.QUERY)
+ params = RegionParams.with_defaults(WINDOW, method="hmm")
+ clusters = {frozenset(c) for c in cluster_references(result, WINDOW, params)}
+ assert clusters == {frozenset({"pa0", "pa1"}), frozenset({"pb0", "pb1"})}
+
+
+def test_hmm_calls_the_tract_on_a_near_identical_panel(tmp_path) -> None:
+ seqs, (lo, hi) = rg.simulate_near_identical(3, recombinant=True, **NEAR)
+ msa = write_fasta(tmp_path / "near.fasta", seqs)
+ result = compute_similarity_informative(str(msa), rg.QUERY)
+ regions, major = _hmm_regions(result)
+ assert major == "R0"
+ donors = [r for r in regions if r.minor_parent == "R1"]
+ assert len(donors) == 1
+ assert donors[0].msa_start < hi and donors[0].msa_end > lo
+```
+
+`test_duplicates_still_merge_when_informative_windowing_is_forced` is a guard: it passes before and after.
+
+- [ ] **Step 2: Run them to see them fail**
+
+Run: `pytest tests/integration/test_regimes.py -q`
+Expected: 2 failed, 5 passed.
+- `test_near_identical_references_are_not_pooled_into_one_lineage`: the only cluster is `{R0, R0_copy, R1, R2, R3, R4}`.
+- `test_hmm_calls_the_tract_on_a_near_identical_panel`: `assert 0 == 1` -- no region.
+
+- [ ] **Step 3: Compare on informative columns**
+
+In `src/tessera/recomb/clusters.py`, replace the import
+
+```python
+from .similarity import WindowSimilarity, _best_per_window, _canonical_mask
+```
+
+with
+
+```python
+from .similarity import (
+ WindowSimilarity,
+ _best_per_window,
+ _canonical_mask,
+ _informative_column_mask,
+)
+```
+
+and in `cluster_references` replace
+
+```python
+ canon = {label: _canonical_mask(result.rows[label]) for label in labels}
+ uf = _UnionFind(labels)
+ for i, a in enumerate(labels):
+ a_row, a_canon = result.rows[a], canon[a]
+ for b in labels[i + 1:]:
+ comp = a_canon & canon[b]
+```
+
+with
+
+```python
+ canon = {label: _canonical_mask(result.rows[label]) for label in labels}
+ # Informative-site windowing is what a near-identical panel is scanned under, and
+ # on such a panel identity over all columns is above the floor for every pair: the
+ # whole panel would pool into one lineage and leave the HMM a single state. Compare
+ # the pair where the panel varies instead. There the floor again separates true
+ # duplicates, which agree, from distinct lineages, which differ at a large share of
+ # the informative columns.
+ informative = (
+ _informative_column_mask(result.rows, labels) if result.window_spans else None
+ )
+ uf = _UnionFind(labels)
+ for i, a in enumerate(labels):
+ a_row, a_canon = result.rows[a], canon[a]
+ for b in labels[i + 1:]:
+ comp = a_canon & canon[b]
+ if informative is not None:
+ comp = comp & informative
+```
+
+The next line (`match = comp & (a_row == result.rows[b])`) and the rest of the loop are unchanged.
+
+- [ ] **Step 4: Run the tests**
+
+Run: `pytest tests/integration/test_regimes.py tests/unit/test_clusters.py -q && pytest -m "not requires_binary" -q && ruff check src tests && mypy src`
+Expected: all pass. The existing clustering tests run under base-pair windowing and are unaffected.
+
+- [ ] **Step 5: Document it**
+
+In `docs/detection-methods.md`, in "Low-divergence panels", add after the paragraph ending "...you cannot localise a switch more finely than the spacing of the discriminating sites.":
+
+```markdown
+Before the HMM runs, near-duplicate references are pooled into lineages so that
+duplicates do not tie every window (`--cluster-lineages`, on by default for panels of 4
+to 200 references). Two references pool when their agreement stays at or above 98.5 % in
+every window. Under informative-site windowing that agreement is measured **on the
+informative columns**: on a near-identical panel every pair is above 98.5 % over all
+columns, and measuring it there would pool the whole panel and leave the HMM nothing to
+choose between. On the informative columns only true duplicates agree.
+```
+
+- [ ] **Step 6: Harness gate**
+
+```bash
+validation/data/audit-c/gate.sh task9
+python validation/run_regimes.py --reps 10 | tee validation/data/audit-c/task9/regimes_mm1.txt
+python validation/run_regimes.py --reps 10 --min-methods 2 | tee validation/data/audit-c/task9/regimes_mm2.txt
+python - <<'EOF' | tee validation/data/audit-c/task9/near_clonal.txt
+import importlib.util, logging, sys
+from collections import Counter
+from pathlib import Path
+spec = importlib.util.spec_from_file_location("run_regimes", Path("validation/run_regimes.py"))
+rg = importlib.util.module_from_spec(spec); sys.modules["run_regimes"] = rg; spec.loader.exec_module(rg)
+log = logging.getLogger("x"); log.addHandler(logging.NullHandler()); log.propagate = False
+for min_methods in (1, 2):
+ bad = 0
+ callers = Counter()
+ for rep in range(20):
+ seqs, _ = rg.simulate_near_identical(4000 + rep, recombinant=False)
+ n, per = rg.score_clonal(rg.scan(seqs, log, min_methods=min_methods))
+ bad += bool(n)
+ callers.update(per)
+ print(f"near-identical clonal, --min-methods {min_methods}: "
+ f"runs with a false region {bad}/20 {dict(callers)}")
+EOF
+python validation/run_hybrids.py > validation/data/audit-c/task9/hybrids.txt 2>&1
+```
+
+Acceptance (the spec's criteria for C1):
+- **The HMM calls the tract.** `near_identical`: `called by the HMM` at least 9/10 at both gates (baseline 0/10; measured 10/10), `detected` 10/10, `correct donor` 10/10.
+- Median breakpoint error on `near_identical` not higher than Task 8's (measured 465 bp before and after).
+- **Specificity no worse on `run_specificity.py`:** `spec_mm1.txt` and `spec_mm2.txt` `TOTAL` lines identical to Task 8's. (Those panels run under base-pair windowing, so this is expected by construction; a difference means the change leaked into base-pair mode.)
+- `validation.txt` and `validation_regions.tsv` identical to Task 8's (measured: identical).
+- **Near-identical clonal:** 0/20 at `--min-methods 2`. At `--min-methods 1` record the count. Measured: 1/20 (one HMM-only region, q = 0.021) against 0/20 on `main`. A count of 0 or 1 is consistent with what was measured; **2 or more of 20 is beyond it -- stop, revert, and take the number to the maintainer.**
+- **Hybrids unchanged or better:** no case from PASS to FAIL. Look in particular at the cases analysed under informative-site windowing (`mode` column: `mpox`, `vzv`, `ebola`, `lowdiv_rsv`, `neg_sarscov2`) and at every `neg_*` case: a negative control that gains an HMM region fails the gate.
+
+If the gate fails: revert, record all five outputs. The recorded 1/20 goes into Task 10's decision record whether or not this task is kept.
+
+- [ ] **Step 7: Commit**
+
+```bash
+git add src/tessera/recomb/clusters.py tests/integration/test_regimes.py docs/detection-methods.md
+git commit -m "$(cat <<'EOF'
+Cluster on informative columns under informative-site windowing
+
+Lineage clustering merges two references whose per-window identity over all
+columns stays at or above 98.5 %. On a panel less than ~1.5 % divergent every
+pair qualifies, the whole panel pools into one lineage and the HMM has a
+single state: with 4-200 references it could not call anything, in exactly
+the regime informative-site windowing was added for. In 10 of 10 simulated
+panels ~0.2 % divergent the HMM was absent from a call that 3SEQ and MaxChi
+made; with --no-cluster-lineages it was present in all 10.
+
+Under informative-site windowing the pairwise agreement is now measured on
+the informative columns, where the floor again separates true duplicates
+from distinct lineages. Base-pair windowing is unchanged.
+
+The cost: the HMM now competes those references individually and can be
+wrong on its own. On 20 clonal near-identical panels: /20 runs with an
+HMM-only false region at --min-methods 1 (0/20 before), 0/20 at
+--min-methods 2. run_specificity.py is unchanged at both gates.
+
+Harness: .
+
+Co-Authored-By: Claude Fable 5.1
+EOF
+)"
+```
+
+---
+
+### Task 10: Characterise the false regions and record the decision (C8)
+
+At the CLI default (`--min-methods 1`) the specificity harness reports a false region in about half of its clonal runs, all from the HMM. `run_specificity.py` defaults to `--min-methods 2`, so its clean result does not describe the shipped default. This task does not change the default. It commits a script that shows what the false regions are, and a decision record that puts the options and their measured costs in front of the maintainer.
+
+**Files:**
+- Create: `validation/characterise_false_regions.py`
+- Create: `docs/superpowers/specs/2026-10-01-min-methods-default-decision.md`
+- Modify: `validation/README.md`
+
+**Interfaces:**
+- Consumes: `validation/run_specificity.py` (`SCENARIOS`, `QUERY`, `simulate_clonal`).
+- Produces: `characterise_false_regions.false_region_rows(scenario, seed, seqs, query) -> list[dict]` and a TSV on stdout with the columns in `COLUMNS`.
+
+- [ ] **Step 1: Write the script**
+
+Create `validation/characterise_false_regions.py`:
+
+```python
+#!/usr/bin/env python
+"""Describe each false region the callers report on the specificity harness's clonal data.
+
+``run_specificity.py`` counts false regions; this lists them, one row per region and
+per caller, with the quantities needed to see *why* each was called: its length, how
+many sites distinguish the two parents and which way they lean, its p- and q-value,
+whether donor and major belong to the same simulated clade, and whether lineage
+clustering pooled anything. It exists to inform a decision about the ``--min-methods``
+default and the HMM's significance gate, not to gate a change.
+
+Needs no aligner, network or downloaded data.
+
+ python validation/characterise_false_regions.py # 10 replicates
+ python validation/characterise_false_regions.py --reps 3
+"""
+
+from __future__ import annotations
+
+import importlib.util
+import sys
+import tempfile
+from pathlib import Path
+
+from tessera.recomb.analyze import analyze
+from tessera.recomb.clusters import all_singletons, cluster_references
+from tessera.recomb.regions import DEFAULT_METHODS, RegionParams, call_regions
+from tessera.recomb.similarity import compute_similarity, discordant_counts
+
+WINDOW, STEP = 1000, 100 # the CLI defaults, as run_specificity.py uses
+COLUMNS = (
+ "scenario", "seed", "caller", "minor", "major", "msa_start", "msa_end", "length",
+ "favor_minor", "favor_major", "pvalue", "qvalue", "sim_minor", "sim_major",
+ "same_clade", "panel_clustered",
+)
+
+
+def _load_specificity():
+ path = Path(__file__).resolve().parent / "run_specificity.py"
+ spec = importlib.util.spec_from_file_location("run_specificity", path)
+ assert spec is not None and spec.loader is not None
+ module = importlib.util.module_from_spec(spec)
+ sys.modules["run_specificity"] = module
+ spec.loader.exec_module(module)
+ return module
+
+
+def false_region_rows(scenario: str, seed: int, seqs: dict[str, str], query: str) -> list[dict]:
+ """One row per region any default caller reports on a clonal alignment."""
+ rows: list[dict] = []
+ with tempfile.TemporaryDirectory() as td:
+ msa = Path(td) / "aln.fasta"
+ msa.write_text("".join(f">{k}\n{v}\n" for k, v in seqs.items()))
+ result = compute_similarity(str(msa), query, WINDOW, STEP)
+ analysis = analyze(result)
+ clustered = not all_singletons(
+ cluster_references(result, WINDOW, RegionParams.with_defaults(WINDOW))
+ )
+ for method in DEFAULT_METHODS:
+ params = RegionParams.with_defaults(WINDOW, method=method)
+ regions, _major, _siblings = call_regions(result, analysis, WINDOW, params)
+ for r in regions:
+ favor_minor, favor_major = discordant_counts(
+ result.rows, query, r.major_parent, r.minor_parent, r.msa_start, r.msa_end
+ )
+ rows.append({
+ "scenario": scenario, "seed": seed, "caller": method,
+ "minor": r.minor_parent, "major": r.major_parent,
+ "msa_start": r.msa_start, "msa_end": r.msa_end,
+ "length": r.msa_end - r.msa_start,
+ "favor_minor": favor_minor, "favor_major": favor_major,
+ "pvalue": "" if r.pvalue is None else f"{r.pvalue:.3g}",
+ "qvalue": "" if r.qvalue is None else f"{r.qvalue:.3g}",
+ "sim_minor": r.mean_sim_minor, "sim_major": r.mean_sim_major,
+ # tips are named , e.g. A2
+ "same_clade": r.minor_parent[0] == r.major_parent[0],
+ "panel_clustered": clustered,
+ })
+ return rows
+
+
+def main(argv: list[str]) -> int:
+ reps = int(argv[argv.index("--reps") + 1]) if "--reps" in argv else 10
+ rs = _load_specificity()
+ print("\t".join(COLUMNS))
+ total = 0
+ for scenario in rs.SCENARIOS:
+ for rep in range(reps):
+ seed = 1000 + rep
+ for row in false_region_rows(scenario, seed,
+ rs.simulate_clonal(scenario, seed=seed), rs.QUERY):
+ print("\t".join(str(row[c]) for c in COLUMNS))
+ total += 1
+ print(f"# {total} false region(s) over {reps * len(rs.SCENARIOS)} clonal alignment(s); "
+ "single-caller rows, before the ensemble merge and any --min-methods gate",
+ file=sys.stderr)
+ return 0
+
+
+if __name__ == "__main__":
+ raise SystemExit(main(sys.argv[1:]))
+```
+
+- [ ] **Step 2: Run it**
+
+```bash
+mkdir -p validation/data/audit-c/task10
+python validation/characterise_false_regions.py --reps 10 > validation/data/audit-c/task10/false_regions.tsv
+column -t -s$'\t' validation/data/audit-c/task10/false_regions.tsv | head -40
+ruff check validation
+```
+
+The table it must produce has one row per region and these columns:
+
+`scenario seed caller minor major msa_start msa_end length favor_minor favor_major pvalue qvalue sim_minor sim_major same_clade panel_clustered`
+
+For reference, on `main` at 3 replicates it produced eight rows, all `caller = hmm`:
+
+| scenario | seed | minor | major | length | favor_minor | favor_major | pvalue | qvalue | same_clade | panel_clustered |
+|---|---|---|---|---|---|---|---|---|---|---|
+| clean | 1000 | A2 | A1 | 1400 | 20 | 8 | 0.0178 | 0.0357 | True | False |
+| clean | 1001 | A1 | A3 | 4400 | 59 | 40 | 0.035 | 0.035 | True | False |
+| asrv | 1000 | A2 | A3 | 2100 | 29 | 14 | 0.0158 | 0.0315 | True | False |
+| asrv | 1002 | A1 | A0 | 1800 | 29 | 14 | 0.0158 | 0.0394 | True | False |
+| asrv | 1002 | A2 | A0 | 1000 | 15 | 3 | 0.00377 | 0.0188 | True | False |
+| lineage_rate | 1000 | A2 | A1 | 1600 | 20 | 8 | 0.0178 | 0.0357 | True | False |
+| lineage_rate | 1001 | A1 | A0 | 1400 | 24 | 5 | 0.000273 | 0.00164 | True | False |
+| rate_shift | 1002 | A3 | A0 | 1300 | 43 | 23 | 0.00933 | 0.0373 | True | False |
+
+(With Task 8 in place the same eight regions appear with lengths 894-4049.)
+
+- [ ] **Step 3: Summarise what the table shows**
+
+Compute, from your 10-replicate table: the number of rows per caller; the share with `same_clade = True`; the share with `panel_clustered = True`; the median `length`; the median `favor_minor + favor_major`; and how many rows have `qvalue > 0.01`. These six numbers go into the decision record in Step 5.
+
+What the 3-replicate table showed, to compare against:
+- every false region is an HMM call;
+- every one is a switch between two tips of the query's **own clade** (the references about 2.4 % apart), never across clades;
+- none of the panels was pooled by lineage clustering (the tips are below the 98.5 % floor), so the tips competed individually;
+- the regions are 1-4.4 kb and rest on 18-99 distinguishing sites;
+- seven of eight have a q-value between 0.01 and 0.04.
+
+The mechanism this points to: the HMM chooses a segment because the donor wins there, and the sign test is then run on the same sites. The test does not know a segment was searched for, so it is anti-conservative; 3SEQ and MaxChi, whose nulls account for the scan, called none of the eight.
+
+- [ ] **Step 4: Measure the two options that do not need a design**
+
+(a) The agreement gate: already in `validation/data/audit-c/task9/spec_mm2.txt` and `regimes_mm2.txt`.
+
+(b) `--alpha 0.01` for the whole scan, which the table suggests would remove most rows. `run_specificity.py` has no alpha flag, so:
+
+```bash
+python - <<'EOF' | tee validation/data/audit-c/task10/alpha_001.txt
+import importlib.util, logging, sys, tempfile
+from pathlib import Path
+from tessera.recomb.run import RecombParams, run_recomb
+spec = importlib.util.spec_from_file_location("run_specificity", Path("validation/run_specificity.py"))
+rs = importlib.util.module_from_spec(spec); sys.modules["run_specificity"] = rs; spec.loader.exec_module(rs)
+log = logging.getLogger("x"); log.addHandler(logging.NullHandler()); log.propagate = False
+
+def scan(seqs):
+ with tempfile.TemporaryDirectory() as td:
+ msa = Path(td) / "aln.fasta"
+ rs._write_fasta(msa, seqs)
+ run_recomb(RecombParams(msa=msa, output=Path(td) / "out", query=rs.QUERY,
+ plot_format="png", min_methods=1, alpha=0.01), log)
+ return rs._read_regions(Path(td) / "out" / "recombination_regions.tsv")
+
+bad = runs = 0
+for scenario in rs.SCENARIOS:
+ for rep in range(10):
+ n, _ = rs.score_negative(scan(rs.simulate_clonal(scenario, seed=1000 + rep)))
+ bad += bool(n)
+ runs += 1
+hits = 0
+for rep in range(10):
+ seqs, tract = rs.simulate_recombinant(seed=2000 + rep)
+ hits += rs.score_positive(scan(seqs), tract, donor_prefix="B")["detected"]
+print(f"--alpha 0.01, --min-methods 1: runs with a false region {bad}/{runs}; "
+ f"positive control detected {hits}/10")
+EOF
+```
+
+Neither option is implemented as a default here; these are numbers for the record.
+
+- [ ] **Step 5: Write the decision record**
+
+Create `docs/superpowers/specs/2026-10-01-min-methods-default-decision.md` with the text below. The numbers in it were measured while this plan was written, at the replicate counts stated (option B's on a scratch copy that is not part of this branch -- keep those as they are, labelled as measured on 2026-10-01); **replace every other number with your own 10-replicate measurement** (the file names are given beside each) and keep the replicate count beside every number.
+
+```markdown
+# Decision record: false regions at the default agreement gate
+
+**Status:** open -- for the maintainer. **Date opened:** 2026-10-01.
+**Question:** at the CLI default (`--min-methods 1`) the specificity harness reports a
+false region in roughly half of its clonal runs. Should the default change, should the
+HMM's significance gate change, or should the default stay and be documented?
+
+## What was measured
+
+All on simulated clonal data (`validation/run_specificity.py`: four clades about 16 %
+apart, four tips each about 2.4 % apart, a clonal query in clade A). Every region is a
+false positive by construction. Replicate counts are small; the intervals are wide.
+
+| setting | runs with a false region | source | positive control (3 kb tract) |
+|---|---|---|---|
+| `--min-methods 1` (CLI default) | 7/12 (58 %, CI 32-81 %) | hmm = 8 of 8 regions | 3/3 detected, 3/3 donor |
+| `--min-methods 2` (harness default) | 0/12 (CI 0-24 %) | -- | 3/3 detected, 3/3 donor |
+
+Source files: `validation/data/audit-c/task9/spec_mm1.txt`, `spec_mm2.txt`.
+
+On near-identical panels (`validation/run_regimes.py`, references about 0.2 % apart),
+after lineage clustering stopped pooling the whole panel (audit item C1):
+
+| setting | clonal runs with a false region | tract detected | HMM among the callers |
+|---|---|---|---|
+| `--min-methods 1` | 1/20 (hmm) | 10/10 | 10/10 |
+| `--min-methods 2` | 0/20 | 10/10 | 10/10 |
+
+Source files: `validation/data/audit-c/task9/near_clonal.txt`, `regimes_mm1.txt`,
+`regimes_mm2.txt`.
+
+## What the false regions are
+
+From `validation/characterise_false_regions.py` (3 replicates, 8 regions):
+
+- all eight are HMM calls; 3SEQ, MaxChi and Bootscan called none of them;
+- all eight are switches between two tips of the query's own clade, about 2.4 % apart;
+ none crosses a clade;
+- none of the panels was pooled by lineage clustering (the tips sit below its 98.5 %
+ floor), so the tips competed as individual genomes;
+- lengths 1.0-4.4 kb, resting on 18-99 distinguishing sites;
+- seven of eight have a q-value between 0.01 and 0.04.
+
+The likely mechanism: the HMM picks a segment because the donor wins there, and the sign
+test is then run on the same sites. It is not aware that a segment was searched for.
+3SEQ and MaxChi test against nulls that account for the scan.
+
+## Options
+
+### A. Change the default to `--min-methods 2`
+
+- For: 0/12 and 0/20 false regions here, with the positive controls intact.
+- Against: this was tried and reverted (PR #47). On the hybrid harness it lost three
+ true detections (`rsv_a`, `mpox`, `masksib_rsv`) and gained nothing, because every
+ negative control already passed. At very low divergence only the HMM has power, so a
+ gate of two is not reachable there.
+- Evidence still needed: `run_hybrids.py` at `--min-methods 2` on the current branch.
+
+### B. Keep `--min-methods 1`; make the HMM's gate aware of the scan
+
+One concrete form was measured, not implemented: an HMM segment must pass both the sign
+test and the 3SEQ max-descent test for its (major, donor) pair, taking the larger
+p-value.
+
+| measurement (3 or 10 replicates as stated) | current gate | scan-aware gate |
+|---|---|---|
+| `run_specificity.py --reps 3 --min-methods 1`: runs with a false region | 7/12 | 2/12 |
+| positive control | 3/3, 3/3, 13 bp | 3/3, 3/3, 13 bp |
+| near-identical, 10 reps: HMM among the callers | 10/10 | 10/10 |
+| near-identical clonal, 10 reps: runs with a false region | 1/10 | 1/10 |
+| `run_validation.py` verdicts | 6 PASS, 1 FAIL, 1 SKIP | the same |
+| `hiv1_crf`, AE_env region 5569-8177 | `hmm,bootscan` | `bootscan` (the HMM vote is lost) |
+| p-value of the four strongest regions | 1e-32 to 1e-67 | 5e-05 (permutation floor) |
+
+- For: removes about two thirds of the false regions without an agreement gate.
+- Against: it removed a real HMM call on HIV-1 CRF01_AE; it replaces exact small
+ p-values by the permutation floor wherever the triplet is too large for the exact
+ test; it was not run on the hybrid harness, where the HMM-only detections live.
+- Evidence still needed: `run_hybrids.py`; a design that keeps the exact p-value.
+
+### C. Keep everything; document the rate and recommend `--min-methods 2` for redundant panels
+
+- For: no behaviour change; the README and `--min-methods` help already describe the
+ trade.
+- Against: a default that reports a false region in about half of clonal runs on a
+ redundant panel is a weak default for a first-time user, and `hcv_clonal_1b` (a real
+ clonal control) currently fails on a single-caller region.
+- Needed: state the measured rate in `README.md` and `validation/README.md`, and make
+ the harness and the CLI use the same default or say plainly that they do not.
+
+### D. Lower `--alpha` for the HMM only
+
+Seven of the eight false regions have q between 0.01 and 0.04.
+
+Measured with `--alpha 0.01` for the whole scan, 3 replicates
+(`validation/data/audit-c/task10/alpha_001.txt`): 1/12 clonal runs with a false region,
+positive control 3/3 detected. It was not measured on real data or on the hybrid harness,
+where weak true regions (several real HMM calls have q between 0.01 and 0.05) would be
+lost, and an alpha for one caller that differs from the others needs a reason beyond "it
+removes these rows".
+
+## Recommendation from the audit
+
+Not A on its own -- the hybrid harness already answered that. B is the option that
+addresses the mechanism, but the form measured here is not ready: it cost a real HMM
+call and coarsened p-values. The audit's suggestion is to take C now (it is
+documentation only) and treat B as a design task of its own, gated on the hybrid
+harness.
+
+## Decision
+
+_To be filled in by the maintainer: option chosen, date, and the harness numbers it was
+taken on._
+```
+
+The final section is left for the maintainer on purpose: this is the one line in the plan that the executor must not fill in.
+
+- [ ] **Step 6: List the script in the validation README**
+
+In `validation/README.md`, in the layout block, add after the `run_regimes.py` line:
+
+```
+ characterise_false_regions.py one row per false region on the clonal simulations
+```
+
+- [ ] **Step 7: Commit**
+
+```bash
+git add validation/characterise_false_regions.py validation/README.md \
+ docs/superpowers/specs/2026-10-01-min-methods-default-decision.md
+git commit -m "$(cat <<'EOF'
+Characterise the false regions at the default gate; open a decision record
+
+run_specificity.py defaults to --min-methods 2 and is clean there. The CLI
+defaults to 1, where the same harness reports a false region in about half
+of its clonal runs. The clean result therefore does not describe what a
+user gets, and nothing in the repository said so.
+
+characterise_false_regions.py lists each false region with the quantities
+needed to see why it was called. On the current callers every one is an HMM
+call between two tips of the query's own clade, resting on a few dozen
+distinguishing sites, with q mostly between 0.01 and 0.04 -- consistent with
+a sign test run on a segment the HMM chose because the donor wins there.
+
+The default is not changed here. The decision record sets out four options
+with what was measured for each and what is still missing (above all the
+hybrid harness), and leaves the choice to the maintainer.
+
+Co-Authored-By: Claude Fable 5.1
+EOF
+)"
+```
+
+- [ ] **Step 8: Ask the maintainer**
+
+Post the decision record's "What was measured", "Options" and "Recommendation" sections in the PR description and ask for a decision. Do not implement any option in this branch.
+
+---
+
+### Task 11: Changelog, final gate, pull request
+
+**Files:**
+- Modify: `CHANGELOG.md`
+
+**Interfaces:**
+- Consumes: the gate outputs under `validation/data/audit-c/` from every earlier task.
+
+- [ ] **Step 1: Write the changelog**
+
+In `CHANGELOG.md`, under `## [Unreleased]`, add the block below. **Delete the bullet of any task whose gate failed and was reverted**, and replace each number with the one you measured.
+
+```markdown
+### Changed
+
+- **HMM region coordinates are the span the donor explains, not the HMM segment.** A
+ segment ran on through columns where donor and major are identical (a 600 bp difference
+ was reported as 11.7 kb when a sibling recombinant was in the panel) and its ends sat on
+ the window grid. The reported span now runs from the first to the last site in the
+ segment where the query matches the donor and not the major, and reaches the alignment
+ end for a terminal tract. **`msa_start`, `msa_end`, `query_start`, `query_end`,
+ `length_bp`, `mean_sim_minor` and `mean_sim_major` change for regions the HMM called**;
+ which regions are called, and their p-values, do not.
+- **On near-identical panels the HMM can contribute again.** Lineage clustering pooled an
+ entire panel less than about 1.5 % divergent into one lineage, leaving the HMM a single
+ state. Under informative-site windowing the pairwise agreement is now measured on the
+ informative columns. On simulated clonal panels of this kind the HMM alone then reported
+ a false region in 1 of 20 runs at `--min-methods 1` (0 of 20 before, and 0 of 20 at
+ `--min-methods 2`).
+- **The PHI window is capped at a tenth of the informative sites.** `--phi-window` is an
+ upper bound; the report states the window used. With the fixed 100-rank window the test
+ could not reject on alignments with few informative sites (p = 1 on the shipped
+ `divergent` example, now 0.001). Alignments with 1000 or more informative sites are
+ unaffected.
+- **Pool clade labels are read as written.** Short and multi-word clades (`B.1`, `XBB`,
+ `D8`, `IIb`, the HIV pure subtypes, `clade 2 wild-type`) were dropped when a Nextclade
+ or NCBI Virus pool was typed, so lineage-aware selection and the recombinant-lineage
+ exclusion skipped those genomes. **Panels seeded from such pools change**; the labels
+ of seeded genomes are recorded in `/pool_labels.tsv`.
+
+### Fixed
+
+- A 3SEQ or MaxChi tract started at the first maximum of the discriminating-site walk
+ before its trough, not the last, so it could begin a few sites early and include a
+ stretch that is not donor tract. MaxChi's permutation null follows the same rule.
+- Exact 3SEQ p-values below about 1e-16 were written as `0.0`. They now keep their
+ magnitude.
+- The ensemble merge could label a region with the same genome as donor and major when
+ the callers disagreed on the backbone, and could report one donor lineage as two
+ overlapping regions on a typed panel.
+
+### Added
+
+- `validation/run_regimes.py`: a self-contained harness for near-identical panels and
+ sibling tracts, the two regimes the specificity harness does not sample.
+- `validation/characterise_false_regions.py`, and a decision record on the false-positive
+ rate at the default agreement gate
+ (`docs/superpowers/specs/2026-10-01-min-methods-default-decision.md`).
+```
+
+- [ ] **Step 2: Final checks**
+
+Run: `ruff check src tests validation && pytest -m "not requires_binary" -q && mypy src`
+Expected with every task kept: `All checks passed!`, `618 passed, 1 deselected`, `Success: no issues found in 79 source files`. (586 at the start; this plan adds 32 tests.) With tasks reverted the count is lower by that task's tests; it must not be below 586.
+
+- [ ] **Step 3: Final gate**
+
+```bash
+validation/data/audit-c/gate.sh final
+python validation/run_regimes.py --reps 10 | tee validation/data/audit-c/final/regimes_mm1.txt
+python validation/run_regimes.py --reps 10 --min-methods 2 | tee validation/data/audit-c/final/regimes_mm2.txt
+python validation/run_hybrids.py > validation/data/audit-c/final/hybrids.txt 2>&1
+diff validation/data/audit-c/baseline/validation_regions.tsv validation/data/audit-c/final/validation_regions.tsv
+```
+
+Acceptance: `final` equals the last kept task's outputs. Build the before/after table for the PR from `baseline/` and `final/`.
+
+- [ ] **Step 4: Commit and open the pull request**
+
+```bash
+git add CHANGELOG.md
+git commit -m "$(cat <<'EOF'
+Record the caller-behaviour changes in the changelog
+
+Co-Authored-By: Claude Fable 5.1
+EOF
+)"
+git push -u origin fix-audit-caller-behaviour
+gh pr create --title "Caller behaviour: audit plan C" --body "$(cat <<'EOF'
+Implements plan C of the post-1.2.0 audit
+(docs/superpowers/specs/2026-10-01-post-1.2.0-audit-design.md, items C1-C8).
+
+## Harness results, before and after
+
+
+
+## Gate results per task
+
+
+
+## Skipped
+
+
+
+## Found, not fixed
+
+- PHI has low power on the Jaya 2023 benchmark under every window rule: the
+ biallelic-column filter leaves no informative sites at mut = 0.1 and about 65 at
+ mut = 0.01.
+- The merged region on the sibling-tract regime is still about 400 bp longer than the
+ true difference: that is the Bootscan member, whose span runs between window centres.
+- `hcv_clonal_1b` still fails (a 12 bp MaxChi-only region); `validation/README.md` still
+ lists it as 0 regions. Plan B owns that text.
+
+## Decision needed
+
+
+
+🤖 Generated with [Claude Code](https://claude.com/claude-code)
+EOF
+)"
+```
+
+Fill every `<...>` in the body from the saved outputs before running `gh pr create`. Do not merge; hand the PR to the maintainer.
diff --git a/docs/superpowers/specs/2026-10-01-post-1.2.0-audit-design.md b/docs/superpowers/specs/2026-10-01-post-1.2.0-audit-design.md
new file mode 100644
index 0000000..9195746
--- /dev/null
+++ b/docs/superpowers/specs/2026-10-01-post-1.2.0-audit-design.md
@@ -0,0 +1,366 @@
+# Post-1.2.0 audit: findings and remediation design
+
+**Date:** 2026-10-01. **Audited commit:** `457bdfb` (main, v1.2.0 plus Dependabot merges).
+**Baseline:** 586 tests pass (`pytest -m "not requires_binary"`), `ruff` and `mypy` clean,
+87 % line coverage.
+
+This is the third audit of Tessera. The first two (v1.1.0, v1.2.0) covered inputs, caches,
+the output contract and the "informative-site identity reported as similarity" defect. This
+one read every source file again and looked at how features interact. The findings below
+were each reproduced unless marked otherwise. Nothing has been fixed yet; this document
+records what was found and the design chosen for each fix. Three implementation plans
+follow from it:
+
+| Plan | Scope | Changes caller behaviour? |
+|---|---|---|
+| A -- `plans/2026-10-01-audit-a-data-safety.md` | data loss, wrong alignments, wrong filters | no (panel building and alignment only) |
+| B -- `plans/2026-10-01-audit-b-report-faithfulness.md` | outputs that misstate what the run did | no region call changes; one flag (`donor_undercovered`) changes |
+| C -- `plans/2026-10-01-audit-c-caller-behaviour.md` | region calling, parent-free signal, panel typing | yes -- every task is harness-gated |
+
+The plans are independent and can land in any order; A and B are the lower-risk pair.
+
+## Constraints that apply to every fix
+
+- No new runtime dependency (CLAUDE.md). `seaborn` is removed, not replaced.
+- Modest scientific language in code, docs and messages; reported numbers must be faithful.
+- A new behaviour needs a test that fails without the change (CONTRIBUTING.md).
+- Anything touching a caller, a default or region calling is validated on
+ `validation/run_specificity.py` (both `--min-methods 1` and `2`) and
+ `validation/run_hybrids.py` before and after, and the numbers are recorded in the PR.
+ Unit-level reasoning about caller behaviour has been contradicted by the harnesses twice.
+- "Could not test" must never be reported as "tested, found nothing".
+- Branch before committing; commit messages end with the project co-author trailer.
+- Append the aligner env to `PATH` (`PATH="$PATH:$HOME/miniforge3/envs/recomfi-aln/bin"`),
+ never prepend it: prepending swaps the Python interpreter.
+
+## Measured context
+
+`python validation/run_specificity.py --reps 3` on the audited commit (clonal simulations,
+every reported region is a false positive):
+
+| gate | runs with a false region | false regions | source | positive control |
+|---|---|---|---|---|
+| `--min-methods 2` (harness default) | 0/12 (CI 0-24 %) | 0 | -- | 3/3 detected, 3/3 donor, 55 bp |
+| `--min-methods 1` (**CLI default**) | 7/12 (58 %, CI 32-81 %) | 8 | hmm = 8 | 3/3 detected, 3/3 donor, 55 bp |
+
+The harness default and the CLI default differ, so the clean harness result does not
+describe the shipped default.
+
+## Plan A -- data safety and alignment correctness
+
+**A1. MAFFT backend ignores strand.** `aligners/mafft.py:79` runs
+`mafft --keeplength --addfragments` without `--adjustdirection`. A reverse-strand genome or
+contig aligns at chance level. Measured with mafft 7.526 on a 6 kb reference and a query
+3 % divergent: forward 0.97 identity; whole query reverse-complemented 0.40-0.43; middle
+contig of three reverse-complemented 0.39 over that third. minimap2 on the same inputs:
+0.97 throughout. *Design:* add `--adjustdirection`. MAFFT renames a reversed record
+`_R_`; `converters/mafft_merge.py` ignores names, so no further change is expected,
+but the test asserts it. Correct `docs/aligners.md` ("handled cleanly" becomes true; the
+statement that MAFFT keeps insertions is false under `--keeplength` and is corrected).
+
+**A2. Reassort alignment-fraction filter uses the wrong unit.** `reassort/assign.py:28`
+sets `MIN_AF = 0.5` and compares it with skani's `Align_fraction_query`, which
+`discover/panel.py:108` returns in percent (0-100). The filter never fires: a tip aligning
+over 20 % of a segment passes. `tests/unit/test_reassort_assign.py` mocks fractions, so it
+pins the mock's unit. *Design:* `MIN_AF = 50.0`, documented as percent; the test mocks move
+to percent.
+
+**A3. `fill-references`: the last round's downloads are never aligned.**
+`discover/iterate.py:307-380` builds the MSA at the top of each round and downloads at the
+bottom. On a `max_rounds` exit the final downloads are in `collection/` and are counted in
+`fill_summary.tsv`, the report caption and `lineages.tsv`, but are absent from
+`panel.msa.fasta`, so detection runs without them.
+
+**A4. `fill-references --curate` does nothing unless a round both finds a gap and downloads
+something** (`iterate.py:333-372`). A starting panel that contains a whole-genome sibling has
+no coverage gap, so the loop converges before curation runs -- the case curation exists for.
+
+**A5. `fill-references --curate --reference X` can delete X.** The curation backbone is
+auto-picked; the user's reference is not protected, and the next round's MSA build fails
+with "Reference 'X' not found among the staged genomes".
+
+*Design for A3-A5 (one restructuring of `_grow_collection`):* each round becomes
+`[curate if --curate] -> build MSA -> scan -> stop checks -> search -> download`. Curation
+therefore runs on round 1 before the first build and on every later round before its build.
+After the loop, if anything was downloaded since the last build, the collection is curated
+(if enabled) and a final MSA is built (`final.msa.fasta`) with no further search, and that
+alignment is what is published as `panel.msa.fasta`. `--reference`, when given, is passed to
+curation as the backbone and is never removed. The copied `/collection` is still the
+only directory mutated.
+
+**A6. `find-references --download --curate` deletes pre-existing genomes.**
+`discover/run.py:289-327` curates the download directory in place; `--download` pointing at
+the collection is the documented invocation. *Design:* files present in the download
+directory before the run are never deleted; curation may only remove files this run
+downloaded. Pre-existing files still take part in the sibling/redundancy comparison.
+
+**A7. `reassort --scan-segments` path escape.** `reassort/scan.py:96` sanitises a segment
+name with `re.sub(r"[^\w.-]+", "_", segment)` but lacks the `.strip(".")` guard that
+`assign.py:88` has; a record named `..` resolves the scan directory to the parent of the
+output and `shutil.rmtree(/collection)` runs. Two names that sanitise to the same
+string share one directory. *Design:* one shared helper (`safe_segment_name`) used by both
+modules, plus a uniqueness suffix on collision.
+
+**A8. MAF conversion (sibeliaz, cactus).** `converters/maf_to_fasta.py:84-108`:
+(i) a multi-contig backbone is laid out in sorted-name order (`contig_10` before
+`contig_2`), not file order; (ii) a backbone contig with no alignment block is dropped, so
+the MSA is narrower than the reference and later coordinates shift; (iii) a genome with no
+block at all gets no row, while `msa/build.py` still counts it. *Design:* the converter
+receives the backbone's contig names and lengths in file order and the full list of
+expected genome labels; uncovered contigs keep their columns; an unplaced genome gets an
+all-gap row and a warning naming it.
+
+**A9. progressiveMauve leaf names.** `aligners/progressivemauve.py:50-98` derives row names
+from the resolved path echoed in the XMFA. Unrecognised extensions, symlinked collections,
+or whitespace in the path give wrong or duplicate row names; a panel symlink whose target
+stem equals the backbone label is silently dropped. Reproduced with a faked XMFA only (the
+tool is not installed). *Design:* map rows by the staged file path given on the command
+line (an explicit `path -> label` map built at staging), never by `Path(name).stem`.
+
+**A10. Small input errors.** `core/io.py:47` raises `IndexError` on a header line `"> "`;
+sequence lines keep trailing spaces/tabs as sequence characters; a truncated MAF row and an
+XMFA without the reference raise bare `IndexError`/`KeyError`. *Design:* strip whitespace
+from sequence lines; raise `UserInputError` naming the file and line for the others.
+
+## Plan B -- reporting faithfulness
+
+**B1. Run provenance names the wrong callers.** `recomb/run.py:407-414` (`_caller_desc`)
+describes every caller other than hmm and 3seq as `heuristic (min ... / margin ... / merge
+...)`. A default run records `hmm + 3seq + heuristic + heuristic` in `run_provenance.json`
+and the report. The record also omits `--min-methods`, sibling exclusion, lineage
+clustering and donor re-attribution. *Design:* one description per caller; add those four
+settings to the provenance.
+
+**B2. Breakpoint-straddling windows are reported as missing references.** A window that
+straddles a breakpoint between parents more than about 10 % apart has a best similarity
+below the adaptive coverage threshold (90th percentile minus 0.05). `recomb/coverage.py`
+calls it a `divergent` gap and `reconcile_gaps` then marks the overlapping region
+`donor_undercovered`. On `example_data/divergent.msa.fasta` (donor identity 1.0, four
+callers agree) the report headline is "low confidence. Possible missing reference", while
+`example_data/README.md` says high confidence. *Design:* after regions are called, a
+coverage gap that lies within one window width of a called region boundary on the
+alignment is relabelled `kind = "breakpoint"`. Such a gap does not caveat a region, is not
+bridged to a donor-absent region, is not counted in the headline, and is listed in
+`coverage_gaps.tsv` and the report table under its own kind with a one-line explanation.
+`call_coverage_gaps` itself is unchanged, so reference recruitment (`fill-references`,
+`find-references`) behaves as before. Because `donor_undercovered` and the confidence
+wording change, run both harnesses before and after.
+
+**B3. The report sums overlapping regions.** `recomb/report_html.py:53,62` adds region
+lengths; the ensemble keeps overlapping regions with different donors separate, so a
+2.2 kb union was reported as "3.3 kb (54.6 %)". *Design:* report the union length in query
+coordinates.
+
+**B4. Method-comparison outputs are stale after `--reattribute-donors`.**
+`recomb/run.py:344-349` re-labels the region but not `method_breakdown`, so
+`recombination_methods.tsv` and the HTML method table name the old donor. *Design:*
+re-attribution returns the old-to-new mapping per region and the breakdown rows are updated
+in step.
+
+**B5. `--lineage-map` pointing at a missing file is ignored (exit 0)** in `recomb`,
+`type-lineages`, `detect`, `fill-references` and `build-panel`. *Design:* `_require_file`
+on the option in each command.
+
+**B6. Barcode caller on an untyped panel.** `recomb/barcode.py:95-101` returns no regions
+and names `labels[0]` -- the first FASTA record -- as major parent. `--method barcode`
+alone then reports "major parent " and a clean negative; in an ensemble the
+methods table shows barcode as "no". *Design:* the caller returns `major = None` when it
+cannot run; `run_recomb` logs a warning, the methods table shows "not run", and a
+barcode-only run on an untyped panel is refused with a `UserInputError`.
+
+**B7. PHI reporting.** (i) `report_html.py:586` always states alpha 0.05, whatever
+`--alpha` is, while the per-region flag uses `--alpha`. (ii) When the number of informative
+sites minus one is at most `--phi-window`, every pair of sites is inside the window, the
+statistic is invariant under permutation and p is always 1; this is reported as "no
+significant signal". *Design:* pass alpha through `ReportContext`; in the invariant case
+report "not testable (N informative sites, window W)" and set `phi_p` to `None`. Changing
+the default window is a caller-behaviour change and is in plan C.
+
+**B8. Plots.** `report_plots.py:49-55,170-178` label a donor-absent region "recombinant:
+" in the backbone's colour. `report.py:106-107` draws `similarity_pair` from the
+top two window winners, which is not "major vs leading minor" when a near-duplicate of the
+backbone is in the panel. With `--top-n 1` the donor swatch is grey. *Design:* donor-absent
+bands get their own label and neutral colour; the pair plot uses the reconciled major and
+the donor of the longest region (falling back to the current behaviour when there is no
+region); the colour map covers every label that appears in a region.
+
+**B9. Stale text.** HTML footer hard-codes `.pdf` under `--plot-format png` and lists files
+that were not written; the HTML methods paragraph and references describe a two-caller
+ensemble; `--method` help says "all but the legacy heuristic" although `geneconv` and
+`barcode` are also outside the default and `geneconv` is not listed;
+`docs/reference-panels.md:244` describes a `--fetch-limit` cap that no longer exists;
+`docs/detection-methods.md` says the methods TSV holds "Y/n" (it holds yes/no) and does not
+describe lineage clustering at all.
+
+**B10. Housekeeping.** `seaborn` is a declared runtime dependency and is never imported.
+`aligners/sibeliaz.py:50` probes the version with `-v`, which the wrapper rejects, so the
+provenance records `illegal option -- v` as the version. `validation/README.md` lists
+`hcv_clonal_1b` as "0 regions" and "7 PASS, 0 FAIL" although that case currently reports a
+12 bp MaxChi-only region, and its phrase "the default agreement gate" refers to the harness
+default, not the CLI default.
+
+**B11. CLI validation gaps.** `find-references --msa ` and `recomb -o ` give "Unexpected error"; `fill-references --max-rounds 0` exits 0 having built
+nothing; `reassort` accepts `--ani-floor 500`, `--margin -3` and a `--dataset` key that
+matches no segment; `type-lineages` rejects `.fasta.gz`/`.fas` collections that
+`collection_genomes()` accepts; a multi-line failure note corrupts `segment_scan.tsv`.
+
+## Plan C -- caller behaviour (harness-gated)
+
+**C1. Lineage clustering silences the HMM on near-identical panels.**
+`recomb/regions.py:279` clusters when there are 4-200 references; `recomb/clusters.py`
+merges two references when their per-window identity over all columns stays at or above
+0.985. On a panel less than about 1.5 % divergent every pair qualifies, all references
+pool into one cluster, the HMM has one state and can call nothing. This is the regime
+informative-site windowing was added for. In 12 of 12 simulations (120 kb, five references
+0.1-0.2 % apart, an 8-15 kb tract) the HMM was silent by default and called the tract with
+`--no-cluster-lineages`; 3SEQ and MaxChi detected all 12 either way, so no missed event was
+shown, but the HMM vote and its corroboration are lost. *Design (evaluate first):* under
+informative-site windowing, measure pairwise agreement on the informative columns only, so
+0.985 again means "near-duplicate" rather than "same species". Acceptance: the HMM calls
+the tract in the 12 simulations; specificity at both gates is no worse; hybrids unchanged
+or better.
+
+**C2. HMM regions overshoot where donor and major are identical.** After a tract the HMM
+has no emission reason to return to the major when the two are identical downstream, and
+the jump penalty keeps it in the donor state. A true 600 bp difference was reported as
+11.7 kb (39 % of the query); the ensemble's union span propagates it. Related: an HMM
+segment is padded by `positions[1] - positions[0]` (`hmm.py:182`), which is the step in
+base-pair mode (so a tract starting at column 0 is reported from 120, and Bootscan from
+150) and an arbitrary first gap in informative-site mode (10 bp against a median of 285 bp
+in one simulation). *Design (evaluate first):* trim each HMM region to the span from its
+first to its last discordant site favouring the donor, and recompute the sign test and
+similarities on the trimmed span. Acceptance: the 600 bp case is reported within one window
+step; median breakpoint error on the specificity positive control and the hybrids harness
+does not increase.
+
+**C3. 3SEQ/MaxChi tract start on ties.** `threeseq.max_descent` takes the first maximum of
+the walk before the trough; the tract should start at the last one. With steps
+`[+1,+1,-1,+1,-1,-1,-1,-1]` the tract is reported as sites 2-8 (support 5/6) instead of 4-8
+(4/4). MaxChi's permutation null (`maxchi._exceedances`) uses the same rule and must change
+with it.
+
+**C4. Exact 3SEQ p-values underflow to 0.0.** `descent_pvalue_exact` returns
+`1.0 - f`, which is exactly 0.0 below about 1e-16; `recombination_regions.tsv` then holds
+`pvalue = 0.0`, `qvalue = 0.0`. *Design:* propagate the complementary probability (reaching
+the depth) so small values keep their magnitude.
+
+**C5. PHI window.** The default `--phi-window 100` is in informative-site ranks. With 159
+informative sites the easy example gives p = 1.0 at window 100 and p = 0.001 at window 20.
+*Design (evaluate first):* cap the effective window relative to the number of informative
+sites; choose the rule on `validation/run_benchmark.py` (PHI power and specificity).
+
+**C6. Structured clade labels are discarded.** Pool FASTA headers are written as
+`>{accession} {clade}` (`discover/nextclade.py:236`, `pool.py:218`) and then re-mined by
+`typing.genotype_from_title`, which needs a token of at least four characters containing a
+digit. `B.1`, `XBB`, `D8`, `IIb` and the HIV pure subtypes come back untyped (measles
+846/846, mpox 1552/1556, HIV-1 413/1052 tips on cached pools). Lineage-aware selection and
+recombinant exclusion skip those genomes; 128 recombinant-named SARS-CoV-2 tips bypass
+exclusion. *Design:* when a header comes from a pool, read the second token directly, as
+`lineage_assign._reference_tips` already does.
+
+**C7. Ensemble merge edge cases** (unit level only, not reached end to end): when callers
+disagree on the backbone a merged region can name the same genome as minor and major
+(`ensemble._merge_group`); on a typed panel grouping is not transitive across labels of one
+lineage, leaving two overlapping regions for one donor.
+
+**C8. False-positive rate at the CLI default.** See "Measured context". *Design:* first
+characterise the eight HMM false regions (length, discordant counts, q-value, whether the
+panel was clustered), then re-measure after C1 and C2, then decide between changing the
+`--min-methods` default and tightening the HMM gate. The decision is the maintainer's; the
+plan produces the numbers for it.
+
+## Revisions after the plans were verified
+
+Each plan's tests and changes were applied in a scratch worktree before the plan was
+written. That showed several designs above to be incomplete or wrong. Where a plan and
+this section disagree with the text above, the plan and this section hold.
+
+**Plan A**
+
+- *A4:* curation before round 1 runs only for a user-supplied `--collection`, not for a
+ freshly seeded one. On a panel under about 1.5 % divergent, sibling filtering classes
+ every genome as a twin of the backbone, so curating the seed would reduce a `detect`
+ panel to one reference. The same collapse exists in `curate-panel` on such panels and is
+ not addressed.
+- *A6:* a second defect: the curation backbone was picked after the download, so with
+ `--download` equal to the collection a downloaded sibling could become the backbone.
+ Fixed in the same task.
+- *A8:* Cactus does not receive the backbone's contig order (the names `hal2maf` emits
+ could not be checked without the binary); its sorted layout stays and is documented.
+- *A10:* stripping whitespace only when reading is not enough -- minimap2 reads the file
+ itself and counts a trailing space as a base (identity 0.25 over half a 6 kb genome). A
+ genome with whitespace in its sequence lines is staged as a cleaned copy.
+- *Not planned:* `maf_to_fasta.py:76` looks up the backbone's genome label in a map keyed
+ by sequence ID, so another genome holding a sequence with that ID becomes the backbone.
+
+**Plan A, after its whole-branch review** (fix commit `351d47d`)
+
+- *A5 reversed:* `--reference` is not the curation backbone. The sibling test is relative
+ to its anchor, so anchoring it on a distant coordinate reference dropped every genome
+ closer to the query (a five-genome panel was cut to the reference alone). The anchor is
+ the query's closest whole-genome relative, as on main; the reference is protected from
+ removal and shown as `sibling-kept` / `redundant-kept` where it would have been dropped.
+- *A8 extended:* a genome aligned only in MAF blocks that lack the backbone also projects
+ to an all-gap row and is named in the same warning.
+- *New, pre-existing defect:* `/collection` was removed unconditionally at the
+ start of `detect`, `fill-references`, `build-panel` and `curate-panel`. A non-empty
+ directory that no earlier run created is now refused. A marker file
+ (`/.tessera-working-collection`) identifies a working copy; outputs of older
+ releases are recognised by the run files beside it.
+- *A4 limit, stated not fixed:* a sibling that is itself the closest genome in a supplied
+ collection becomes the curation anchor and is kept. Changing how the anchor is chosen
+ alters every `--curate` panel and is left as a design decision.
+- Deferred minors from that review are listed in the pull request.
+
+**Plan B**
+
+- *B2:* distance to a region boundary alone would relabel a truly divergent stretch next
+ to a breakpoint. A gap is a `breakpoint` gap only if it is also explained by the
+ region's two parents (the query matches at least one of them at or above the threshold).
+- *B6:* the refusal also covers a typed panel with fewer than two marked clades, and the
+ `--min-methods` clamp counts only callers that ran.
+- *B7:* `RecombinationSignal.phi_p` becomes `float | None`.
+- *B9:* `example_data/README.md` is corrected too (four callers, not two; the printed
+ q-value is 0.0 until C4 lands).
+- Measured with all of plan B applied: specificity unchanged at both gates (0/12 and
+ 7/12, positive control 3/3, 55 bp); `run_validation.py` 6 PASS, 1 FAIL
+ (`hcv_clonal_1b`), 1 SKIP; the divergent example reports high confidence.
+
+**Plan C**
+
+- *C2:* the sign test stays on the HMM segment and only the reported span is trimmed.
+ Recomputing it on a span chosen from the data would lower p-values in the caller that
+ produces every false region. A segment touching the first or last window is searched to
+ the alignment end.
+- *C5:* the rule is `min(--phi-window, sites // 10)`. With it, B7's "not testable" case is
+ unreachable at the default. PHI power on the Jaya benchmark is at most 14 % under every
+ rule tried; the window is not the limit there (few biallelic sites survive the filter).
+- *C6:* "read the second token" is wrong for multi-word clades (`clade 2 wild-type`,
+ `Clade VI`); the label is everything after the accession, and whether a header is
+ structured is carried in a sidecar by provenance, not inferred. `run_hybrids.py` types
+ pools from the tree and never calls the CLI's selection path, so it cannot gate C6.
+- *C1:* depends on C2 (without trimming, an active HMM raises the merged breakpoint error
+ on near-identical panels). Cost measured: 1/20 near-identical clonal runs gain an HMM
+ false region at `--min-methods 1`, 0/20 at 2.
+- *C8:* characterised. All eight false regions are HMM switches between two tips of the
+ query's own clade (about 2.4 % apart, not pooled by clustering), q mostly 0.01-0.04 --
+ consistent with a sign test run on a segment selected from the same sites. The plan ends
+ in a decision record; no default is changed.
+- Not run for any plan: `validation/run_hybrids.py` (it needs network). It is an executor
+ gate in plans B and C.
+
+**Landing order.** A, then B, then C. B edits `recomb/run.py` in five tasks and C edits it
+again; B's clustering paragraph in `docs/detection-methods.md` describes the C1 defect as
+current behaviour and C1 rewrites it; B's reassort `--dataset` check calls
+`core.io.read_fasta`, which A10 changes.
+
+## Not covered by the audit
+
+Cactus and real progressiveMauve (not installed); the live network paths of `detect`,
+`fill-references` and `build-panel`; whether live Nextclade trees contain mutations whose
+reference state is a gap (`nextclade.py:45` skips them); large-input memory behaviour.
+Smaller discover-side items recorded for a later pass: the stop test watches the worst gap
+while the search targets the longest gaps (`iterate.py:325-347`); the RefSeq-to-complete
+broadening does not handle zero RefSeq genomes; the Nextclade alias fallback maps RSV-B to
+the RSV-A dataset and HIV-2 to HIV-1; curated-away siblings are re-downloaded each round.
diff --git a/src/tessera/aligners/cactus.py b/src/tessera/aligners/cactus.py
index 2b9aa1c..f34e22e 100644
--- a/src/tessera/aligners/cactus.py
+++ b/src/tessera/aligners/cactus.py
@@ -81,7 +81,14 @@ def align(
# dotted query filename would otherwise be unfindable downstream).
name_map = {_sample_name(g): g.stem for g in genomes}
# Drop the Minigraph-Cactus backbone pseudo-genome so it is not a taxon.
- maf_to_fasta(maf, ref_name, msa, name_map=name_map, exclude={"_MINIGRAPH_"})
+ # `expected` gives a genome that hal2maf placed in no block an all-gap row
+ # instead of leaving it out. The backbone contig order is not passed: the
+ # sequence names hal2maf emits for a pangenome HAL have not been verified
+ # against the input FASTA, so the layout stays as the MAF implies.
+ maf_to_fasta(
+ maf, ref_name, msa, name_map=name_map, exclude={"_MINIGRAPH_"},
+ expected=[g.stem for g in genomes], logger=logger,
+ )
return AlignResult(msa_fasta=msa, native_format=hal)
diff --git a/src/tessera/aligners/mafft.py b/src/tessera/aligners/mafft.py
index 28aa705..9311780 100644
--- a/src/tessera/aligners/mafft.py
+++ b/src/tessera/aligners/mafft.py
@@ -7,8 +7,10 @@
``mafft --addfragments --keeplength ``: ``--keeplength``
keeps the output in reference coordinates (insertions relative to the backbone
are dropped) and ``--addfragments`` is designed for fragmented assemblies, so a
-multi-contig query is handled cleanly. A genome's contigs are then merged into a
-single reference-anchored row.
+multi-contig query is handled cleanly. ``--adjustdirection`` lets MAFFT reverse-
+complement an added sequence that is on the opposite strand to the backbone (it renames
+such a record ``_R_``; the merge below is positional and ignores names). A genome's
+contigs are then merged into a single reference-anchored row.
"""
from __future__ import annotations
@@ -76,8 +78,8 @@ def add_genome(genome: Path) -> tuple[str, str]:
aligned = out_dir / f"{genome.stem}.aln.fasta"
run_tool(
self.capabilities,
- ["mafft", "--thread", threads, "--keeplength", *tuning,
- "--addfragments", str(genome.resolve()), str(ref_fasta.resolve())],
+ ["mafft", "--thread", threads, "--keeplength", "--adjustdirection",
+ *tuning, "--addfragments", str(genome.resolve()), str(ref_fasta.resolve())],
logger=logger,
log_prefix=f"mafft:{genome.stem}",
stdout_path=aligned,
diff --git a/src/tessera/aligners/progressivemauve.py b/src/tessera/aligners/progressivemauve.py
index 8d75881..c0f8b5c 100644
--- a/src/tessera/aligners/progressivemauve.py
+++ b/src/tessera/aligners/progressivemauve.py
@@ -17,6 +17,7 @@
from ..converters.xmfa_to_fasta import xmfa_to_fasta
from ..core.binaries import BinarySpec
+from ..core.errors import OutputError
from ..core.executors import parallel_map
from ..core.io import normalize_reference, read_fasta, write_fasta_record
from ..core.plugins import ToolCapabilities
@@ -68,7 +69,7 @@ def align(
# resolved a yet-unexplained progressiveMauve error on some systems.
workers = 1 if params.flag("single") else params.threads
- def align_query(query: Path) -> Path:
+ def align_query(query: Path) -> tuple[str, Path]:
stem = query.stem
xmfa = xmfa_dir / f"{stem}.xmfa"
fa = xmfa_dir / f"{stem}.fa"
@@ -79,25 +80,37 @@ def align_query(query: Path) -> Path:
log_prefix=f"progressivemauve:{stem}",
)
xmfa_to_fasta(xmfa, ref_arg, 0, fa, reference_length=ref_length)
- return fa
+ return stem, fa
- per_query_fastas = parallel_map(align_query, queries, workers, logger=logger)
+ per_query = parallel_map(align_query, queries, workers, logger=logger)
msa = out_dir / "msa.fasta"
- _concatenate(per_query_fastas, reference, msa)
+ _concatenate(per_query, reference.stem, msa)
return AlignResult(msa_fasta=msa)
-def _concatenate(per_query_fastas: list[Path], reference: Path, out_path: Path) -> None:
- """Write the reference row once, then each query row; leaf names are stems."""
- ref_stem = reference.stem
- written_ref = False
+def _concatenate(
+ per_query: list[tuple[str, Path]], reference_label: str, out_path: Path
+) -> None:
+ """Write the reference row once, then one row per query, named by staged label.
+
+ Each per-query FASTA holds exactly two records, in a fixed order: the reference
+ projection, then the query's. The names inside those files are the paths
+ progressiveMauve echoed from its command line -- resolved symlink targets, which can
+ carry an unrecognised extension, whitespace, or a basename shared with another
+ genome -- so rows are identified by position and labelled from the staged file,
+ never from those names.
+ """
with open(out_path, "w") as out:
- for fa in per_query_fastas:
- for name, seq in read_fasta(fa):
- leaf = Path(name).stem
- if leaf == ref_stem:
- if written_ref:
- continue
- written_ref = True
- write_fasta_record(out, leaf, seq)
+ for i, (label, fa) in enumerate(per_query):
+ records = read_fasta(fa)
+ if len(records) != 2:
+ raise OutputError(
+ f"Expected a reference row and one query row in {fa} (the "
+ f"projection of '{label}' onto '{reference_label}'), found "
+ f"{len(records)} record(s). progressiveMauve's output is not a "
+ "pairwise alignment."
+ )
+ if i == 0:
+ write_fasta_record(out, reference_label, records[0][1])
+ write_fasta_record(out, label, records[1][1])
diff --git a/src/tessera/aligners/sibeliaz.py b/src/tessera/aligners/sibeliaz.py
index 0ad80be..7884c4c 100644
--- a/src/tessera/aligners/sibeliaz.py
+++ b/src/tessera/aligners/sibeliaz.py
@@ -16,7 +16,7 @@
from ..converters.maf_to_fasta import maf_to_fasta
from ..core.binaries import BinarySpec
from ..core.errors import OutputError, UserInputError
-from ..core.io import normalize_reference
+from ..core.io import normalize_reference, read_fasta
from ..core.plugins import ToolCapabilities
from ..core.process import run_tool
from .base import Aligner, AlignParams, AlignResult
@@ -112,7 +112,15 @@ def align(
# genome filenames; build the seqid -> genome-stem map for the converter.
name_map = _build_seqid_map(genomes)
msa = out_dir / "msa.fasta"
- maf_to_fasta(maf, reference.stem, msa, name_map=name_map)
+ # The MAF names only the backbone contigs that fall in a block, in no particular
+ # order, and only the genomes it placed. Hand the converter the backbone's own
+ # contig order and the full genome list so neither is inferred from the MAF.
+ maf_to_fasta(
+ maf, reference.stem, msa, name_map=name_map,
+ ref_contigs=[(seqid, len(seq)) for seqid, seq in read_fasta(reference)],
+ expected=[g.stem for g in genomes],
+ logger=logger,
+ )
return AlignResult(msa_fasta=msa, native_format=maf)
@@ -148,9 +156,16 @@ def _build_seqid_map(genomes) -> dict[str, str]:
for genome in genomes:
stem = genome.stem
with open(genome) as fo:
- for line in fo:
+ for lineno, line in enumerate(fo, start=1):
if line.startswith(">"):
- seqid = line[1:].split()[0]
+ tokens = line[1:].split()
+ if not tokens:
+ raise UserInputError(
+ f"{genome} has a record with no sequence ID (line {lineno}). "
+ "The sibeliaz backend identifies genomes by sequence ID, so "
+ "every record needs a name after '>'."
+ )
+ seqid = tokens[0]
owner = name_map.setdefault(seqid, stem)
if owner != stem:
# SibeliaZ names alignment rows by sequence ID alone, so two
diff --git a/src/tessera/cli/cmd_fill_references.py b/src/tessera/cli/cmd_fill_references.py
index bf7ece4..1849949 100644
--- a/src/tessera/cli/cmd_fill_references.py
+++ b/src/tessera/cli/cmd_fill_references.py
@@ -121,7 +121,9 @@ def fill_references(
),
curate: bool = typer.Option(
False, "--curate",
- help="Drop the query's siblings and dereplicate each round (needs skani/skDER).",
+ help="Drop the query's siblings and dereplicate before each alignment build: the "
+ "supplied collection, then each round's downloads (needs skani/skDER). "
+ "--reference, when given, is the curation backbone and is never removed.",
),
sibling_margin: float = typer.Option(
3.0, "--sibling-margin",
diff --git a/src/tessera/cli/cmd_find_references.py b/src/tessera/cli/cmd_find_references.py
index 7896bc0..c38217a 100644
--- a/src/tessera/cli/cmd_find_references.py
+++ b/src/tessera/cli/cmd_find_references.py
@@ -61,8 +61,9 @@ def find_references(
),
curate: bool = typer.Option(
False, "--curate",
- help="After download, drop the query's siblings and dereplicate (needs skani/skDER, "
- "--collection as the backbone source).",
+ help="After download, drop the query's siblings and near-duplicates among the "
+ "new downloads (needs skani/skDER, --collection as the backbone source). Genomes "
+ "already in the download directory are never removed.",
),
sibling_margin: float = typer.Option(
3.0, "--sibling-margin",
diff --git a/src/tessera/converters/maf_to_fasta.py b/src/tessera/converters/maf_to_fasta.py
index f5dd6d0..4525a6e 100644
--- a/src/tessera/converters/maf_to_fasta.py
+++ b/src/tessera/converters/maf_to_fasta.py
@@ -17,9 +17,12 @@
from __future__ import annotations
+import logging
+from collections.abc import Sequence
from dataclasses import dataclass
from pathlib import Path
+from ..core.errors import OutputError
from ..core.io import write_fasta_record
_COMPLEMENTS = bytes.maketrans(
@@ -53,6 +56,9 @@ def maf_to_fasta(
out_path: str | Path,
name_map: dict[str, str] | None = None,
exclude: set[str] | None = None,
+ ref_contigs: Sequence[tuple[str, int]] | None = None,
+ expected: Sequence[str] | None = None,
+ logger: logging.Logger | None = None,
) -> Path:
"""Project a MAF onto ``reference`` coordinates as an MSA-FASTA.
@@ -63,6 +69,18 @@ def maf_to_fasta(
``exclude`` drops genomes by label, e.g. ``{"_MINIGRAPH_"}`` to remove the
Minigraph-Cactus backbone pseudo-genome so it is not emitted as a taxon.
+
+ ``ref_contigs`` gives the backbone's contigs as ``(MAF source name, length)`` in
+ the order of its FASTA file. The MAF alone cannot supply this: it lists only
+ contigs that fall in some block, in no particular order. With it, the backbone is
+ laid out as its file is, and a contig no block covers keeps its (all-gap) columns
+ instead of vanishing and shifting everything after it. Without it the contigs seen
+ in the MAF are laid out in sorted-name order.
+
+ ``expected`` lists every genome label that should have a row. A genome the aligner
+ placed in no block is absent from the MAF; it is written as an all-gap row. Such a
+ genome, and one aligned only in blocks that do not include the backbone, is named in
+ a warning on ``logger``, so the panel is never silently smaller than the collection.
"""
maf_path = Path(maf_path)
out_path = Path(out_path)
@@ -81,36 +99,71 @@ def genome_of(src: str) -> str:
# each distinct reference source name is one contig (length = its src_size),
# placed at a cumulative offset. Collapsing them into a single contig's
# coordinate space would make later contigs overwrite earlier ones.
- ref_contigs: dict[str, int] = {} # source name -> contig length
+ seen_contigs: dict[str, int] = {} # source name -> contig length
species: set[str] = set()
+ placed: set[str] = set() # genomes sharing at least one block with the backbone
for block in blocks:
- for row in block:
- label = genome_of(row.name)
- species.add(label)
+ labels = [genome_of(row.name) for row in block]
+ species.update(labels)
+ if ref_key in labels:
+ placed.update(labels)
+ for row, label in zip(block, labels, strict=True):
if label == ref_key:
- ref_contigs.setdefault(row.name, row.src_size)
+ seen_contigs.setdefault(row.name, row.src_size)
species -= exclude
- if not ref_contigs:
+ if not seen_contigs:
raise ValueError(
f"MAF projection onto reference '{ref_key}' found no reference rows. "
f"Check that the reference label matches the MAF/name_map sequence names."
)
+ if ref_contigs is not None:
+ unknown = sorted(set(seen_contigs) - {name for name, _ in ref_contigs})
+ if unknown:
+ raise OutputError(
+ f"{maf_path} aligns backbone sequence(s) {', '.join(unknown)} that are "
+ f"not in the backbone '{ref_key}' as staged. The aligner's sequence "
+ "names do not match the input FASTA."
+ )
+ layout = list(ref_contigs)
+ else:
+ layout = sorted(seen_contigs.items())
ref_offsets: dict[str, int] = {}
ref_length = 0
- for name in sorted(ref_contigs):
+ for name, length in layout:
ref_offsets[name] = ref_length
- ref_length += ref_contigs[name]
-
+ ref_length += length
+
+ # A genome in no block at all, and one aligned only in blocks that lack the backbone,
+ # both project to an all-gap row; name both.
+ unplaced = sorted((set(expected or ()) | species) - placed - exclude - {ref_key})
+ if unplaced and logger is not None:
+ logger.warning(
+ "%d genome(s) share no alignment block with the backbone '%s' and are "
+ "written as all-gap rows: %s. They contribute nothing to the scan; they "
+ "may be too divergent for this aligner.",
+ len(unplaced), ref_key, ", ".join(unplaced),
+ )
species.discard(ref_key)
- ordered_species = [ref_key, *sorted(species)]
+ ordered_species = [ref_key, *sorted(species | set(unplaced))]
out: dict[str, bytearray] = {s: bytearray(b"-" * ref_length) for s in ordered_species}
for block in blocks:
ref_row = next((r for r in block if genome_of(r.name) == ref_key), None)
if ref_row is None:
continue
+ # Every row of a block has the same aligned width. A shorter one means the file
+ # was cut off mid-write (a full disk, a killed aligner); indexing into it below
+ # would fail with a bare IndexError several frames from the file's name.
+ for row in block:
+ if len(row.text) != len(ref_row.text):
+ raise OutputError(
+ f"Truncated alignment block in {maf_path}: row '{row.name}' has "
+ f"{len(row.text)} column(s), the backbone row '{ref_row.name}' has "
+ f"{len(ref_row.text)}. The aligner's output looks incomplete; check "
+ "that it finished."
+ )
contig_offset = ref_offsets[ref_row.name]
if ref_row.strand == "-":
# Reverse-complement the whole block into forward-reference orientation.
diff --git a/src/tessera/converters/mafft_merge.py b/src/tessera/converters/mafft_merge.py
index 3abbcbb..17eecc8 100644
--- a/src/tessera/converters/mafft_merge.py
+++ b/src/tessera/converters/mafft_merge.py
@@ -13,6 +13,7 @@
from pathlib import Path
+from ..core.errors import OutputError
from ..core.io import read_fasta
@@ -24,7 +25,10 @@ def merge_added_fragments(aligned_path: str | Path) -> tuple[str, str]:
"""
records = read_fasta(aligned_path)
if not records:
- raise ValueError(f"Empty MAFFT alignment: {aligned_path}")
+ raise OutputError(
+ f"MAFFT wrote an empty alignment at {aligned_path}. Check that mafft "
+ "completed and that the genome being added is a nucleotide FASTA."
+ )
reference_row = records[0][1]
width = len(reference_row)
merged = bytearray(b"-" * width)
diff --git a/src/tessera/converters/xmfa_to_fasta.py b/src/tessera/converters/xmfa_to_fasta.py
index 5821200..85bc44e 100644
--- a/src/tessera/converters/xmfa_to_fasta.py
+++ b/src/tessera/converters/xmfa_to_fasta.py
@@ -112,6 +112,13 @@ def xmfa_to_fasta(
seq[curr_pos : curr_pos + length_of_line] = stripped
curr_pos += length_of_line
+ if reference_name not in name2num:
+ listed = ", ".join(sorted(name2num)) or "none"
+ raise UserInputError(
+ f"{xmfa_path} does not list the reference {reference_name} among its "
+ f"sequence files (found: {listed}). The aligner's output does not belong "
+ "to this reference, or its header is incomplete."
+ )
reference_num = name2num[reference_name]
# Output width: the full reference length when known, otherwise the furthest
diff --git a/src/tessera/core/io.py b/src/tessera/core/io.py
index 62d524d..c99434a 100644
--- a/src/tessera/core/io.py
+++ b/src/tessera/core/io.py
@@ -11,10 +11,11 @@
import gzip
import logging
+import re
import shutil
-from collections.abc import Sequence
+from collections.abc import Iterable, Sequence
from pathlib import Path
-from typing import TextIO
+from typing import BinaryIO, TextIO
from .errors import UserInputError
@@ -34,20 +35,25 @@ def read_fasta(path: str | Path) -> list[tuple[str, str]]:
A ``.gz`` input is read through :mod:`gzip`: staging already accepts compressed
genomes, so the readers that look inside a query must accept them too.
+
+ Whitespace inside a sequence line is dropped: a trailing space or tab is not a
+ base, and kept it lengthens the sequence (the backbone row then no longer matches
+ the rows aligned to it). A header with no name -- ``>`` or ``> `` -- reads as an
+ unnamed record (``""``).
"""
records: list[tuple[str, str]] = []
name: str | None = None
seq: list[str] = []
with _open_text(path) as fo:
for line in fo:
- line = line.rstrip("\r\n")
if line.startswith(">"):
if name is not None:
records.append((name, "".join(seq)))
- name = line[1:].split()[0] if len(line) > 1 else ""
+ tokens = line[1:].split()
+ name = tokens[0] if tokens else ""
seq = []
else:
- seq.append(line)
+ seq.append("".join(line.split()))
if name is not None:
records.append((name, "".join(seq)))
return records
@@ -99,6 +105,16 @@ def strip_sequence_extension(name: str) -> str:
return name
+def safe_filename_stem(name: str, fallback: str = "sequence") -> str:
+ """A file or directory name derived from an untrusted label (a FASTA header).
+
+ Path separators and other punctuation become ``_``, and leading/trailing dots are
+ removed so the result can never be ``.`` or ``..`` or climb out of the directory it
+ is joined to. ``fallback`` is returned when nothing usable is left.
+ """
+ return re.sub(r"[^\w.-]+", "_", name).strip(".") or fallback
+
+
def collection_genomes(directory: Path) -> list[Path]:
"""The genome files of a collection directory, sorted by name.
@@ -112,6 +128,53 @@ def collection_genomes(directory: Path) -> list[Path]:
)
+# Written beside a working collection when Tessera creates it, so a later run can tell
+# its own scratch directory from one the user happened to name the same.
+_WORKING_COPY_MARKER = ".tessera-working-collection"
+# Files a run leaves beside its working copy. Output directories written before the
+# marker existed are recognised by these, so re-running into one still works.
+_RUN_ARTEFACTS = ("round1.msa.fasta", "panel.msa.fasta", "fill_summary.tsv", "panel_lineages.tsv")
+
+
+def _clear_working_copy(dest: Path) -> None:
+ """Remove a previous run's working collection at ``dest``.
+
+ The working copy is cleared at the start of every run. That is only safe for a
+ directory Tessera made: ``/collection`` is also a natural place for a person
+ to keep their own genomes, and clearing that would destroy them. An existing,
+ non-empty directory is removed only when a previous run left its marker (or, for
+ output written by an older release, its other files) beside it; otherwise refuse.
+ """
+ if not dest.exists():
+ return
+ parent = dest.parent
+ ours = (parent / _WORKING_COPY_MARKER).exists() or any(
+ (parent / name).exists() for name in _RUN_ARTEFACTS
+ )
+ if not ours and (not dest.is_dir() or any(dest.iterdir())):
+ raise UserInputError(
+ f"{dest} already exists and was not created by Tessera. A run clears that "
+ "directory to hold its working copy of the references, which would delete "
+ "what is in it. Move it elsewhere or choose a different output directory."
+ )
+ shutil.rmtree(dest)
+
+
+def _mark_working_copy(dest: Path) -> None:
+ (dest.parent / _WORKING_COPY_MARKER).write_text(
+ "The collection/ directory here is Tessera's working copy; it is cleared and "
+ "rebuilt at the start of every run.\n"
+ )
+
+
+def new_working_collection(dest: Path) -> None:
+ """Start an empty working collection at ``dest``, clearing a previous run's."""
+ dest = Path(dest)
+ _clear_working_copy(dest)
+ dest.mkdir(parents=True)
+ _mark_working_copy(dest)
+
+
def copy_collection(source: Path, dest: Path) -> None:
"""Replace ``dest`` with a fresh copy of the collection at ``source``.
@@ -129,9 +192,9 @@ def copy_collection(source: Path, dest: Path) -> None:
f"copy ({dest}), which is cleared at the start of every run. Choose a "
"different output directory, or point --collection at a copy elsewhere."
)
- if dest.exists():
- shutil.rmtree(dest)
+ _clear_working_copy(dest)
shutil.copytree(source, dest)
+ _mark_working_copy(dest)
def _require_fasta(source: Path) -> None:
@@ -153,14 +216,58 @@ def _require_fasta(source: Path) -> None:
)
+_WHITESPACE = re.compile(rb"\s")
+
+
+def _has_sequence_whitespace(source: Path) -> bool:
+ """True when a plain FASTA carries whitespace inside a sequence line.
+
+ A trailing space or tab, a space within the line, or a carriage return (CRLF line
+ endings). Blank lines and the header's own spaces do not count.
+ """
+ with open(source, "rb") as fo:
+ for line in fo:
+ if line.startswith(b">"):
+ continue
+ body = line[:-1] if line.endswith(b"\n") else line
+ if body and _WHITESPACE.search(body):
+ return True
+ return False
+
+
+def _write_clean(src: Iterable[bytes], dst: BinaryIO) -> None:
+ """Copy a FASTA, dropping whitespace from sequence lines (and blank lines)."""
+ for line in src:
+ if line.startswith(b">"):
+ dst.write(line.rstrip(b"\r\n") + b"\n")
+ else:
+ body = b"".join(line.split())
+ if body:
+ dst.write(body + b"\n")
+
+
def _stage_one(source: Path, target_dir: Path, logger: logging.Logger) -> Path:
- """Place one genome into ``target_dir`` as ``.fasta``; return the path."""
+ """Place one genome into ``target_dir`` as ``.fasta``; return the path.
+
+ The staged file is what the aligner reads, while Tessera reads the same genome
+ through :func:`read_fasta`, which ignores whitespace in sequence lines. The two
+ must agree on the genome's length, and not every aligner ignores a trailing space
+ (minimap2 counts it as a base), so a genome that carries such whitespace is staged
+ as a cleaned copy rather than a link to the original.
+ """
label = strip_sequence_extension(source.name)
target = target_dir / f"{label}.fasta"
if source.name.endswith(".gz"):
logger.debug("Decompressing %s -> %s", source, target)
with gzip.open(source, "rb") as src, open(target, "wb") as dst:
- shutil.copyfileobj(src, dst)
+ _write_clean(src, dst)
+ elif _has_sequence_whitespace(source):
+ logger.warning(
+ "%s has whitespace inside its sequence lines (trailing spaces, tabs or "
+ "Windows line endings); aligning a cleaned copy.", source,
+ )
+ with open(source, "rb") as src, open(target, "wb") as dst:
+ _write_clean(src, dst)
else:
logger.debug("Linking %s -> %s", source, target)
target.symlink_to(source.resolve())
diff --git a/src/tessera/discover/iterate.py b/src/tessera/discover/iterate.py
index 223179b..05b547a 100644
--- a/src/tessera/discover/iterate.py
+++ b/src/tessera/discover/iterate.py
@@ -4,7 +4,9 @@
gaps, BLASTs the worst gaps against NCBI, and downloads the best new reference per
gap into the collection. The loop stops when the gaps close, when no new reference
can be found, when coverage stops improving (a stubborn residual is reported, not
-chased forever), or at ``max_rounds``.
+chased forever), or at ``max_rounds``. If the last round downloaded references, one
+further MSA is built from them, so the published panel always contains every
+reference the run reports.
Because every round rebuilds the alignment, this needs an aligner binary and
Entrez Direct, and it contacts NCBI over the network.
@@ -23,6 +25,7 @@
from ..core.io import (
collection_genomes,
copy_collection,
+ new_working_collection,
read_fasta,
strip_sequence_extension,
)
@@ -209,9 +212,7 @@ def fill_references(params: FillParams, logger: logging.Logger) -> list[RoundRes
if params.collection is not None:
copy_collection(params.collection, collection)
else:
- if collection.exists():
- shutil.rmtree(collection)
- collection.mkdir(parents=True)
+ new_working_collection(collection)
exclude = {_base_accession(e) for e in params.exclude}
# The query's own GenBank record matches itself almost perfectly and would be
@@ -279,6 +280,76 @@ def fill_references(params: FillParams, logger: logging.Logger) -> list[RoundRes
return trace
+def _curation_backbone(
+ params: FillParams, collection: Path, logger: logging.Logger
+) -> Path | None:
+ """The genome curation is anchored on: the query's closest whole-genome relative.
+
+ ``--reference`` is deliberately not the anchor. It is the alignment's coordinate
+ reference and may be far from the query; the sibling test is relative to its anchor,
+ so anchored there every genome closer to the query than the reference would be
+ dropped as a sibling. The reference is protected from removal instead
+ (:func:`_user_reference`).
+ """
+ backbone = pick_backbone(
+ params.query, collection_genomes(collection), af_min=params.af_min, logger=logger
+ )
+ if backbone is not None:
+ logger.info("Curation backbone (the query's closest whole-genome relative): %s",
+ strip_sequence_extension(backbone.name))
+ return backbone
+
+
+def _user_reference(params: FillParams, collection: Path) -> list[Path]:
+ """The genome named by ``--reference`` in ``collection`` (``[]`` if none or absent).
+
+ The next MSA build resolves the same name, so curation must never remove it. An
+ absent reference is left for ``build_msa`` to report with its own message.
+ """
+ if not params.reference:
+ return []
+ wanted = strip_sequence_extension(Path(params.reference).name)
+ return [
+ genome for genome in collection_genomes(collection)
+ if strip_sequence_extension(genome.name) == wanted
+ ]
+
+
+def _curate_round(
+ params: FillParams,
+ collection: Path,
+ backbone: Path | None,
+ panel_rows: dict[str, dict],
+ logger: logging.Logger,
+) -> None:
+ """Drop siblings and near-duplicates from ``collection`` in place, recording roles."""
+ if backbone is None or not backbone.exists():
+ backbone = _curation_backbone(params, collection, logger)
+ if backbone is None:
+ return
+ curation = curate_collection_dir(
+ params.query, collection, backbone,
+ ani_margin=params.sibling_margin, af_min=params.af_min,
+ derep_ani=params.derep_ani, protect=_user_reference(params, collection),
+ logger=logger,
+ )
+ for row in curation.table:
+ panel_rows[row["genome"]] = row
+ remaining = len(collection_genomes(collection))
+ if remaining < 2:
+ logger.warning(
+ "Curation left %d reference(s) in the panel; detection needs at least 2. "
+ "Every other genome was classed as a sibling of the query or a near-duplicate "
+ "of the backbone (see panel_lineages.tsv). If the panel is meant to be this "
+ "close to the query, run without --curate.",
+ remaining,
+ )
+
+
+def _labels(collection: Path) -> set[str]:
+ return {strip_sequence_extension(p.name) for p in collection_genomes(collection)}
+
+
def _grow_collection(
params: FillParams,
collection: Path,
@@ -286,13 +357,22 @@ def _grow_collection(
exclude: set[str],
logger: logging.Logger,
) -> tuple[list[RoundResult], dict[str, dict], Path | None]:
- """Run the build -> scan -> find -> download (-> curate) loop.
+ """Run the (curate ->) build -> scan -> find -> download loop.
Each round rebuilds the MSA from the (growing) collection, scans for coverage
- gaps, downloads the best new reference per gap, and (when ``--curate``)
- dereplicates. Stops when the gaps close, when coverage stops improving, when
- no new reference is found, or at ``max_rounds``. Mutates ``collection``;
- returns the per-round trace, the curated panel-role table, and the last MSA.
+ gaps and downloads the best new reference per gap. Stops when the gaps close, when
+ coverage stops improving, when no new reference is found, or at ``max_rounds``.
+
+ With ``--curate`` the collection is dereplicated and cleared of the query's siblings
+ *before* a build whenever it holds genomes that have not been through that filter: a
+ collection the user supplied (before round 1), and anything a round downloaded
+ (before the next build). A freshly seeded collection is not curated before round 1 --
+ seeding applies its own sibling filter.
+
+ Whatever the exit, the alignment returned was built from the collection as it stands:
+ if the last round downloaded references, one more MSA (``final.msa.fasta``) is built
+ from them without a further search. Mutates ``collection``; returns the per-round
+ trace, the curated panel-role table, and the last MSA.
"""
cov = CoverageParams.with_defaults(
params.window_size, floor=params.coverage_floor, rel_drop=params.coverage_rel_drop,
@@ -303,8 +383,25 @@ def _grow_collection(
# dropped genome keeps the role it had when removed, even after later rounds).
panel_rows: dict[str, dict] = {}
last_msa: Path | None = None
+ built_from: set[str] = set() # labels the last MSA was built from
prev_worst: float | None = None
+ # Curation owed before the next build, and the backbone to anchor it on.
+ curate_pending = params.curate and params.collection is not None
+ backbone: Path | None = None
+ pending_round: RoundResult | None = None # the round whose downloads await curation
+
+ def curate_if_pending() -> None:
+ nonlocal curate_pending, pending_round
+ if not curate_pending:
+ return
+ _curate_round(params, collection, backbone, panel_rows, logger)
+ if pending_round is not None:
+ kept = _labels(collection)
+ pending_round.added = [a for a in pending_round.added if a in kept]
+ curate_pending, pending_round = False, None
+
for rnd in range(1, params.max_rounds + 1):
+ curate_if_pending()
msa = params.output / f"round{rnd}.msa.fasta"
logger.info("=== Round %d: building MSA from %d reference(s) ===",
rnd, len(collection_genomes(collection)))
@@ -316,6 +413,7 @@ def _grow_collection(
logger,
)
last_msa = msa
+ built_from = _labels(collection)
result = compute_similarity(
str(msa), query_label,
window_size=params.window_size, window_step=params.window_step,
@@ -353,32 +451,34 @@ def _grow_collection(
)
# Pick the backbone from the pre-download (curated, sibling-free) collection
# so a freshly-downloaded sibling cannot be mistaken for it.
- backbone = None
if params.curate:
- backbone = pick_backbone(
- params.query, collection_genomes(collection),
- af_min=params.af_min, logger=logger,
- )
+ backbone = _curation_backbone(params, collection, logger)
downloaded = _download(candidates, collection, logger)
rr.added = [c.hit.accession for c in downloaded]
if not downloaded:
logger.info("Stopping: no new references available to add.")
break
- if params.curate and backbone is not None:
- curation = curate_collection_dir(
- params.query, collection, backbone,
- ani_margin=params.sibling_margin, af_min=params.af_min,
- derep_ani=params.derep_ani, logger=logger,
- )
- for row in curation.table:
- panel_rows[row["genome"]] = row
- dropped = {c.hit.accession for c in downloaded} - {
- strip_sequence_extension(p.name) for p in collection_genomes(collection)
- }
- rr.added = [a for a in rr.added if a not in dropped]
+ curate_pending, pending_round = params.curate, rr
else:
logger.info("Reached the maximum of %d round(s).", params.max_rounds)
+ # The last round may have downloaded references after its own MSA was built. Curate
+ # them like any other round's, then align whatever the collection now holds, so the
+ # published panel and the reported reference count describe the same set.
+ curate_if_pending()
+ if last_msa is not None and _labels(collection) != built_from:
+ final_msa = params.output / "final.msa.fasta"
+ logger.info("=== Final build: aligning %d reference(s) (no further search) ===",
+ len(collection_genomes(collection)))
+ build_msa(
+ MsaParams(
+ query=params.query, collection=collection, output=final_msa,
+ aligner=params.aligner, reference=params.reference, threads=params.threads,
+ ),
+ logger,
+ )
+ last_msa = final_msa
+
return trace, panel_rows, last_msa
diff --git a/src/tessera/discover/panel.py b/src/tessera/discover/panel.py
index 3f87d10..822a1a5 100644
--- a/src/tessera/discover/panel.py
+++ b/src/tessera/discover/panel.py
@@ -26,6 +26,7 @@
import logging
import shutil
import tempfile
+from collections.abc import Iterable
from dataclasses import dataclass, field
from pathlib import Path
@@ -266,22 +267,48 @@ def curate_collection_dir(
ani_margin: float = DEFAULT_SIBLING_MARGIN,
af_min: float = DEFAULT_AF_MIN,
derep_ani: float = DEFAULT_DEREP_ANI,
+ protect: Iterable[Path] = (),
logger: logging.Logger,
) -> CurationResult:
"""Curate the genomes in ``collection`` in place: drop siblings/redundant files.
Runs :func:`curate_panel`, then deletes the dropped genome files from disk so a
subsequent MSA rebuild sees only the diverse, sibling-free panel.
+
+ ``protect`` lists files that must stay on disk whatever the comparison says -- the
+ genomes a directory held before this run added to it. They still take part in the
+ sibling and redundancy comparison (a new download that duplicates one is dropped),
+ and a protected genome that would have been removed is reported with the role
+ ``sibling-kept`` / ``redundant-kept`` instead, so the table says what is on disk.
"""
genomes = collection_genomes(collection)
result = curate_panel(
query_fasta, genomes, backbone,
ani_margin=ani_margin, af_min=af_min, derep_ani=derep_ani, logger=logger,
)
+ protected = {Path(p).resolve() for p in protect}
keep = {p.resolve() for p in result.kept} | {backbone.resolve()}
+ spared: list[Path] = []
for g in genomes:
- if g.resolve() not in keep:
+ if g.resolve() in keep:
+ continue
+ if g.resolve() in protected:
+ spared.append(g)
+ else:
g.unlink()
+ if spared:
+ spared_labels = {strip_sequence_extension(g.name) for g in spared}
+ for row in result.table:
+ if row["genome"] in spared_labels:
+ row["role"] = row["role"].replace("-dropped", "-kept")
+ spared_set = {g.resolve() for g in spared}
+ result.kept = [*result.kept, *spared]
+ result.siblings = [g for g in result.siblings if g.resolve() not in spared_set]
+ result.redundant = [g for g in result.redundant if g.resolve() not in spared_set]
+ logger.info(
+ "Kept %d pre-existing genome(s) that curation would have dropped: %s.",
+ len(spared), ", ".join(sorted(spared_labels)),
+ )
return result
@@ -372,10 +399,15 @@ def write_panel_tsv(
_ROLE_LABEL = {
"backbone": "backbone",
"representative": "kept (parent)",
+ "sibling-kept": "kept (sibling, pre-existing)",
+ "redundant-kept": "kept (redundant, pre-existing)",
"sibling-dropped": "dropped (sibling)",
"redundant-dropped": "dropped (redundant)",
}
-_ROLE_ORDER = {"backbone": 0, "representative": 1, "sibling-dropped": 2, "redundant-dropped": 3}
+_ROLE_ORDER = {
+ "backbone": 0, "representative": 1, "sibling-kept": 2, "redundant-kept": 3,
+ "sibling-dropped": 4, "redundant-dropped": 5,
+}
def panel_table_html(table: list[dict], lineage_map: LineageMap | None = None) -> str:
diff --git a/src/tessera/discover/run.py b/src/tessera/discover/run.py
index 8019581..6b68fee 100644
--- a/src/tessera/discover/run.py
+++ b/src/tessera/discover/run.py
@@ -104,10 +104,16 @@ def find_references(params: FindRefParams, logger: logging.Logger) -> list[Candi
_print_candidates(candidates, logger)
if params.download is not None:
+ # What the download directory held before this run. It is often the user's own
+ # collection (`--download ` is the documented way to grow one in
+ # place), so curation below may remove only what this run adds to it.
+ preexisting = (
+ collection_genomes(params.download) if params.download.is_dir() else []
+ )
downloaded = _download(candidates, params.download, logger)
_write_downloaded(params.output, downloaded, logger)
if params.curate and downloaded:
- _curate_download(params, query_label, query_row, logger)
+ _curate_download(params, query_label, query_row, preexisting, logger)
elif candidates:
logger.info(
"Re-run with --download to add the new references, "
@@ -287,13 +293,18 @@ def _download(
def _curate_download(
- params: FindRefParams, query_label: str, query_row: str, logger: logging.Logger
+ params: FindRefParams, query_label: str, query_row: str,
+ preexisting: list[Path], logger: logging.Logger,
) -> None:
- """Drop the query's siblings and dereplicate the download directory in place.
+ """Drop the query's siblings and near-duplicates among this run's downloads.
The backbone (the query's whole-genome anchor) is chosen from the existing
``--collection``, so a freshly-downloaded sibling cannot be mistaken for it. The
query is reconstructed from its (de-gapped) MSA row, as skani needs a FASTA.
+
+ ``preexisting`` are the files the download directory held before this run; they are
+ compared against but never deleted, so pointing ``--download`` at the collection
+ itself cannot remove the user's own genomes.
"""
from .panel import (
curate_collection_dir,
@@ -311,18 +322,26 @@ def _curate_download(
return
qfasta = params.output / "query.degapped.fasta"
qfasta.write_text(f">{query_label}\n{query_row.replace('-', '')}\n")
+ assert params.download is not None # only called from the `download is not None` branch
+ # When --download is the collection itself, the collection now also holds this run's
+ # downloads; leave them out so a downloaded sibling cannot become the backbone.
+ held_before = {p.resolve() for p in preexisting}
+ new_files = {
+ p.resolve() for p in collection_genomes(params.download)
+ if p.resolve() not in held_before
+ }
backbone = pick_backbone(
- qfasta, collection_genomes(params.collection),
+ qfasta,
+ [g for g in collection_genomes(params.collection) if g.resolve() not in new_files],
af_min=params.af_min, logger=logger,
)
if backbone is None:
logger.warning("Could not determine a backbone from --collection; skipping curation.")
return
- assert params.download is not None # only called from the `download is not None` branch
curation = curate_collection_dir(
qfasta, params.download, backbone,
ani_margin=params.sibling_margin, af_min=params.af_min,
- derep_ani=params.derep_ani, logger=logger,
+ derep_ani=params.derep_ani, protect=preexisting, logger=logger,
)
write_panel_tsv(params.output / "panel_lineages.tsv", curation.table, logger)
diff --git a/src/tessera/reassort/assign.py b/src/tessera/reassort/assign.py
index 7a57912..04ba6b0 100644
--- a/src/tessera/reassort/assign.py
+++ b/src/tessera/reassort/assign.py
@@ -10,22 +10,28 @@
from __future__ import annotations
import logging
-import re
import tempfile
from dataclasses import dataclass, field
from pathlib import Path
from ..core.cache import nextclade_cache
from ..core.errors import ToolExecutionError, UserInputError
-from ..core.io import read_fasta, strip_sequence_extension, write_fasta_record
+from ..core.io import (
+ read_fasta,
+ safe_filename_stem,
+ strip_sequence_extension,
+ write_fasta_record,
+)
from ..discover.nextclade import NON_CLADE_MARKERS, build_pool, resolve_dataset
from ..discover.panel import skani_available, skani_query_ani
from ..recomb.typing import first_header
from .constellation import DEFAULT_MARGIN, ParentGroup, call_constellation
-from .scan import SegmentScan, require_aligner, scan_segment
+from .scan import SegmentScan, require_aligner, scan_segment, unique_scan_dirs
DEFAULT_ANI_FLOOR = 80.0 # a segment below this ANI to every tip is left unassigned
-MIN_AF = 0.5 # a tip aligning over less than this fraction of the segment is ignored
+# A tip aligning over less than this share of the segment is ignored. In PERCENT (0-100),
+# the unit skani reports Align_fraction_query in and `skani_query_ani` returns.
+MIN_AF = 50.0
TOP_K = 25 # internal cap on candidate strains kept per segment
@@ -85,7 +91,7 @@ def _type_segment(seg, seq, overrides, ani_floor, margin, email, cache_dir, tmp,
error, or any unexpected error) propagates so it surfaces rather than reading as unassigned."""
# The segment name comes from a FASTA header, which may hold path separators
# ("A/California/07/2009|HA") or climb out of the temp directory ("../x").
- safe = re.sub(r"[^\w.-]+", "_", strip_sequence_extension(seg)).strip(".") or "segment"
+ safe = safe_filename_stem(strip_sequence_extension(seg), fallback="segment")
seg_fasta = Path(tmp) / f"{safe}.fasta"
with open(seg_fasta, "w") as fo:
write_fasta_record(fo, seg, seq)
@@ -184,13 +190,15 @@ def assign_segments(
result.pair_notes = call.pair_notes
if scan_segments:
+ scan_dirs = unique_scan_dirs([s.segment for s in result.segments])
for s in result.segments:
if s.status == "assigned":
seq, dataset = to_scan[s.segment]
assert output is not None # guarded above when scan_segments is set
result.scans.append(scan_segment(
s.segment, seq, dataset, output,
- aligner=aligner, cache_dir=cache_dir, logger=logger))
+ aligner=aligner, cache_dir=cache_dir, logger=logger,
+ dir_name=scan_dirs[s.segment]))
else:
result.scans.append(SegmentScan(s.segment, False, False, 0, "unassigned"))
return result
diff --git a/src/tessera/reassort/scan.py b/src/tessera/reassort/scan.py
index 44f492c..d5138fd 100644
--- a/src/tessera/reassort/scan.py
+++ b/src/tessera/reassort/scan.py
@@ -10,14 +10,13 @@
from __future__ import annotations
import logging
-import re
import shutil
from dataclasses import dataclass
from pathlib import Path
from ..core.cache import nextclade_cache
from ..core.errors import UserInputError
-from ..core.io import strip_sequence_extension, write_fasta_record
+from ..core.io import safe_filename_stem, strip_sequence_extension, write_fasta_record
from ..discover.nextclade import NON_CLADE_MARKERS, build_pool
from ..msa.build import MsaParams, build_msa
from ..recomb.regions import DEFAULT_METHODS
@@ -87,13 +86,40 @@ def _summarize_regions(path: Path) -> tuple[int, bool]:
return n, n > 0
+def unique_scan_dirs(segments: list[str]) -> dict[str, str]:
+ """Map each segment name to its own scan-directory name.
+
+ Segment names are FASTA headers, so two different names can sanitise to the same
+ string (``seg/1`` and ``seg_1``); their scans would then overwrite each other. Later
+ duplicates get a numeric suffix, in input order.
+ """
+ out: dict[str, str] = {}
+ used: set[str] = set()
+ for segment in segments:
+ base = safe_filename_stem(segment, fallback="segment")
+ name, n = base, 1
+ while name in used:
+ n += 1
+ name = f"{base}_{n}"
+ used.add(name)
+ out[segment] = name
+ return out
+
+
def scan_segment(
segment: str, seq: str, dataset, out_dir: Path, *,
aligner: str, cache_dir: Path | None, logger: logging.Logger,
+ dir_name: str | None = None,
) -> SegmentScan:
"""Scan one segment for intragenic recombination. Never raises: a failure is recorded as
- ``scanned=False`` so the caller can continue with the other segments."""
- seg_name = re.sub(r"[^\w.-]+", "_", segment)
+ ``scanned=False`` so the caller can continue with the other segments.
+
+ ``dir_name`` is the scan directory under ``out_dir`` (see :func:`unique_scan_dirs`);
+ by default it is derived from the segment name. Either way it is a sanitised single
+ path component: the segment name comes from a FASTA header and must not be able to
+ name ``..`` or the output root itself.
+ """
+ seg_name = safe_filename_stem(dir_name or segment, fallback="segment")
seg_dir = out_dir / seg_name
try:
pool = build_pool(
diff --git a/tests/integration/test_mafft_strand.py b/tests/integration/test_mafft_strand.py
new file mode 100644
index 0000000..6ab4fe5
--- /dev/null
+++ b/tests/integration/test_mafft_strand.py
@@ -0,0 +1,83 @@
+"""MAFFT backend on reverse-strand input (needs the mafft binary).
+
+An assembly carries no promise about strand: a whole genome, or one contig of a draft,
+may be the reverse complement of the backbone. The row must still be a real alignment.
+"""
+
+from __future__ import annotations
+
+import logging
+import random
+import shutil
+from pathlib import Path
+
+import pytest
+
+from tessera.aligners.base import AlignParams
+from tessera.aligners.mafft import MafftAligner
+from tessera.core.io import read_fasta
+
+_LOG = logging.getLogger("tessera.test")
+
+pytestmark = pytest.mark.requires_binary
+
+_COMPLEMENT = str.maketrans("ACGT", "TGCA")
+
+
+def _revcomp(seq: str) -> str:
+ return seq.translate(_COMPLEMENT)[::-1]
+
+
+def _genomes() -> tuple[str, str]:
+ """A 6 kb random reference and a query 3 % diverged from it."""
+ rng = random.Random(1)
+ ref = "".join(rng.choice("ACGT") for _ in range(6000))
+ qry = "".join(
+ rng.choice([b for b in "ACGT" if b != c]) if rng.random() < 0.03 else c for c in ref
+ )
+ return ref, qry
+
+
+def _identity_by_third(ref_row: str, qry_row: str) -> list[float]:
+ out = []
+ for lo in range(0, len(ref_row), 2000):
+ pairs = [
+ (a, b) for a, b in zip(ref_row[lo:lo + 2000].upper(), qry_row[lo:lo + 2000].upper(),
+ strict=True)
+ if a in "ACGT" and b in "ACGT"
+ ]
+ out.append(sum(a == b for a, b in pairs) / len(pairs) if pairs else 0.0)
+ return out
+
+
+def _align(tmp_path: Path, query_fasta: str) -> tuple[str, str]:
+ ref, _ = _genomes()
+ (tmp_path / "ref.fasta").write_text(f">ref\n{ref}\n")
+ (tmp_path / "qry.fasta").write_text(query_fasta)
+ result = MafftAligner().align(
+ [tmp_path / "ref.fasta", tmp_path / "qry.fasta"], tmp_path / "ref.fasta",
+ tmp_path / "out", AlignParams(threads=1), _LOG,
+ )
+ msa = dict(read_fasta(result.msa_fasta))
+ return msa["ref"], msa["qry"]
+
+
+@pytest.fixture(autouse=True)
+def _needs_mafft() -> None:
+ if shutil.which("mafft") is None:
+ pytest.skip("mafft not installed")
+
+
+def test_reverse_complemented_query_aligns(tmp_path: Path) -> None:
+ _, qry = _genomes()
+ ref_row, qry_row = _align(tmp_path, f">q\n{_revcomp(qry)}\n")
+ assert len(ref_row) == len(qry_row) == 6000
+ assert all(identity > 0.95 for identity in _identity_by_third(ref_row, qry_row))
+
+
+def test_draft_query_with_one_reversed_contig_aligns(tmp_path: Path) -> None:
+ _, qry = _genomes()
+ contigs = f">c1\n{qry[:2000]}\n>c2\n{_revcomp(qry[2000:4000])}\n>c3\n{qry[4000:]}\n"
+ ref_row, qry_row = _align(tmp_path, contigs)
+ assert len(ref_row) == len(qry_row) == 6000
+ assert all(identity > 0.95 for identity in _identity_by_third(ref_row, qry_row))
diff --git a/tests/integration/test_whitespace_genomes.py b/tests/integration/test_whitespace_genomes.py
new file mode 100644
index 0000000..3fd9e09
--- /dev/null
+++ b/tests/integration/test_whitespace_genomes.py
@@ -0,0 +1,58 @@
+"""Genomes with whitespace in their sequence lines, through each pairwise aligner.
+
+Needs the aligner binaries. A reference saved with a trailing space on every line used
+to give a ragged MSA with mafft and, once the reader ignored the spaces, a silently
+shifted one with minimap2 (which counts them as bases).
+"""
+
+from __future__ import annotations
+
+import logging
+import random
+import shutil
+from pathlib import Path
+
+import pytest
+
+from tessera.core.io import read_fasta
+from tessera.msa.build import MsaParams, build_msa
+
+_LOG = logging.getLogger("tessera.test")
+
+pytestmark = pytest.mark.requires_binary
+
+
+@pytest.mark.parametrize("aligner", ["mafft", "minimap2"])
+def test_reference_with_trailing_spaces_aligns_in_register(aligner: str, tmp_path: Path) -> None:
+ if shutil.which(aligner) is None:
+ pytest.skip(f"{aligner} not installed")
+ rng = random.Random(1)
+ ref = "".join(rng.choice("ACGT") for _ in range(6000))
+ qry = "".join(
+ rng.choice([b for b in "ACGT" if b != c]) if rng.random() < 0.03 else c for c in ref
+ )
+ other = "".join(
+ rng.choice([b for b in "ACGT" if b != c]) if rng.random() < 0.05 else c for c in ref
+ )
+ collection = tmp_path / "collection"
+ collection.mkdir()
+ spaced = "".join(ref[i:i + 60] + " \n" for i in range(0, len(ref), 60))
+ (collection / "ref.fasta").write_text(f">r\n{spaced}")
+ (collection / "other.fasta").write_text(f">o\n{other}\n")
+ query = tmp_path / "query.fasta"
+ query.write_text(f">q\n{qry}\n")
+
+ msa = build_msa(
+ MsaParams(query=query, collection=collection, output=tmp_path / "msa.fasta",
+ aligner=aligner, reference="ref", threads=1),
+ _LOG,
+ )
+ rows = dict(read_fasta(msa))
+ assert {len(row) for row in rows.values()} == {6000}
+ tail_pairs = [
+ (a, b) for a, b in zip(rows["ref"][3000:].upper(), rows["query"][3000:].upper(),
+ strict=True)
+ if a in "ACGT" and b in "ACGT"
+ ]
+ assert len(tail_pairs) > 2500
+ assert sum(a == b for a, b in tail_pairs) / len(tail_pairs) > 0.95
diff --git a/tests/unit/test_aligner_backends.py b/tests/unit/test_aligner_backends.py
index 76ca2c6..237de54 100644
--- a/tests/unit/test_aligner_backends.py
+++ b/tests/unit/test_aligner_backends.py
@@ -201,3 +201,135 @@ def fake_run(caps, cmd, **kw):
assert set(msa) == {"ref", "qry"}
assert msa["ref"] == "ACGTACGTACGT"
assert msa["qry"] == "ACGTACGTACGT"
+
+
+def test_mafft_adjusts_direction_of_added_sequences(monkeypatch, tmp_path: Path) -> None:
+ """A draft contig or a whole genome may be on the opposite strand to the backbone.
+ Without --adjustdirection MAFFT aligns it as given and the row is noise."""
+ captured: dict[str, list[str]] = {}
+
+ def fake_run(caps, cmd, **kw):
+ captured["cmd"] = [str(c) for c in cmd]
+ raise RuntimeError("stop after capture")
+
+ monkeypatch.setattr(mafft_mod, "run_tool", fake_run)
+ ref, qry = _two_genomes(tmp_path)
+ with pytest.raises(RuntimeError):
+ mafft_mod.MafftAligner().align([ref, qry], ref, tmp_path / "out",
+ AlignParams(threads=1), _LOG)
+
+ cmd = captured["cmd"]
+ assert "--adjustdirection" in cmd
+ # The option must precede --addfragments, whose two arguments are positional.
+ assert cmd.index("--adjustdirection") < cmd.index("--addfragments")
+
+
+def test_merge_added_fragments_ignores_reversed_name_prefix(tmp_path: Path) -> None:
+ # `mafft --adjustdirection` renames a record it reverse-complemented to `_R_`.
+ # The merge is positional (first record = reference, the rest = contigs), so the
+ # renamed contig must still be merged.
+ aligned = tmp_path / "a.fasta"
+ aligned.write_text(
+ ">ref\nACGTACGT\n"
+ ">contig1\nAC------\n"
+ ">_R_contig2\n----ACGT\n"
+ )
+ _, merged = merge_added_fragments(aligned)
+ assert merged == "AC--ACGT"
+
+
+# --- progressiveMauve row names come from the staged labels ----------------------
+def _fake_mauve(caps, cmd, **kw):
+ """Write the XMFA the way progressiveMauve does: sequence names are the paths it
+ was given on the command line (which the adapter passes resolved)."""
+ cmd = [str(c) for c in cmd]
+ xmfa = Path(cmd[cmd.index("--output") + 1])
+ ref_path, query_path = cmd[-2], cmd[-1]
+ ref_seq = "".join(
+ line for line in Path(ref_path).read_text().splitlines() if not line.startswith(">")
+ )
+ qry_seq = "".join(
+ line for line in Path(query_path).read_text().splitlines() if not line.startswith(">")
+ )
+ xmfa.write_text(
+ f"#Sequence1File\t{ref_path}\n#Sequence2File\t{query_path}\n"
+ f"> 1:1-8 + {ref_path}\n{ref_seq}\n> 2:1-8 + {query_path}\n{qry_seq}\n=\n"
+ )
+ return ""
+
+
+def _mauve_rows(monkeypatch, staged: list[Path], out_dir: Path) -> list[tuple[str, str]]:
+ monkeypatch.setattr(pm_mod, "run_tool", _fake_mauve)
+ result = pm_mod.ProgressiveMauveAligner().align(
+ staged, staged[0], out_dir, AlignParams(threads=1), _LOG
+ )
+ rows: list[tuple[str, str]] = []
+ for line in result.msa_fasta.read_text().splitlines():
+ if line.startswith(">"):
+ rows.append((line[1:], ""))
+ else:
+ rows[-1] = (rows[-1][0], rows[-1][1] + line)
+ return rows
+
+
+def test_progressivemauve_names_rows_by_staged_label_not_symlink_target(
+ monkeypatch, tmp_path: Path
+) -> None:
+ # Staged genomes are symlinks named .fasta. Two of them point at files that
+ # are both called genome.fna, and one target is named like the backbone's label.
+ store = tmp_path / "store"
+ for sub in ("x", "y", "z"):
+ (store / sub).mkdir(parents=True)
+ (store / "x" / "genome.fna").write_text(">r\nACGTACGT\n")
+ (store / "y" / "genome.fna").write_text(">b\nACGTACGA\n")
+ (store / "z" / "aRef.fna").write_text(">c\nTTTTTTTT\n")
+ stage = tmp_path / "stage"
+ stage.mkdir()
+ staged = []
+ for label, target in (("aRef", "x/genome.fna"), ("panelB", "y/genome.fna"),
+ ("panelC", "z/aRef.fna")):
+ link = stage / f"{label}.fasta"
+ link.symlink_to((store / target).resolve())
+ staged.append(link)
+
+ rows = _mauve_rows(monkeypatch, staged, tmp_path / "out")
+
+ assert rows == [("aRef", "ACGTACGT"), ("panelB", "ACGTACGA"), ("panelC", "TTTTTTTT")]
+
+
+def test_progressivemauve_tolerates_whitespace_in_the_path(monkeypatch, tmp_path: Path) -> None:
+ stage = tmp_path / "dir with space"
+ stage.mkdir()
+ staged = []
+ for label, seq in (("ref", "ACGTACGT"), ("qryA", "ACGTACGA"), ("qryB", "ACGTACGC")):
+ path = stage / f"{label}.fasta"
+ path.write_text(f">{label}\n{seq}\n")
+ staged.append(path)
+
+ rows = _mauve_rows(monkeypatch, staged, tmp_path / "out")
+
+ assert [name for name, _ in rows] == ["ref", "qryA", "qryB"]
+
+
+def test_progressivemauve_rejects_a_projection_that_is_not_pairwise(
+ monkeypatch, tmp_path: Path
+) -> None:
+ from tessera.core.errors import OutputError
+
+ def one_sequence_xmfa(caps, cmd, **kw):
+ cmd = [str(c) for c in cmd]
+ xmfa = Path(cmd[cmd.index("--output") + 1])
+ ref_path = cmd[-2]
+ xmfa.write_text(f"#Sequence1File\t{ref_path}\n> 1:1-8 + {ref_path}\nACGTACGT\n=\n")
+ return ""
+
+ monkeypatch.setattr(pm_mod, "run_tool", one_sequence_xmfa)
+ genomes = []
+ for label in ("ref", "qryA", "qryB"):
+ path = tmp_path / f"{label}.fasta"
+ path.write_text(f">{label}\nACGTACGT\n")
+ genomes.append(path)
+ with pytest.raises(OutputError, match=r"qryA\.fa.*found 1 record"):
+ pm_mod.ProgressiveMauveAligner().align(
+ genomes, genomes[0], tmp_path / "out", AlignParams(threads=1), _LOG
+ )
diff --git a/tests/unit/test_aligner_params.py b/tests/unit/test_aligner_params.py
index 06bfc4b..c50f946 100644
--- a/tests/unit/test_aligner_params.py
+++ b/tests/unit/test_aligner_params.py
@@ -98,3 +98,56 @@ def fake_run(caps, cmd, **kw):
assert cmd[cmd.index("-k") + 1] == "15"
assert cmd[cmd.index("-f") + 1] == "12"
assert cmd[cmd.index("-t") + 1] == "4"
+
+
+def test_sibeliaz_lays_out_the_backbone_as_its_file_and_keeps_every_genome(
+ monkeypatch, tmp_path: Path, caplog
+) -> None:
+ # Backbone: three contigs in the order contig_2, contig_10, contig_3. SibeliaZ gives
+ # contig_10 no block (nothing is homologous to it) and gives the panel genome
+ # `far` no block at all.
+ ref = tmp_path / "ref.fasta"
+ ref.write_text(">contig_2\nAAAA\n>contig_10\nGGGG\n>contig_3\nCCCC\n")
+ qry = tmp_path / "qry.fasta"
+ qry.write_text(">q1\nAAAACCCC\n")
+ far = tmp_path / "far.fasta"
+ far.write_text(">f1\nTTTTTTTT\n")
+
+ def fake_run(caps, cmd, **kw):
+ out_dir = Path(cmd[cmd.index("-o") + 1])
+ (out_dir / "alignment.maf").write_text(
+ "a\n"
+ "s contig_3 0 4 + 4 CCCC\n"
+ "s q1 4 4 + 8 CCCC\n"
+ "\n"
+ "a\n"
+ "s contig_2 0 4 + 4 AAAA\n"
+ "s q1 0 4 + 8 AAAA\n"
+ )
+ return ""
+
+ monkeypatch.setattr(sz, "run_tool", fake_run)
+ monkeypatch.setattr(sz, "_sibeliaz_invocation", lambda out_dir, logger: ["sibeliaz"])
+
+ # Not under the "tessera" logger: the CLI tests switch its propagation off, and
+ # caplog only sees records that reach the root logger.
+ log = logging.getLogger("sibeliaz_adapter_test")
+ with caplog.at_level(logging.WARNING, logger="sibeliaz_adapter_test"):
+ result = sz.SibeliazAligner().align(
+ [ref, qry, far], ref, tmp_path / "out", AlignParams(threads=1), log
+ )
+
+ rows: dict[str, str] = {}
+ name = ""
+ for line in result.msa_fasta.read_text().splitlines():
+ if line.startswith(">"):
+ name = line[1:]
+ rows[name] = ""
+ else:
+ rows[name] += line
+ assert rows == {
+ "ref": "AAAA----CCCC", # file order; the uncovered contig keeps its columns
+ "far": "------------", # placed in no block, but still a row
+ "qry": "AAAA----CCCC",
+ }
+ assert "far" in caplog.text
diff --git a/tests/unit/test_converters.py b/tests/unit/test_converters.py
index 407eae0..e2daacf 100644
--- a/tests/unit/test_converters.py
+++ b/tests/unit/test_converters.py
@@ -98,3 +98,102 @@ def test_maf_name_map_relabels_to_real_stems(tmp_path: Path) -> None:
seqs = _read_fasta(out)
assert set(seqs) == {"cowpox_KC813504", "sample.1"} # dotted stem restored
assert seqs["sample.1"] == "ACGA"
+
+
+# --- multi-contig backbone layout and unplaced genomes ---------------------------
+
+def test_maf_backbone_contigs_follow_file_order_when_given(tmp_path: Path) -> None:
+ # The backbone FASTA lists contig_2 then contig_10. Sorted by name, contig_10 comes
+ # first, which puts the backbone in a different order from its own file (and from
+ # the minimap2 / mafft backends on the same input).
+ maf = tmp_path / "order.maf"
+ maf.write_text(
+ "a\n"
+ "s ref.contig_2 0 4 + 4 AAAA\n"
+ "s qry.x 0 4 + 8 AAAA\n"
+ "\n"
+ "a\n"
+ "s ref.contig_10 0 4 + 4 CCCC\n"
+ "s qry.x 4 4 + 8 CCCC\n"
+ )
+ seqs = _read_fasta(maf_to_fasta(
+ maf, "ref", tmp_path / "msa.fasta",
+ ref_contigs=[("ref.contig_2", 4), ("ref.contig_10", 4)],
+ ))
+ assert seqs["ref"] == "AAAACCCC"
+ assert seqs["qry"] == "AAAACCCC"
+
+
+def test_maf_backbone_contig_without_a_block_keeps_its_columns(tmp_path: Path) -> None:
+ # c2 has no homolog in any genome, so no MAF block mentions it. It is still part of
+ # the backbone: dropping it would make the MSA 8 wide and shift c3 from 8-11 to 4-7.
+ maf = tmp_path / "missing.maf"
+ maf.write_text(
+ "a\n"
+ "s ref.c1 0 4 + 4 AAAA\n"
+ "s qry.x 0 4 + 8 AAAA\n"
+ "\n"
+ "a\n"
+ "s ref.c3 0 4 + 4 CCCC\n"
+ "s qry.x 4 4 + 8 CCCC\n"
+ )
+ seqs = _read_fasta(maf_to_fasta(
+ maf, "ref", tmp_path / "msa.fasta",
+ ref_contigs=[("ref.c1", 4), ("ref.c2", 4), ("ref.c3", 4)],
+ ))
+ assert seqs["ref"] == "AAAA----CCCC"
+ assert seqs["qry"] == "AAAA----CCCC"
+
+
+def test_maf_rejects_a_backbone_contig_it_was_not_told_about(tmp_path: Path) -> None:
+ import pytest
+
+ from tessera.core.errors import OutputError
+
+ maf = tmp_path / "extra.maf"
+ maf.write_text("a\ns ref.c9 0 4 + 4 AAAA\ns qry.x 0 4 + 4 AAAA\n")
+ with pytest.raises(OutputError, match="ref.c9"):
+ maf_to_fasta(maf, "ref", tmp_path / "msa.fasta", ref_contigs=[("ref.c1", 4)])
+
+
+def test_maf_genome_without_any_block_gets_an_all_gap_row(tmp_path: Path, caplog) -> None:
+ # A panel member too divergent to be placed in any block used to have no row at all,
+ # so the panel was silently one genome smaller than the collection.
+ import logging
+
+ maf = tmp_path / "vanish.maf"
+ maf.write_text("a\ns r1 0 4 + 4 AAAA\ns q1 0 4 + 4 AAAT\n")
+ name_map = {"r1": "ref", "q1": "qry", "p1": "divergent_panel"}
+ # Not under the "tessera" logger: the CLI tests switch its propagation off, and
+ # caplog only sees records that reach the root logger.
+ log = logging.getLogger("maf_converter_test")
+ with caplog.at_level(logging.WARNING, logger="maf_converter_test"):
+ seqs = _read_fasta(maf_to_fasta(
+ maf, "ref", tmp_path / "msa.fasta", name_map=name_map,
+ expected=["ref", "qry", "divergent_panel"], logger=log,
+ ))
+ assert seqs == {"ref": "AAAA", "divergent_panel": "----", "qry": "AAAT"}
+ assert "divergent_panel" in caplog.text
+
+
+def test_maf_genome_aligned_only_away_from_the_backbone_is_named(tmp_path: Path, caplog) -> None:
+ # SibeliaZ also emits blocks between non-backbone genomes. A genome seen only in such
+ # blocks projects to an all-gap row just like one seen in no block, and must be named
+ # in the same warning.
+ import logging
+
+ maf = tmp_path / "offref.maf"
+ maf.write_text(
+ "a\ns r1 0 4 + 4 AAAA\ns a1 0 4 + 4 AAAT\n\n"
+ "a\ns a1 0 4 + 4 AAAT\ns b1 0 4 + 4 AAAT\n"
+ )
+ name_map = {"r1": "ref", "a1": "A", "b1": "B"}
+ log = logging.getLogger("maf_converter_test")
+ with caplog.at_level(logging.WARNING, logger="maf_converter_test"):
+ seqs = _read_fasta(maf_to_fasta(
+ maf, "ref", tmp_path / "msa.fasta", name_map=name_map,
+ expected=["ref", "A", "B", "C"], logger=log,
+ ))
+ assert seqs == {"ref": "AAAA", "A": "AAAT", "B": "----", "C": "----"}
+ assert "2 genome(s)" in caplog.text
+ assert "B, C" in caplog.text
diff --git a/tests/unit/test_discover.py b/tests/unit/test_discover.py
index a32af20..f8ffe95 100644
--- a/tests/unit/test_discover.py
+++ b/tests/unit/test_discover.py
@@ -150,6 +150,109 @@ def fake_efetch(accession, collection_dir, logger):
assert manifest[1].split("\t")[0] == "NEW123"
+def test_curate_after_download_keeps_the_users_own_genomes(monkeypatch, tmp_path, logger):
+ """`find-references -c C --download C --curate` is the documented way to grow a
+ collection in place. Curation used to prune C itself, deleting the user's files."""
+ from tessera.discover import panel
+
+ coll = tmp_path / "collection"
+ coll.mkdir()
+ for name in ("refA", "refB", "userC"):
+ (coll / f"{name}.fasta").write_text(f">{name}\nACGT\n")
+
+ def fake_blast(seq, *, max_hits, logger, email=None, cache_dir=None):
+ return [
+ Hit("NEW123", "novel donor virus", 90.0, 95.0, 1e-40),
+ Hit("NEWSIB", "a relative of the query", 97.0, 96.0, 1e-60),
+ ]
+
+ def fake_efetch(accession, collection_dir, logger):
+ path = collection_dir / f"{accession}.fasta"
+ path.write_text(f">{accession}\nACGT\n")
+ return path
+
+ def fake_ani(query, refs, logger):
+ # refA is the backbone; refB and userC are its whole-genome twins (would be
+ # dropped as siblings); NEWSIB is a downloaded sibling; NEW123 a regional donor.
+ table = {
+ "refA": (97.0, 95.0), "refB": (96.0, 95.0), "userC": (96.8, 96.0),
+ "NEW123": (88.0, 30.0), "NEWSIB": (99.0, 98.0),
+ }
+ return {r: table[r.name.split(".")[0]] for r in refs}
+
+ monkeypatch.setattr(discover_run, "blast_subsequence", fake_blast)
+ monkeypatch.setattr(discover_run, "efetch_available", lambda: True)
+ monkeypatch.setattr(discover_run, "efetch_fasta", fake_efetch)
+ monkeypatch.setattr(panel, "skani_available", lambda: True)
+ monkeypatch.setattr(panel, "skder_available", lambda: False)
+ monkeypatch.setattr(panel, "skani_query_ani", fake_ani)
+
+ find_references(
+ FindRefParams(
+ msa=_msa(tmp_path), query="q", output=tmp_path / "out",
+ window_size=60, window_step=30, top_gaps=1,
+ collection=coll, download=coll, curate=True,
+ ),
+ logger,
+ )
+
+ assert sorted(p.name for p in coll.iterdir()) == [
+ "NEW123.fasta", "refA.fasta", "refB.fasta", "userC.fasta",
+ ]
+ rows = dict(
+ line.split("\t")[:2]
+ for line in (tmp_path / "out" / "panel_lineages.tsv").read_text().splitlines()[1:]
+ )
+ assert rows["userC"] == "sibling-kept"
+ assert rows["NEWSIB"] == "sibling-dropped"
+
+
+def test_curate_after_download_into_a_new_directory(monkeypatch, tmp_path, logger):
+ """--download may name a directory that does not exist yet. Nothing in it predates
+ the run, so curation is free to drop a downloaded sibling."""
+ from tessera.discover import panel
+
+ coll = tmp_path / "collection"
+ coll.mkdir()
+ (coll / "refA.fasta").write_text(">refA\nACGT\n")
+ fresh = tmp_path / "downloads"
+
+ def fake_blast(seq, *, max_hits, logger, email=None, cache_dir=None):
+ return [
+ Hit("NEW123", "novel donor virus", 90.0, 95.0, 1e-40),
+ Hit("NEWSIB", "a relative of the query", 97.0, 96.0, 1e-60),
+ ]
+
+ def fake_efetch(accession, collection_dir, logger):
+ collection_dir.mkdir(parents=True, exist_ok=True)
+ path = collection_dir / f"{accession}.fasta"
+ path.write_text(f">{accession}\nACGT\n")
+ return path
+
+ def fake_ani(query, refs, logger):
+ table = {"refA": (92.0, 95.0), "NEW123": (88.0, 30.0), "NEWSIB": (99.0, 98.0)}
+ return {r: table[r.name.split(".")[0]] for r in refs}
+
+ monkeypatch.setattr(discover_run, "blast_subsequence", fake_blast)
+ monkeypatch.setattr(discover_run, "efetch_available", lambda: True)
+ monkeypatch.setattr(discover_run, "efetch_fasta", fake_efetch)
+ monkeypatch.setattr(panel, "skani_available", lambda: True)
+ monkeypatch.setattr(panel, "skder_available", lambda: False)
+ monkeypatch.setattr(panel, "skani_query_ani", fake_ani)
+
+ find_references(
+ FindRefParams(
+ msa=_msa(tmp_path), query="q", output=tmp_path / "out",
+ window_size=60, window_step=30, top_gaps=1,
+ collection=coll, download=fresh, curate=True,
+ ),
+ logger,
+ )
+
+ assert sorted(p.name for p in fresh.iterdir()) == ["NEW123.fasta"]
+ assert sorted(p.name for p in coll.iterdir()) == ["refA.fasta"]
+
+
def test_download_without_efetch_is_a_clear_error(monkeypatch, tmp_path, logger):
monkeypatch.setattr(discover_run, "efetch_available", lambda: False)
from tessera.core.errors import UserInputError
diff --git a/tests/unit/test_input_safety.py b/tests/unit/test_input_safety.py
index 110fbc0..498f8ad 100644
--- a/tests/unit/test_input_safety.py
+++ b/tests/unit/test_input_safety.py
@@ -51,8 +51,10 @@ def test_copy_collection_refuses_an_overlapping_source(tmp_path: Path, relative:
def test_copy_collection_replaces_a_previous_working_copy(tmp_path: Path) -> None:
+ stale = _collection(tmp_path / "earlier", names=("stale",))
source = _collection(tmp_path / "coll")
- dest = _collection(tmp_path / "out" / "collection", names=("stale",))
+ dest = tmp_path / "out" / "collection"
+ copy_collection(stale, dest) # an earlier run's working copy
copy_collection(source, dest)
assert sorted(p.name for p in dest.iterdir()) == ["refA.fasta", "refB.fasta"]
@@ -319,3 +321,101 @@ def fake_build_msa(params, logger):
assert (out / "panel.msa.fasta").exists()
assert provenance_path(out / "panel.msa.fasta").read_text() == '{"aligner": "mafft"}\n'
+
+
+# --- malformed records and tool output ---------------------------------------
+
+def test_read_fasta_drops_whitespace_inside_sequence_lines(tmp_path: Path) -> None:
+ """A trailing space or tab is not a base. Kept, it widened the backbone row and made
+ the mafft MSA ragged (reference 6100 columns, query 6000)."""
+ path = tmp_path / "ws.fasta"
+ path.write_text(">ref desc\nACGT \nAC GT\t\n\nTTAA\n")
+ assert read_fasta(path) == [("ref", "ACGTACGTTTAA")]
+
+
+def test_read_fasta_accepts_a_header_with_no_name(tmp_path: Path) -> None:
+ # ">" alone already read as an unnamed record; "> " raised IndexError instead.
+ path = tmp_path / "blank.fasta"
+ path.write_text("> \nACGT\n>\nTTAA\n")
+ assert read_fasta(path) == [("", "ACGT"), ("", "TTAA")]
+
+
+def test_sibeliaz_rejects_a_record_with_no_sequence_id(tmp_path: Path) -> None:
+ path = tmp_path / "g1.fasta"
+ path.write_text(">ok\nACGT\n> \nACGT\n")
+ with pytest.raises(UserInputError, match=r"g1\.fasta.*line 3"):
+ sibeliaz._build_seqid_map([path])
+
+
+def test_truncated_maf_row_is_reported_against_the_file(tmp_path: Path) -> None:
+ from tessera.converters.maf_to_fasta import maf_to_fasta
+
+ maf = tmp_path / "cut.maf"
+ maf.write_text("a\ns ref.c 0 8 + 8 ACGTACGT\ns qry.x 0 8 + 8 ACGT")
+ with pytest.raises(OutputError, match=r"cut\.maf.*qry\.x"):
+ maf_to_fasta(maf, "ref", tmp_path / "msa.fasta")
+
+
+def test_xmfa_without_the_reference_is_reported_against_the_file(tmp_path: Path) -> None:
+ from tessera.converters.xmfa_to_fasta import xmfa_to_fasta
+
+ xmfa = tmp_path / "other.xmfa"
+ xmfa.write_text(
+ "#Sequence1File\t/data/a.fasta\n#Sequence2File\t/data/b.fasta\n"
+ "> 1:1-4 + /data/a.fasta\nACGT\n> 2:1-4 + /data/b.fasta\nACGT\n=\n"
+ )
+ with pytest.raises(UserInputError, match=r"other\.xmfa.*/data/ref\.fasta"):
+ xmfa_to_fasta(xmfa, "/data/ref.fasta", 0, tmp_path / "out.fasta")
+
+
+def test_empty_mafft_output_is_a_tessera_error(tmp_path: Path) -> None:
+ from tessera.converters.mafft_merge import merge_added_fragments
+
+ empty = tmp_path / "empty.aln.fasta"
+ empty.write_text("")
+ with pytest.raises(OutputError, match=r"empty\.aln\.fasta"):
+ merge_added_fragments(empty)
+
+
+# --- the working copy is only cleared when Tessera made it ---------------------------
+
+def test_copy_collection_refuses_to_replace_a_directory_it_did_not_create(tmp_path: Path) -> None:
+ source = tmp_path / "src"
+ source.mkdir()
+ (source / "a.fasta").write_text(">a\nACGT\n")
+ dest = tmp_path / "project" / "collection"
+ dest.mkdir(parents=True)
+ (dest / "mine.fasta").write_text(">mine\nACGT\n")
+
+ with pytest.raises(UserInputError, match="was not created by Tessera"):
+ copy_collection(source, dest)
+
+ assert (dest / "mine.fasta").exists()
+
+
+def test_copy_collection_replaces_the_working_copy_of_an_older_release(tmp_path: Path) -> None:
+ # Output directories written before the marker existed are recognised by the files a
+ # run leaves beside the working copy, so re-running into one still works.
+ source = tmp_path / "src"
+ source.mkdir()
+ (source / "a.fasta").write_text(">a\nACGT\n")
+ out = tmp_path / "out"
+ (out / "collection").mkdir(parents=True)
+ (out / "collection" / "old.fasta").write_text(">old\nACGT\n")
+ (out / "round1.msa.fasta").write_text(">q\nACGT\n")
+
+ copy_collection(source, out / "collection")
+
+ assert [p.name for p in collection_genomes(out / "collection")] == ["a.fasta"]
+
+
+def test_copy_collection_accepts_an_empty_existing_directory(tmp_path: Path) -> None:
+ source = tmp_path / "src"
+ source.mkdir()
+ (source / "a.fasta").write_text(">a\nACGT\n")
+ dest = tmp_path / "out" / "collection"
+ dest.mkdir(parents=True)
+
+ copy_collection(source, dest)
+
+ assert [p.name for p in collection_genomes(dest)] == ["a.fasta"]
diff --git a/tests/unit/test_io.py b/tests/unit/test_io.py
index fc5b90a..3546254 100644
--- a/tests/unit/test_io.py
+++ b/tests/unit/test_io.py
@@ -96,3 +96,37 @@ def test_query_colliding_with_a_collection_member_is_rejected(tmp_path: Path) ->
with pytest.raises(UserInputError, match="share the label 'sample'"):
stage_genomes(query, coll, tmp_path / "staged", _LOG)
+
+
+def test_stage_genomes_cleans_whitespace_out_of_sequence_lines(tmp_path: Path) -> None:
+ """The aligner reads the staged file; Tessera reads the same genome through
+ read_fasta. If only one of them ignores a trailing space the two disagree about the
+ genome's length -- minimap2 counted the spaces and every later column was shifted."""
+ coll = tmp_path / "coll"
+ coll.mkdir()
+ (coll / "clean.fasta").write_text(">c desc\nACGT\nTTAA\n")
+ (coll / "spaces.fasta").write_text(">s desc\nACGT \nTT AA\t\n")
+ (coll / "crlf.fasta").write_bytes(b">w desc\r\nACGT\r\nTTAA\r\n")
+ query = tmp_path / "query.fasta"
+ query.write_text(">q\nACGT\n")
+
+ staged, _ = stage_genomes(query, coll, tmp_path / "stage", _LOG)
+ by_stem = {p.stem: p for p in staged}
+
+ assert by_stem["clean"].is_symlink() # untouched input is still linked, not copied
+ for stem, header in (("spaces", ">s desc"), ("crlf", ">w desc")):
+ assert not by_stem[stem].is_symlink()
+ assert by_stem[stem].read_text() == f"{header}\nACGT\nTTAA\n"
+ # The user's own files are not modified.
+ assert (coll / "spaces.fasta").read_text() == ">s desc\nACGT \nTT AA\t\n"
+
+
+def test_stage_genomes_cleans_a_gzipped_genome(tmp_path: Path) -> None:
+ coll = tmp_path / "coll"
+ coll.mkdir()
+ with gzip.open(coll / "z.fasta.gz", "wt") as fo:
+ fo.write(">z\nACGT \n\nTTAA\n")
+ query = tmp_path / "query.fasta"
+ query.write_text(">q\nACGT\n")
+ staged, _ = stage_genomes(query, coll, tmp_path / "stage", _LOG)
+ assert {p.stem: p for p in staged}["z"].read_text() == ">z\nACGT\nTTAA\n"
diff --git a/tests/unit/test_iterate.py b/tests/unit/test_iterate.py
index 5849dee..742d6a3 100644
--- a/tests/unit/test_iterate.py
+++ b/tests/unit/test_iterate.py
@@ -6,6 +6,7 @@
import pytest
+from tessera.core.errors import UserInputError
from tessera.discover import iterate
from tessera.discover.blast import Hit
from tessera.discover.iterate import FillParams, fill_references
@@ -504,3 +505,306 @@ def fake_select(params, genomes, logger):
)
assert captured["dataset"] == "nextstrain/sars-cov-2/XBB"
assert (out / "collection" / "REF1.fasta").exists()
+
+
+# --- the round structure: what is aligned, and when the panel is curated ----------
+
+def _recording_build(monkeypatch, builds):
+ """Replace build_msa with a stub that records which genomes each MSA was built from
+ and writes them into the file, so the published panel can be inspected."""
+ def fake_build(p, logger):
+ members = sorted(f.name for f in p.collection.iterdir())
+ builds.append((p.output.name, members))
+ p.output.write_text("".join(f">{m}\nACGT\n" for m in ["q", *members]))
+ monkeypatch.setattr(iterate, "build_msa", fake_build)
+
+
+def _one_new_hit_per_round(monkeypatch):
+ counter = iter(range(1, 50))
+
+ def collect(*a, **k):
+ return [Candidate(_gap(0.8), Hit(f"NEW{next(counter)}", "t", 90.0, 95.0, 1e-9), False)]
+
+ def download(cands, dest, logger):
+ for c in cands:
+ (dest / f"{c.hit.accession}.fasta").write_text(f">{c.hit.accession}\nACGT\n")
+ return cands
+
+ monkeypatch.setattr(iterate, "collect_candidates", collect)
+ monkeypatch.setattr(iterate, "_download", download)
+
+
+def _fake_skani(monkeypatch, table):
+ """Stub skani at the panel layer: ``table`` maps a genome label to (ANI %, AF %)."""
+ from tessera.discover import panel
+
+ def fake_ani(query, refs, logger):
+ return {r: table[r.name.split(".")[0]] for r in refs}
+
+ monkeypatch.setattr(panel, "skani_query_ani", fake_ani)
+ monkeypatch.setattr(panel, "skani_available", lambda: True)
+ monkeypatch.setattr(panel, "skder_available", lambda: False)
+ monkeypatch.setattr(iterate, "skani_available", lambda: True)
+
+
+def test_last_round_downloads_are_aligned(monkeypatch, tmp_path, logger):
+ """On a max_rounds exit the final round's downloads used to sit in collection/ and be
+ counted in the summary without ever reaching the published alignment."""
+ query, coll, out = _setup(tmp_path)
+ _common_mocks(monkeypatch, [([_gap(0.80)], 0.94), ([_gap(0.90)], 0.94)])
+ builds: list[tuple[str, list[str]]] = []
+ _recording_build(monkeypatch, builds)
+ _one_new_hit_per_round(monkeypatch)
+
+ trace = fill_references(
+ FillParams(query=query, collection=coll, output=out, max_rounds=2), logger
+ )
+
+ assert [(r.round, r.added) for r in trace] == [(1, ["NEW1"]), (2, ["NEW2"])]
+ final_members = ["NEW1.fasta", "NEW2.fasta", "refA.fasta"]
+ assert sorted(p.name for p in (out / "collection").iterdir()) == final_members
+ assert builds == [
+ ("round1.msa.fasta", ["refA.fasta"]),
+ ("round2.msa.fasta", ["NEW1.fasta", "refA.fasta"]),
+ ("final.msa.fasta", final_members),
+ ]
+ published = (out / "panel.msa.fasta").read_text()
+ assert ">NEW2.fasta" in published
+
+
+def test_no_extra_build_when_the_last_round_adds_nothing(monkeypatch, tmp_path, logger):
+ query, coll, out = _setup(tmp_path)
+ _common_mocks(monkeypatch, [([_gap(0.84)], 0.94), ([], 0.95)])
+ builds: list[tuple[str, list[str]]] = []
+ _recording_build(monkeypatch, builds)
+ _one_new_hit_per_round(monkeypatch)
+
+ fill_references(FillParams(query=query, collection=coll, output=out, max_rounds=2), logger)
+
+ assert [name for name, _ in builds] == ["round1.msa.fasta", "round2.msa.fasta"]
+ assert not (out / "final.msa.fasta").exists()
+
+
+def test_curate_runs_on_a_supplied_collection_before_the_first_build(
+ monkeypatch, tmp_path, logger
+):
+ """A whole-genome sibling in the starting collection leaves no coverage gap, so the
+ loop converges in round 1. Curation used to sit after the download step and never ran
+ -- in exactly the case it exists for."""
+ query, coll, out = _setup(tmp_path)
+ (coll / "SIBLING.fasta").write_text(">SIBLING\nACGT\n")
+ (coll / "PARENT.fasta").write_text(">PARENT\nACGT\n")
+ _common_mocks(monkeypatch, [([], 0.95)])
+ builds: list[tuple[str, list[str]]] = []
+ _recording_build(monkeypatch, builds)
+ # refA is picked as backbone only if SIBLING does not outrank it; make the sibling a
+ # whole-genome twin of refA (comparable ANI, full coverage) and PARENT a regional donor.
+ _fake_skani(monkeypatch, {
+ "refA": (97.0, 95.0), "SIBLING": (96.5, 95.0), "PARENT": (88.0, 30.0),
+ })
+ sections = {}
+ monkeypatch.setattr(
+ iterate, "run_recomb", lambda p, logger, **kw: sections.update(kw),
+ )
+
+ fill_references(FillParams(query=query, collection=coll, output=out, curate=True), logger)
+
+ assert builds == [("round1.msa.fasta", ["PARENT.fasta", "refA.fasta"])]
+ assert (coll / "SIBLING.fasta").exists() # the user's own collection is untouched
+ panel_tsv = (out / "panel_lineages.tsv").read_text()
+ assert "SIBLING\tsibling-dropped" in panel_tsv
+ assert "Reference panel" in [title for title, _ in sections["extra_sections"]]
+
+
+def test_curate_does_not_run_on_a_freshly_seeded_collection(monkeypatch, tmp_path, logger):
+ """Seeding already filters siblings (parents mode, pool selection). Curating the seed
+ again before round 1 would change what `detect` recruits, so it is not done."""
+ query = tmp_path / "q.fasta"
+ query.write_text(">q\n" + "ACGT" * 100 + "\n")
+ out = tmp_path / "out"
+ _common_mocks(monkeypatch, [([], 0.95)])
+ monkeypatch.setattr(iterate, "skani_available", lambda: True)
+
+ def seed(params, collection, query_records, exclude, logger):
+ for name in ("S1", "S2"):
+ (collection / f"{name}.fasta").write_text(f">{name}\nACGT\n")
+
+ monkeypatch.setattr(iterate, "_seed_collection", seed)
+ calls = []
+ monkeypatch.setattr(iterate, "curate_collection_dir",
+ lambda *a, **k: calls.append("curate"))
+ monkeypatch.setattr(iterate, "pick_backbone", lambda *a, **k: calls.append("backbone"))
+
+ fill_references(FillParams(query=query, collection=None, output=out, curate=True), logger)
+
+ assert calls == []
+
+
+def test_curate_never_removes_the_user_reference(monkeypatch, tmp_path, logger):
+ """`--curate --reference X` used to auto-pick another backbone, drop X as its twin,
+ and fail the next round with "Reference 'X' not found among the staged genomes".
+ X is the alignment's coordinate reference, not the sibling test's anchor: the anchor
+ stays the query's closest relative and X is kept whatever the comparison says."""
+ from tessera.core.io import select_reference
+
+ query, coll, out = _setup(tmp_path)
+ (coll / "MYREF.fasta").write_text(">MYREF\nACGT\n")
+ _common_mocks(monkeypatch, [([_gap(0.80)], 0.94), ([_gap(0.90)], 0.94)])
+ _one_new_hit_per_round(monkeypatch)
+ # Auto-picking would choose refA (highest ANI); MYREF is then its whole-genome twin.
+ _fake_skani(monkeypatch, {
+ "refA": (97.0, 95.0), "MYREF": (96.5, 95.0),
+ "NEW1": (88.0, 30.0), "NEW2": (87.0, 30.0),
+ })
+ builds: list[str] = []
+
+ def fake_build(p, logger): # resolve the reference the way the real build_msa does
+ select_reference(sorted(p.collection.iterdir()), p.query, False, p.reference)
+ builds.append(p.output.name)
+ p.output.write_text(">q\nACGT\n")
+
+ monkeypatch.setattr(iterate, "build_msa", fake_build)
+
+ fill_references(
+ FillParams(query=query, collection=coll, output=out, curate=True,
+ reference="MYREF", max_rounds=2),
+ logger,
+ )
+
+ assert builds == ["round1.msa.fasta", "round2.msa.fasta", "final.msa.fasta"]
+ assert (out / "collection" / "MYREF.fasta").exists()
+ rows = dict(
+ line.split("\t")[:2]
+ for line in (out / "panel_lineages.tsv").read_text().splitlines()[1:]
+ )
+ assert rows["refA"] == "backbone"
+ assert rows["MYREF"] == "sibling-kept"
+
+
+def test_last_round_downloads_are_curated_before_the_final_build(monkeypatch, tmp_path, logger):
+ """The final build sees a curated collection. Here the only download is a sibling, so
+ after curation the collection is what round 1 was built from and no rebuild is needed."""
+ query, coll, out = _setup(tmp_path)
+ _common_mocks(monkeypatch, [([_gap(0.80)], 0.94)])
+ builds: list[tuple[str, list[str]]] = []
+ _recording_build(monkeypatch, builds)
+ _one_new_hit_per_round(monkeypatch)
+ # The one download (NEW1) is a sibling: closer than the backbone, whole-genome.
+ _fake_skani(monkeypatch, {"refA": (92.0, 95.0), "NEW1": (99.0, 98.0)})
+
+ trace = fill_references(
+ FillParams(query=query, collection=coll, output=out, curate=True, max_rounds=1),
+ logger,
+ )
+
+ assert builds == [("round1.msa.fasta", ["refA.fasta"])]
+ assert not (out / "final.msa.fasta").exists()
+ assert sorted(p.name for p in (out / "collection").iterdir()) == ["refA.fasta"]
+ assert trace[0].added == [] # the sibling was downloaded, then curated away
+
+
+def test_curate_accepts_a_reference_given_with_its_extension(monkeypatch, tmp_path, logger):
+ query, coll, out = _setup(tmp_path)
+ (coll / "MYREF.fasta").write_text(">MYREF\nACGT\n")
+ _common_mocks(monkeypatch, [([], 0.95)])
+ _fake_skani(monkeypatch, {"refA": (97.0, 95.0), "MYREF": (96.5, 95.0)})
+
+ fill_references(
+ FillParams(query=query, collection=coll, output=out, curate=True,
+ reference="MYREF.fasta"),
+ logger,
+ )
+
+ rows = dict(
+ line.split("\t")[:2]
+ for line in (out / "panel_lineages.tsv").read_text().splitlines()[1:]
+ )
+ assert rows == {"MYREF": "sibling-kept", "refA": "backbone"}
+
+
+def test_curate_warns_when_it_leaves_too_few_references(monkeypatch, tmp_path, caplog):
+ """On a panel as close to the query as its own lineage, every genome but the backbone
+ is classed as a sibling. Detection cannot run on one reference; say why."""
+ import logging
+
+ query, coll, out = _setup(tmp_path)
+ for name in ("refB", "refC"):
+ (coll / f"{name}.fasta").write_text(f">{name}\nACGT\n")
+ _common_mocks(monkeypatch, [([], 0.95)])
+ _fake_skani(monkeypatch, {
+ "refA": (99.6, 99.0), "refB": (99.5, 99.0), "refC": (99.4, 99.0),
+ })
+ # Not under the "tessera" logger: the CLI tests switch its propagation off, and
+ # caplog only sees records that reach the root logger.
+ log = logging.getLogger("fill_loop_test")
+ with caplog.at_level(logging.WARNING, logger="fill_loop_test"):
+ fill_references(FillParams(query=query, collection=coll, output=out, curate=True), log)
+
+ assert sorted(p.name for p in (out / "collection").iterdir()) == ["refA.fasta"]
+ assert "Curation left 1 reference" in caplog.text
+ assert "without --curate" in caplog.text
+
+
+def test_curate_with_a_distant_reference_keeps_the_closer_genomes(monkeypatch, tmp_path, logger):
+ """The sibling test is relative to its anchor. Anchored on a distant coordinate
+ reference, every genome closer to the query than that reference looks like a sibling
+ and the panel loses its parental lineages. The anchor is the closest relative."""
+ query, coll, out = _setup(tmp_path)
+ for name in ("MYREF", "P1", "P2", "P3"):
+ (coll / f"{name}.fasta").write_text(f">{name}\nACGT\n")
+ _common_mocks(monkeypatch, [([], 0.95)])
+ builds: list[tuple[str, list[str]]] = []
+ _recording_build(monkeypatch, builds)
+ _fake_skani(monkeypatch, {
+ "MYREF": (85.0, 95.0), "refA": (95.0, 95.0),
+ "P1": (91.0, 95.0), "P2": (89.0, 95.0), "P3": (84.0, 95.0),
+ })
+
+ fill_references(
+ FillParams(query=query, collection=coll, output=out, curate=True, reference="MYREF"),
+ logger,
+ )
+
+ assert builds == [(
+ "round1.msa.fasta",
+ ["MYREF.fasta", "P1.fasta", "P2.fasta", "P3.fasta", "refA.fasta"],
+ )]
+
+
+def test_fresh_seed_refuses_to_clear_a_collection_directory_it_did_not_create(
+ monkeypatch, tmp_path, logger
+):
+ """`detect -o project/` clears `project/collection/` to seed it. If that directory
+ holds the user's own genomes and no earlier run made it, refuse, do not delete."""
+ query = tmp_path / "q.fasta"
+ query.write_text(">q\n" + "ACGT" * 100 + "\n")
+ out = tmp_path / "project"
+ (out / "collection").mkdir(parents=True)
+ precious = out / "collection" / "my_genome.fasta"
+ precious.write_text(">mine\nACGT\n")
+ _common_mocks(monkeypatch, [([], 0.95)])
+ monkeypatch.setattr(iterate, "_seed_collection", lambda *a, **k: None)
+
+ with pytest.raises(UserInputError, match="was not created by Tessera"):
+ fill_references(FillParams(query=query, collection=None, output=out), logger)
+
+ assert precious.exists()
+
+
+def test_fresh_seed_clears_its_own_working_copy_on_a_rerun(monkeypatch, tmp_path, logger):
+ query = tmp_path / "q.fasta"
+ query.write_text(">q\n" + "ACGT" * 100 + "\n")
+ out = tmp_path / "out"
+ _common_mocks(monkeypatch, [([], 0.95), ([], 0.95)])
+ names = iter(["S1", "S2"])
+
+ def seed(params, collection, query_records, exclude, logger):
+ name = next(names)
+ (collection / f"{name}.fasta").write_text(f">{name}\nACGT\n")
+ (collection / "other.fasta").write_text(">other\nACGT\n")
+
+ monkeypatch.setattr(iterate, "_seed_collection", seed)
+ for _ in range(2):
+ fill_references(FillParams(query=query, collection=None, output=out), logger)
+
+ assert sorted(p.name for p in (out / "collection").iterdir()) == ["S2.fasta", "other.fasta"]
diff --git a/tests/unit/test_panel.py b/tests/unit/test_panel.py
index aee1e3e..cf520e8 100644
--- a/tests/unit/test_panel.py
+++ b/tests/unit/test_panel.py
@@ -128,6 +128,40 @@ def test_curate_panel_table_roles(monkeypatch, tmp_path, logger):
assert roles == {"A1": "backbone", "AE_rel": "sibling-dropped", "C": "representative"}
+def test_curate_collection_dir_never_deletes_protected_files(monkeypatch, tmp_path, logger):
+ """Files the caller marks as protected take part in the comparison but stay on disk."""
+ coll = tmp_path / "coll"
+ coll.mkdir()
+ backbone, own_sibling, new_sibling, parent = _genomes(
+ coll, ["A1", "user_rel", "downloaded_rel", "C"]
+ )
+ ani = {
+ backbone: (92.0, 81.0), own_sibling: (97.0, 94.0),
+ new_sibling: (97.5, 95.0), parent: (89.0, 94.0),
+ }
+ monkeypatch.setattr(panel, "skani_available", lambda: True)
+ monkeypatch.setattr(panel, "skder_available", lambda: False)
+ monkeypatch.setattr(panel, "skani_query_ani", lambda *a, **k: ani)
+
+ result = panel.curate_collection_dir(
+ tmp_path / "q.fasta", coll, backbone, protect=[own_sibling], logger=logger,
+ )
+
+ assert sorted(p.name for p in coll.iterdir()) == ["A1.fasta", "C.fasta", "user_rel.fasta"]
+ roles = {r["genome"]: r["role"] for r in result.table}
+ assert roles == {
+ "A1": "backbone", "C": "representative",
+ "user_rel": "sibling-kept", "downloaded_rel": "sibling-dropped",
+ }
+ assert own_sibling in result.kept
+ assert result.siblings == [new_sibling]
+
+
+def test_panel_html_labels_a_kept_sibling():
+ html = panel.panel_table_html([_panel_row("user_rel", role="sibling-kept")])
+ assert "kept (sibling, pre-existing)" in html
+
+
# --- typed Lineage column (conditional) ----------------------------------------
def _panel_row(genome: str, role: str = "representative") -> dict:
diff --git a/tests/unit/test_reassort_assign.py b/tests/unit/test_reassort_assign.py
index 4289bec..94489be 100644
--- a/tests/unit/test_reassort_assign.py
+++ b/tests/unit/test_reassort_assign.py
@@ -51,8 +51,8 @@ def resolve(fasta, override, *, email, logger):
_patch(monkeypatch,
resolve=resolve,
tips_by_path={"HA_ds": [ha_tip], "NA_ds": [na_tip]},
- ani_by_path={"HA_pool": {ha_tip: (99.0, 0.10)}, # AF 0.10 < MIN_AF -> dropped
- "NA_pool": {na_tip: (99.0, 0.99)}})
+ ani_by_path={"HA_pool": {ha_tip: (99.0, 10.0)}, # AF 10 % < MIN_AF -> dropped
+ "NA_pool": {na_tip: (99.0, 99.0)}})
q = _write_query(tmp_path, [("HA", "HAxx"), ("NA", "NAyy")])
result = assign_segments(q, logger=LOG)
status = {s.segment: s.status for s in result.segments}
@@ -116,7 +116,7 @@ def resolve(fasta, override, *, email, logger):
def skani(q, refs, logger):
if refs[0].parent.name == "HA_pool":
raise ToolExecutionError(["skani", "dist"], 1, "sequence too short")
- return {na_tip: (99.0, 0.99)}
+ return {na_tip: (99.0, 99.0)}
def build_pool(ds, *, cache_dir, logger):
return [ha_tip] if ds.path == "HA_ds" else [na_tip]
@@ -152,7 +152,7 @@ def resolve(fasta, override, *, email, logger):
monkeypatch.setattr(assign, "nextclade_cache", lambda p, t, override=None: Path("/x"))
monkeypatch.setattr(assign, "build_pool", lambda ds, *, cache_dir, logger: [na_tip])
monkeypatch.setattr(assign, "skani_query_ani",
- lambda q, refs, logger: {na_tip: (99.0, 0.99)})
+ lambda q, refs, logger: {na_tip: (99.0, 99.0)})
monkeypatch.setattr(assign, "_clade_of_tip", lambda tip: "cladeX")
q = _write_query(tmp_path, [("HA", "HAxx"), ("NA", "NAyy")])
@@ -208,12 +208,13 @@ def resolve(fasta, override, *, email, logger):
_patch(monkeypatch, resolve=resolve,
tips_by_path={"HA_ds": [ha_tip], "NA_ds": [na_tip]},
- ani_by_path={"HA_pool": {ha_tip: (99.0, 0.99)},
- "NA_pool": {na_tip: (99.0, 0.99)}})
+ ani_by_path={"HA_pool": {ha_tip: (99.0, 99.0)},
+ "NA_pool": {na_tip: (99.0, 99.0)}})
monkeypatch.setattr(assign, "require_aligner", lambda aligner: None)
seen = []
- def fake_scan(segment, seq, dataset, out_dir, *, aligner, cache_dir, logger):
+ def fake_scan(segment, seq, dataset, out_dir, *, aligner, cache_dir, logger,
+ dir_name=None):
from tessera.reassort.scan import SegmentScan
seen.append(segment)
return SegmentScan(segment, True, segment == "HA", 1 if segment == "HA" else 0,
@@ -240,11 +241,12 @@ def resolve(fasta, override, *, email, logger):
_patch(monkeypatch, resolve=resolve,
tips_by_path={"HA_ds": [ha_tip], "NA_ds": [na_tip]},
- ani_by_path={"HA_pool": {ha_tip: (10.0, 0.99)}, # below ani_floor -> unassigned
- "NA_pool": {na_tip: (99.0, 0.99)}})
+ ani_by_path={"HA_pool": {ha_tip: (10.0, 99.0)}, # below ani_floor -> unassigned
+ "NA_pool": {na_tip: (99.0, 99.0)}})
monkeypatch.setattr(assign, "require_aligner", lambda aligner: None)
- def fake_scan(segment, seq, dataset, out_dir, *, aligner, cache_dir, logger):
+ def fake_scan(segment, seq, dataset, out_dir, *, aligner, cache_dir, logger,
+ dir_name=None):
from tessera.reassort.scan import SegmentScan
return SegmentScan(segment, True, False, 0, "none")
monkeypatch.setattr(assign, "scan_segment", fake_scan)
@@ -291,3 +293,56 @@ def test_cap_candidates_still_bounds_the_tail():
def test_cap_candidates_empty():
assert cap_candidates([], margin=0.5, top_k=25) == []
+
+
+def test_alignment_fraction_filter_uses_skani_percent_scale(tmp_path, monkeypatch):
+ # skani reports Align_fraction_query in percent (0-100). A tip that aligns over a fifth
+ # of the segment at a higher ANI must not outrank a full-length match: the values below
+ # are what skani 0.3 printed for a 30 kb query against a full-length tip and a tip
+ # covering only its first 6 kb.
+ full = tmp_path / "HA_pool" / "full.fasta"
+ partial = tmp_path / "HA_pool" / "partial.fasta"
+ na_tip = tmp_path / "NA_pool" / "strainB.fasta"
+ for t in (full, partial, na_tip):
+ t.parent.mkdir(parents=True, exist_ok=True)
+ t.write_text(">x\nACGT\n")
+
+ def resolve(fasta, override, *, email, logger):
+ return _DS("HA_ds") if "HA" in fasta.read_text() else _DS("NA_ds")
+
+ _patch(monkeypatch,
+ resolve=resolve,
+ tips_by_path={"HA_ds": [full, partial], "NA_ds": [na_tip]},
+ ani_by_path={"HA_pool": {full: (98.76, 100.0), partial: (99.29, 20.26)},
+ "NA_pool": {na_tip: (99.0, 99.0)}})
+ q = _write_query(tmp_path, [("HA", "HAxx"), ("NA", "NAyy")])
+ result = assign_segments(q, logger=LOG)
+ ha = next(s for s in result.segments if s.segment == "HA")
+ assert ha.status == "assigned"
+ assert ha.strain == "full"
+ assert ha.ani == pytest.approx(98.76)
+
+
+def test_scan_segments_gives_alike_named_segments_separate_directories(tmp_path, monkeypatch):
+ # "seg/1" and "seg_1" sanitise to the same directory name; each scan must get its own.
+ tip = tmp_path / "pool" / "strainA.fasta"
+ tip.parent.mkdir(parents=True)
+ tip.write_text(">x\nACGT\n")
+ _patch(monkeypatch,
+ resolve=lambda fasta, override, *, email, logger: _DS("ds"),
+ tips_by_path={"ds": [tip]},
+ ani_by_path={"pool": {tip: (99.0, 99.0)}})
+ monkeypatch.setattr(assign, "require_aligner", lambda aligner: None)
+ dirs = {}
+
+ def fake_scan(segment, seq, dataset, out_dir, *, aligner, cache_dir, logger,
+ dir_name=None):
+ from tessera.reassort.scan import SegmentScan
+ dirs[segment] = dir_name
+ return SegmentScan(segment, True, False, 0, "none")
+ monkeypatch.setattr(assign, "scan_segment", fake_scan)
+
+ q = _write_query(tmp_path, [("seg/1", "AAAA"), ("seg_1", "CCCC")])
+ assign_segments(q, output=tmp_path / "out", scan_segments=True, logger=LOG)
+
+ assert dirs == {"seg/1": "seg_1", "seg_1": "seg_1_2"}
diff --git a/tests/unit/test_reassort_scan.py b/tests/unit/test_reassort_scan.py
index 89bfb45..c9cb48a 100644
--- a/tests/unit/test_reassort_scan.py
+++ b/tests/unit/test_reassort_scan.py
@@ -139,3 +139,58 @@ def boom(params, logger):
cache_dir=None, logger=LOG)
assert result.scanned is False
assert "scan failed" in result.note
+
+
+def _two_clade_pool(tmp_path, monkeypatch):
+ pool = tmp_path / "pool"
+ pool.mkdir()
+ tips = []
+ for c in ("A", "B"):
+ t = pool / f"{c}_consensus.fasta"
+ t.write_text(f">{c}_consensus {c}\nACGTACGT\n")
+ tips.append(t)
+ _stub_pool(monkeypatch, tips)
+ monkeypatch.setattr(scan, "build_msa", lambda params, logger: params.output)
+ monkeypatch.setattr(scan, "run_recomb", lambda params, logger: "bp")
+
+
+def test_scan_segment_name_cannot_leave_the_output_directory(tmp_path, monkeypatch):
+ # The segment name is a FASTA header. ".." used to resolve the scan directory to the
+ # parent of the output, whose `collection/` was then removed and rebuilt.
+ _two_clade_pool(tmp_path, monkeypatch)
+ out = tmp_path / "parent" / "out"
+ out.mkdir(parents=True)
+ precious = tmp_path / "parent" / "collection" / "precious.fasta"
+ precious.parent.mkdir()
+ precious.write_text(">x\nACGT\n")
+
+ for name in ("..", ".", "../../escaped", "A/California/07/2009|HA"):
+ scan_segment(name, "ACGTACGT", _DS(), out, aligner="mafft", cache_dir=None, logger=LOG)
+
+ assert precious.exists()
+ outside = [p for p in (tmp_path / "parent").rglob("*")
+ if p.is_file() and out not in p.parents and p != precious]
+ assert outside == []
+ # Nothing is written into the output root itself: every segment has its own directory.
+ assert [p for p in out.iterdir() if p.is_file()] == []
+
+
+def test_safe_filename_stem():
+ from tessera.core.io import safe_filename_stem
+
+ assert safe_filename_stem("HA") == "HA"
+ assert safe_filename_stem("A/California/07/2009|HA") == "A_California_07_2009_HA"
+ assert safe_filename_stem("..", fallback="segment") == "segment"
+ assert safe_filename_stem(".", fallback="segment") == "segment"
+ assert safe_filename_stem("../x") == "_x"
+ assert safe_filename_stem("", fallback="segment") == "segment"
+
+
+def test_unique_scan_dirs_separates_names_that_sanitise_alike():
+ from tessera.reassort.scan import unique_scan_dirs
+
+ assert unique_scan_dirs(["HA", "NA"]) == {"HA": "HA", "NA": "NA"}
+ # "seg/1" and "seg_1" both sanitise to "seg_1"; they must not share a directory.
+ dirs = unique_scan_dirs(["seg/1", "seg_1", "seg 1"])
+ assert dirs == {"seg/1": "seg_1", "seg_1": "seg_1_2", "seg 1": "seg_1_3"}
+ assert len(set(dirs.values())) == 3