From 8c1c6fb9335081b3725960751f9e945d81911010 Mon Sep 17 00:00:00 2001 From: clintval Date: Thu, 8 Oct 2026 11:37:30 -0400 Subject: [PATCH 1/2] feat: let a labelled level drop its rows from General Statistics A labelled level now takes `keep_rows`, which defaults to true. With `keep_rows: false` its columns still fold onto the group row, renamed, placed and titled as before, but the original rows are no longer kept beneath it. Setting it to false on a level without a label, or on a level with a table, is a validation error. --- README.md | 21 +++++++++- multiqc_pivot/config.py | 12 +++++- multiqc_pivot/pivot.py | 30 +++++++++----- tests/data/multiqc_config_keep_rows.yml | 18 +++++++++ tests/test_config.py | 17 ++++++++ tests/test_pivot.py | 54 +++++++++++++++++++++++++ tests/test_plugin.py | 35 ++++++++++++++++ 7 files changed, 172 insertions(+), 15 deletions(-) create mode 100644 tests/data/multiqc_config_keep_rows.yml diff --git a/README.md b/README.md index 3b0e0ff..df7e3cb 100644 --- a/README.md +++ b/README.md @@ -22,7 +22,7 @@ MultiQC's own [sample grouping](https://docs.seqera.io/multiqc/reports/customisa This plugin runs after every module has reported and rebuilds the General Statistics table: 1. Rows for one group fold into a single row, and every folded column is prefixed by which method it came from. -2. The original rows stay beneath the group row, so they still can be viewed. +2. The original rows stay beneath the group row, so they still can be viewed, unless a level asks for them to be dropped. 3. Rows for a level that do not belong in the table, such as per-library read QC, move out into their own table under General Statistics with whatever grouping they already had. 4. Hover text, color scales, formats and hidden-by-default state carry over from the module that produced each column. @@ -71,7 +71,8 @@ And your report will look like: | `group` | A regular expression searched in every matched sample name. Its `(?P...)` capture names the folded row. Required. | | `levels` | An ordered list; the first level whose `match` is found in a sample name wins. Required. | | `levels[].match` | A regular expression searched in the sample name. Named captures are available to `label`. | -| `levels[].label` | A format string built from the captures of `match`. Columns of matching rows are renamed with it and folded onto the group row; the row itself stays beneath. Omit it, and omit `table`, to fold the row's columns onto the group row unchanged. | +| `levels[].label` | A format string built from the captures of `match`. Columns of matching rows are renamed with it and folded onto the group row; the row itself stays beneath unless `keep_rows` is `false`. Omit it, and omit `table`, to fold the row's columns onto the group row unchanged. | +| `levels[].keep_rows` | Whether rows of a labelled level stay beneath the group row. Set it to `false` to keep only the folded columns on the group row. Allowed only alongside `label`. Default `true`. | | `levels[].table` | The name of a table that receives matching rows instead of General Statistics. Rows keep their grouping, so paired reads stay nested under their library. Tables sit directly under General Statistics in the order their levels are listed. | | `column_title` | How a pivoted column is titled. `{label}` is the label as written, `{Label}` has its first letter upper-cased, `{title}` is the module's title. Default `{Label} {title}`. | | `label_order` | Labels in the order their column blocks should appear. Labels not listed follow in order of first appearance. | @@ -82,6 +83,22 @@ And your report will look like: > A sample that matches a level but not `group` is left alone as well, with a warning in the log. > Columns that a module did not declare a header for are dropped from folded rows, as MultiQC would have dropped them anyway. +### Dropping Labelled Rows + +When the group row already says everything you need, set `keep_rows: false` on a labelled level. +Its columns still fold onto the group row, renamed and titled as before, but the original rows no longer sit beneath it. +They are removed from General Statistics only, so each module's own section still shows them. + +```yaml +sample_pivot: + group: '^(?P[^. ]+)\.' + levels: + - match: '\.subject$' + - match: '\.(?PtissueA|tissueB)$' + label: '{analyte}' + keep_rows: false +``` + ### Limitations 1. Sample names are matched after MultiQC has cleaned them, so you must write patterns against the names you see in an un-pivoted report. diff --git a/multiqc_pivot/config.py b/multiqc_pivot/config.py index f3596b5..8be4391 100644 --- a/multiqc_pivot/config.py +++ b/multiqc_pivot/config.py @@ -28,8 +28,9 @@ class Level(BaseModel): A level with neither `label` nor `table` folds its columns onto the group row as they are. A level with a `label` renames its columns after the label, folds them onto the group row and - keeps the original row underneath. A level with a `table` moves its rows out of General - Statistics into a table of that name, keeping whatever grouping they already had. + keeps the original row underneath, unless `keep_rows` is false, which drops the original row + from General Statistics. A level with a `table` moves its rows out of General Statistics into a + table of that name, keeping whatever grouping they already had. """ model_config: ClassVar[ConfigDict] = ConfigDict(extra="forbid") @@ -37,6 +38,7 @@ class Level(BaseModel): match: str label: str | None = None table: str | None = None + keep_rows: bool = True @field_validator("match") @classmethod @@ -56,6 +58,12 @@ def _label_or_table(self) -> Level: raise ValueError(f"{message} ({exc})") from exc return self + @model_validator(mode="after") + def _keep_rows_needs_label(self) -> Level: + if not self.keep_rows and self.label is None: + raise ValueError("a level may set keep_rows to false only with a label") + return self + @property def pattern(self) -> re.Pattern[str]: """The compiled `match` expression.""" diff --git a/multiqc_pivot/pivot.py b/multiqc_pivot/pivot.py index 5ef430b..0ddc590 100644 --- a/multiqc_pivot/pivot.py +++ b/multiqc_pivot/pivot.py @@ -68,10 +68,15 @@ class Moved: @dataclass(frozen=True) class Folded: - """The row belongs on a group's row, under a label unless it is the group's own row.""" + """ + The row belongs on a group's row, under a label unless it is the group's own row. + + A labelled row also stays beneath the group row unless `keep_row` is false. + """ group: str label: str | None = None + keep_row: bool = True Route = Moved | Folded @@ -97,7 +102,7 @@ def classify(name: str, settings: SamplePivotConfig) -> Route | None: ) return None label = level.label.format(**match.groupdict()) if level.label is not None else None - return Folded(group.group("group"), label) + return Folded(group.group("group"), label, level.keep_rows) return None @@ -175,11 +180,12 @@ def fold(self, group: str, row: InputRow) -> None: if key in self.headers: _fold(target, key, value, group) - def fold_labelled(self, group: str, label: str, row: InputRow) -> None: + def fold_labelled(self, group: str, label: str, row: InputRow, keep_row: bool = True) -> None: """ Rename a row's declared columns after the label and put them onto the group row. - The row itself stays beneath the group row, carrying the same renamed columns. + The row itself stays beneath the group row, carrying the same renamed columns, unless + `keep_row` is false. """ target = self._group_rows.setdefault(group, {}) renamed: RowData = {} @@ -193,7 +199,8 @@ def fold_labelled(self, group: str, label: str, row: InputRow) -> None: ) renamed[new_key] = value _fold(target, new_key, value, group) - self._sub_rows.setdefault(group, []).append(InputRow(sample=row.sample, data=renamed)) + if keep_row: + self._sub_rows.setdefault(group, []).append(InputRow(sample=row.sample, data=renamed)) def finish(self) -> SectionRows: """The rebuilt rows, each group row first with its folded rows beneath.""" @@ -209,11 +216,12 @@ def pivot(rows: Rows, headers: Headers, settings: SamplePivotConfig) -> PivotRes Rebuild General Statistics with one row per group. Rows that match a labelled level have their declared columns renamed after the label and copied - onto the group's row; the original rows stay beneath it so the group can still be expanded. - Rows that match an unlabelled level are folded onto the group's row as they are. Rows that match - a level with a table are moved into that table with their grouping intact; the tables come back - in the order their levels are listed, without the ones that received no rows. Rows that match no - level, and columns a module did not declare a header for, are left alone. + onto the group's row; the original rows stay beneath it so the group can still be expanded, + unless the level sets `keep_rows` to false. Rows that match an unlabelled level are folded onto + the group's row as they are. Rows that match a level with a table are moved into that table with + their grouping intact; the tables come back in the order their levels are listed, without the + ones that received no rows. Rows that match no level, and columns a module did not declare a + header for, are left alone. """ placement = Placement(settings.label_order) out_rows: Rows = {} @@ -232,7 +240,7 @@ def pivot(rows: Rows, headers: Headers, settings: SamplePivotConfig) -> PivotRes elif route.label is None: pivoted.fold(route.group, row) else: - pivoted.fold_labelled(route.group, route.label, row) + pivoted.fold_labelled(route.group, route.label, row, route.keep_row) out_rows[section] = pivoted.finish() out_headers[section] = pivoted.new_headers return PivotResult(out_rows, out_headers, {n: t for n, t in tables.items() if t.rows}) diff --git a/tests/data/multiqc_config_keep_rows.yml b/tests/data/multiqc_config_keep_rows.yml new file mode 100644 index 0000000..96e6350 --- /dev/null +++ b/tests/data/multiqc_config_keep_rows.yml @@ -0,0 +1,18 @@ +disable_version_detection: true +no_ai: true + +sample_pivot: + group: '^(?P[^. ]+)\.' + levels: + - match: '\.subject$' + - match: '\.(?PtissueA|tissueB)$' + label: '{analyte}' + keep_rows: false + - match: '\.(?PtissueA|tissueB) \(filtered\)$' + label: '{analyte} (filtered)' + - match: '\.library\.' + table: Library statistics + label_order: [tissueA, tissueB, tissueB (filtered)] + tables: + Library statistics: + description: Per-library read QC. diff --git a/tests/test_config.py b/tests/test_config.py index d85422f..2f40c2c 100644 --- a/tests/test_config.py +++ b/tests/test_config.py @@ -53,6 +53,23 @@ def test_label_and_table_are_exclusive() -> None: Level(match=r"\.x$", label="x", table="Table") +def test_keep_rows_defaults_to_true() -> None: + assert Level(match=r"\.subject$").keep_rows + level = Level.model_validate({ + "match": r"\.(?Px)$", + "label": "{analyte}", + "keep_rows": False, + }) + assert not level.keep_rows + + +def test_keep_rows_false_needs_a_label() -> None: + with pytest.raises(ValidationError, match="keep_rows to false only with a label"): + Level(match=r"\.subject$", keep_rows=False) + with pytest.raises(ValidationError, match="keep_rows to false only with a label"): + Level(match=r"\.library\.", table="Library statistics", keep_rows=False) + + def test_label_must_use_captures_of_match() -> None: with pytest.raises(ValidationError, match="lacks"): Level(match=r"\.(?Px)$", label="{tissue}") diff --git a/tests/test_pivot.py b/tests/test_pivot.py index a3d61e9..c2cb05d 100644 --- a/tests/test_pivot.py +++ b/tests/test_pivot.py @@ -127,6 +127,60 @@ def test_labelled_rows_fold_into_one_row_per_group() -> None: assert new_headers[ColumnKey("median")] == headers[coverage][ColumnKey("median")] +def test_keep_rows_false_drops_only_the_rows_beneath_the_group_row() -> None: + coverage = SectionKey("coverage") + rows = {coverage: section(row("101.tissueA", median=743), row("101.tissueB", median=419))} + headers = {coverage: {ColumnKey("median"): header("Median", suffix="X")}} + match = r"\.(?PtissueA|tissueB)$" + kept = SETTINGS.model_copy(update={"levels": [Level(match=match, label="{analyte}")]}) + dropped = SETTINGS.model_copy( + update={"levels": [Level(match=match, label="{analyte}", keep_rows=False)]} + ) + + with_rows = pivot(rows, headers, kept) + without_rows = pivot(rows, headers, dropped) + + group_row = row("101", median__tissuea=743, median__tissueb=419) + assert with_rows.rows[coverage] == { + SampleGroup("101"): [ + group_row, + row("101.tissueA", median__tissuea=743), + row("101.tissueB", median__tissueb=419), + ] + } + assert without_rows.rows[coverage] == {SampleGroup("101"): [group_row]} + assert without_rows.headers == with_rows.headers + + +def test_keep_rows_applies_per_level() -> None: + settings = SETTINGS.model_copy( + update={ + "levels": [ + Level(match=r"\.(?PtissueA|tissueB)$", label="{analyte}", keep_rows=False), + Level(match=r"\.(?PtissueB) \(filtered\)$", label="{analyte} (filtered)"), + ] + } + ) + alignment = SectionKey("alignment") + rows = { + alignment: section( + row("101.tissueB", aligned=36.5), row("101.tissueB (filtered)", aligned=35.4) + ) + } + headers = {alignment: {ColumnKey("aligned"): header("% Aligned")}} + + result = pivot(rows, headers, settings) + + assert classify("101.tissueB", settings) == Folded("101", "tissueB", keep_row=False) + assert classify("101.tissueB (filtered)", settings) == Folded("101", "tissueB (filtered)") + assert result.rows[alignment] == { + SampleGroup("101"): [ + row("101", aligned__tissueb=36.5, aligned__tissueb_filtered=35.4), + row("101.tissueB (filtered)", aligned__tissueb_filtered=35.4), + ] + } + + def test_unlabelled_rows_become_the_group_row() -> None: concordance = SectionKey("concordance") rows = {concordance: section(row("101.subject", concordance=99.7, undeclared="x"))} diff --git a/tests/test_plugin.py b/tests/test_plugin.py index 6e52192..0a39c51 100644 --- a/tests/test_plugin.py +++ b/tests/test_plugin.py @@ -73,5 +73,40 @@ def test_report(tmp_path: Path) -> None: assert html.count("Per-library read QC.") == 1 +def test_report_without_labelled_rows(tmp_path: Path) -> None: + multiqc.reset() # type: ignore[no-untyped-call] + multiqc.parse_logs( + str(DATA / "report"), config_files=[str(DATA / "multiqc_config_keep_rows.yml")] + ) + + samples = { + str(row.sample) + for section in report.general_stats_data.values() + for rows in section.values() + for row in rows + } + assert samples == {"101", "102", "101.tissueB (filtered)", "102.tissueB (filtered)"} + titles = { + str(column.get("title")) + for section in report.general_stats_headers.values() + for column in section.values() + } + assert { + "Concordance", + "TissueA Median", + "TissueB Median", + "TissueB (filtered) % Aligned", + } <= titles + + multiqc.write_report(output_dir=str(tmp_path), filename="report.html", force=True) + html = (tmp_path / "report.html").read_text() + general_stats = re.search(r'', html, re.S) + assert general_stats is not None + assert general_stats.group(0).count('class="expandable-row-primary"') == 2 + assert general_stats.group(0).count('class="expandable-row-secondary') == 2 + assert 'data-original-sn="101.tissueA"' not in general_stats.group(0) + assert "TissueA Median" in general_stats.group(0) + + def test_table_module_is_none_for_an_empty_table() -> None: assert table_module("Empty", TableSettings(), Table()) is None From e2145fa54fb60d4c3c5bfc613bbf3c56b2e5c76a Mon Sep 17 00:00:00 2001 From: clintval Date: Thu, 8 Oct 2026 11:37:54 -0400 Subject: [PATCH 2/2] chore(release): bump to 0.2.0 --- pyproject.toml | 2 +- uv.lock | 2 +- 2 files changed, 2 insertions(+), 2 deletions(-) diff --git a/pyproject.toml b/pyproject.toml index 147e9d7..f6d0687 100644 --- a/pyproject.toml +++ b/pyproject.toml @@ -4,7 +4,7 @@ build-backend = "hatchling.build" [project] name = "multiqc-pivot" -version = "0.1.1" +version = "0.2.0" description = "A MultiQC plugin that folds related samples into one General Statistics row per group with labelled metric columns." readme = "README.md" authors = [{ name = "Clint Valentine", email = "valentine.clint@gmail.com" }] diff --git a/uv.lock b/uv.lock index 25e63dc..8c8ad81 100644 --- a/uv.lock +++ b/uv.lock @@ -701,7 +701,7 @@ wheels = [ [[package]] name = "multiqc-pivot" -version = "0.1.1" +version = "0.2.0" source = { editable = "." } dependencies = [ { name = "multiqc" },