Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
21 changes: 19 additions & 2 deletions README.md
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,7 @@ MultiQC's own [sample grouping](https://docs.seqera.io/multiqc/reports/customisa
This plugin runs after every module has reported and rebuilds the General Statistics table:

1. Rows for one group fold into a single row, and every folded column is prefixed by which method it came from.
2. The original rows stay beneath the group row, so they still can be viewed.
2. The original rows stay beneath the group row, so they still can be viewed, unless a level asks for them to be dropped.
3. Rows for a level that do not belong in the table, such as per-library read QC, move out into their own table under General Statistics with whatever grouping they already had.
4. Hover text, color scales, formats and hidden-by-default state carry over from the module that produced each column.

Expand Down Expand Up @@ -71,7 +71,8 @@ And your report will look like:
| `group` | A regular expression searched in every matched sample name. Its `(?P<group>...)` capture names the folded row. Required. |
| `levels` | An ordered list; the first level whose `match` is found in a sample name wins. Required. |
| `levels[].match` | A regular expression searched in the sample name. Named captures are available to `label`. |
| `levels[].label` | A format string built from the captures of `match`. Columns of matching rows are renamed with it and folded onto the group row; the row itself stays beneath. Omit it, and omit `table`, to fold the row's columns onto the group row unchanged. |
| `levels[].label` | A format string built from the captures of `match`. Columns of matching rows are renamed with it and folded onto the group row; the row itself stays beneath unless `keep_rows` is `false`. Omit it, and omit `table`, to fold the row's columns onto the group row unchanged. |
| `levels[].keep_rows` | Whether rows of a labelled level stay beneath the group row. Set it to `false` to keep only the folded columns on the group row. Allowed only alongside `label`. Default `true`. |
| `levels[].table` | The name of a table that receives matching rows instead of General Statistics. Rows keep their grouping, so paired reads stay nested under their library. Tables sit directly under General Statistics in the order their levels are listed. |
| `column_title` | How a pivoted column is titled. `{label}` is the label as written, `{Label}` has its first letter upper-cased, `{title}` is the module's title. Default `{Label} {title}`. |
| `label_order` | Labels in the order their column blocks should appear. Labels not listed follow in order of first appearance. |
Expand All @@ -82,6 +83,22 @@ And your report will look like:
> A sample that matches a level but not `group` is left alone as well, with a warning in the log.
> Columns that a module did not declare a header for are dropped from folded rows, as MultiQC would have dropped them anyway.

### Dropping Labelled Rows

When the group row already says everything you need, set `keep_rows: false` on a labelled level.
Its columns still fold onto the group row, renamed and titled as before, but the original rows no longer sit beneath it.
They are removed from General Statistics only, so each module's own section still shows them.

```yaml
sample_pivot:
group: '^(?P<group>[^. ]+)\.'
levels:
- match: '\.subject$'
- match: '\.(?P<analyte>tissueA|tissueB)$'
label: '{analyte}'
keep_rows: false
```

### Limitations

1. Sample names are matched after MultiQC has cleaned them, so you must write patterns against the names you see in an un-pivoted report.
Expand Down
12 changes: 10 additions & 2 deletions multiqc_pivot/config.py
Original file line number Diff line number Diff line change
Expand Up @@ -28,15 +28,17 @@ class Level(BaseModel):

A level with neither `label` nor `table` folds its columns onto the group row as they are. A
level with a `label` renames its columns after the label, folds them onto the group row and
keeps the original row underneath. A level with a `table` moves its rows out of General
Statistics into a table of that name, keeping whatever grouping they already had.
keeps the original row underneath, unless `keep_rows` is false, which drops the original row
from General Statistics. A level with a `table` moves its rows out of General Statistics into a
table of that name, keeping whatever grouping they already had.
"""

model_config: ClassVar[ConfigDict] = ConfigDict(extra="forbid")

match: str
label: str | None = None
table: str | None = None
keep_rows: bool = True

@field_validator("match")
@classmethod
Expand All @@ -56,6 +58,12 @@ def _label_or_table(self) -> Level:
raise ValueError(f"{message} ({exc})") from exc
return self

@model_validator(mode="after")
def _keep_rows_needs_label(self) -> Level:
if not self.keep_rows and self.label is None:
raise ValueError("a level may set keep_rows to false only with a label")
return self

@property
def pattern(self) -> re.Pattern[str]:
"""The compiled `match` expression."""
Expand Down
30 changes: 19 additions & 11 deletions multiqc_pivot/pivot.py
Original file line number Diff line number Diff line change
Expand Up @@ -68,10 +68,15 @@ class Moved:

@dataclass(frozen=True)
class Folded:
"""The row belongs on a group's row, under a label unless it is the group's own row."""
"""
The row belongs on a group's row, under a label unless it is the group's own row.

A labelled row also stays beneath the group row unless `keep_row` is false.
"""

group: str
label: str | None = None
keep_row: bool = True


Route = Moved | Folded
Expand All @@ -97,7 +102,7 @@ def classify(name: str, settings: SamplePivotConfig) -> Route | None:
)
return None
label = level.label.format(**match.groupdict()) if level.label is not None else None
return Folded(group.group("group"), label)
return Folded(group.group("group"), label, level.keep_rows)
return None


Expand Down Expand Up @@ -175,11 +180,12 @@ def fold(self, group: str, row: InputRow) -> None:
if key in self.headers:
_fold(target, key, value, group)

def fold_labelled(self, group: str, label: str, row: InputRow) -> None:
def fold_labelled(self, group: str, label: str, row: InputRow, keep_row: bool = True) -> None:
"""
Rename a row's declared columns after the label and put them onto the group row.

The row itself stays beneath the group row, carrying the same renamed columns.
The row itself stays beneath the group row, carrying the same renamed columns, unless
`keep_row` is false.
"""
target = self._group_rows.setdefault(group, {})
renamed: RowData = {}
Expand All @@ -193,7 +199,8 @@ def fold_labelled(self, group: str, label: str, row: InputRow) -> None:
)
renamed[new_key] = value
_fold(target, new_key, value, group)
self._sub_rows.setdefault(group, []).append(InputRow(sample=row.sample, data=renamed))
if keep_row:
self._sub_rows.setdefault(group, []).append(InputRow(sample=row.sample, data=renamed))

def finish(self) -> SectionRows:
"""The rebuilt rows, each group row first with its folded rows beneath."""
Expand All @@ -209,11 +216,12 @@ def pivot(rows: Rows, headers: Headers, settings: SamplePivotConfig) -> PivotRes
Rebuild General Statistics with one row per group.

Rows that match a labelled level have their declared columns renamed after the label and copied
onto the group's row; the original rows stay beneath it so the group can still be expanded.
Rows that match an unlabelled level are folded onto the group's row as they are. Rows that match
a level with a table are moved into that table with their grouping intact; the tables come back
in the order their levels are listed, without the ones that received no rows. Rows that match no
level, and columns a module did not declare a header for, are left alone.
onto the group's row; the original rows stay beneath it so the group can still be expanded,
unless the level sets `keep_rows` to false. Rows that match an unlabelled level are folded onto
the group's row as they are. Rows that match a level with a table are moved into that table with
their grouping intact; the tables come back in the order their levels are listed, without the
ones that received no rows. Rows that match no level, and columns a module did not declare a
header for, are left alone.
"""
placement = Placement(settings.label_order)
out_rows: Rows = {}
Expand All @@ -232,7 +240,7 @@ def pivot(rows: Rows, headers: Headers, settings: SamplePivotConfig) -> PivotRes
elif route.label is None:
pivoted.fold(route.group, row)
else:
pivoted.fold_labelled(route.group, route.label, row)
pivoted.fold_labelled(route.group, route.label, row, route.keep_row)
out_rows[section] = pivoted.finish()
out_headers[section] = pivoted.new_headers
return PivotResult(out_rows, out_headers, {n: t for n, t in tables.items() if t.rows})
Expand Down
2 changes: 1 addition & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@ build-backend = "hatchling.build"

[project]
name = "multiqc-pivot"
version = "0.1.1"
version = "0.2.0"
description = "A MultiQC plugin that folds related samples into one General Statistics row per group with labelled metric columns."
readme = "README.md"
authors = [{ name = "Clint Valentine", email = "valentine.clint@gmail.com" }]
Expand Down
18 changes: 18 additions & 0 deletions tests/data/multiqc_config_keep_rows.yml
Original file line number Diff line number Diff line change
@@ -0,0 +1,18 @@
disable_version_detection: true
no_ai: true

sample_pivot:
group: '^(?P<group>[^. ]+)\.'
levels:
- match: '\.subject$'
- match: '\.(?P<analyte>tissueA|tissueB)$'
label: '{analyte}'
keep_rows: false
- match: '\.(?P<analyte>tissueA|tissueB) \(filtered\)$'
label: '{analyte} (filtered)'
- match: '\.library\.'
table: Library statistics
label_order: [tissueA, tissueB, tissueB (filtered)]
tables:
Library statistics:
description: Per-library read QC.
17 changes: 17 additions & 0 deletions tests/test_config.py
Original file line number Diff line number Diff line change
Expand Up @@ -53,6 +53,23 @@ def test_label_and_table_are_exclusive() -> None:
Level(match=r"\.x$", label="x", table="Table")


def test_keep_rows_defaults_to_true() -> None:
assert Level(match=r"\.subject$").keep_rows
level = Level.model_validate({
"match": r"\.(?P<analyte>x)$",
"label": "{analyte}",
"keep_rows": False,
})
assert not level.keep_rows


def test_keep_rows_false_needs_a_label() -> None:
with pytest.raises(ValidationError, match="keep_rows to false only with a label"):
Level(match=r"\.subject$", keep_rows=False)
with pytest.raises(ValidationError, match="keep_rows to false only with a label"):
Level(match=r"\.library\.", table="Library statistics", keep_rows=False)


def test_label_must_use_captures_of_match() -> None:
with pytest.raises(ValidationError, match="lacks"):
Level(match=r"\.(?P<analyte>x)$", label="{tissue}")
Expand Down
54 changes: 54 additions & 0 deletions tests/test_pivot.py
Original file line number Diff line number Diff line change
Expand Up @@ -127,6 +127,60 @@ def test_labelled_rows_fold_into_one_row_per_group() -> None:
assert new_headers[ColumnKey("median")] == headers[coverage][ColumnKey("median")]


def test_keep_rows_false_drops_only_the_rows_beneath_the_group_row() -> None:
coverage = SectionKey("coverage")
rows = {coverage: section(row("101.tissueA", median=743), row("101.tissueB", median=419))}
headers = {coverage: {ColumnKey("median"): header("Median", suffix="X")}}
match = r"\.(?P<analyte>tissueA|tissueB)$"
kept = SETTINGS.model_copy(update={"levels": [Level(match=match, label="{analyte}")]})
dropped = SETTINGS.model_copy(
update={"levels": [Level(match=match, label="{analyte}", keep_rows=False)]}
)

with_rows = pivot(rows, headers, kept)
without_rows = pivot(rows, headers, dropped)

group_row = row("101", median__tissuea=743, median__tissueb=419)
assert with_rows.rows[coverage] == {
SampleGroup("101"): [
group_row,
row("101.tissueA", median__tissuea=743),
row("101.tissueB", median__tissueb=419),
]
}
assert without_rows.rows[coverage] == {SampleGroup("101"): [group_row]}
assert without_rows.headers == with_rows.headers


def test_keep_rows_applies_per_level() -> None:
settings = SETTINGS.model_copy(
update={
"levels": [
Level(match=r"\.(?P<analyte>tissueA|tissueB)$", label="{analyte}", keep_rows=False),
Level(match=r"\.(?P<analyte>tissueB) \(filtered\)$", label="{analyte} (filtered)"),
]
}
)
alignment = SectionKey("alignment")
rows = {
alignment: section(
row("101.tissueB", aligned=36.5), row("101.tissueB (filtered)", aligned=35.4)
)
}
headers = {alignment: {ColumnKey("aligned"): header("% Aligned")}}

result = pivot(rows, headers, settings)

assert classify("101.tissueB", settings) == Folded("101", "tissueB", keep_row=False)
assert classify("101.tissueB (filtered)", settings) == Folded("101", "tissueB (filtered)")
assert result.rows[alignment] == {
SampleGroup("101"): [
row("101", aligned__tissueb=36.5, aligned__tissueb_filtered=35.4),
row("101.tissueB (filtered)", aligned__tissueb_filtered=35.4),
]
}


def test_unlabelled_rows_become_the_group_row() -> None:
concordance = SectionKey("concordance")
rows = {concordance: section(row("101.subject", concordance=99.7, undeclared="x"))}
Expand Down
35 changes: 35 additions & 0 deletions tests/test_plugin.py
Original file line number Diff line number Diff line change
Expand Up @@ -73,5 +73,40 @@ def test_report(tmp_path: Path) -> None:
assert html.count("Per-library read QC.") == 1


def test_report_without_labelled_rows(tmp_path: Path) -> None:
multiqc.reset() # type: ignore[no-untyped-call]
multiqc.parse_logs(
str(DATA / "report"), config_files=[str(DATA / "multiqc_config_keep_rows.yml")]
)

samples = {
str(row.sample)
for section in report.general_stats_data.values()
for rows in section.values()
for row in rows
}
assert samples == {"101", "102", "101.tissueB (filtered)", "102.tissueB (filtered)"}
titles = {
str(column.get("title"))
for section in report.general_stats_headers.values()
for column in section.values()
}
assert {
"Concordance",
"TissueA Median",
"TissueB Median",
"TissueB (filtered) % Aligned",
} <= titles

multiqc.write_report(output_dir=str(tmp_path), filename="report.html", force=True)
html = (tmp_path / "report.html").read_text()
general_stats = re.search(r'<table id="general_stats_table_table".*?</table>', html, re.S)
assert general_stats is not None
assert general_stats.group(0).count('class="expandable-row-primary"') == 2
assert general_stats.group(0).count('class="expandable-row-secondary') == 2
assert 'data-original-sn="101.tissueA"' not in general_stats.group(0)
assert "TissueA Median" in general_stats.group(0)


def test_table_module_is_none_for_an_empty_table() -> None:
assert table_module("Empty", TableSettings(), Table()) is None
2 changes: 1 addition & 1 deletion uv.lock

Some generated files are not rendered by default. Learn more about how customized files appear on GitHub.

Loading