Skip to content
Merged
Show file tree
Hide file tree
Changes from 2 commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
27 changes: 21 additions & 6 deletions src/pixelator/pna/anndata.py
Original file line number Diff line number Diff line change
Expand Up @@ -41,14 +41,19 @@ def add_panel_information(adata: AnnData, panel: PNAAntibodyPanel) -> AnnData:


def pna_edgelist_to_anndata(
pixel_connection: duckdb.DuckDBPyConnection, panel: PNAAntibodyPanel
pixel_connection: duckdb.DuckDBPyConnection,
panel: PNAAntibodyPanel,
markers: list[str] | None = None,
) -> AnnData:
"""Build an AnnData object from a DuckDB connection to a pixel file and a panel object.

Args:
pixel_connection: A DuckDB connection to a pixel file. The connection must contain an 'edgelist' table
with the required columns (e.g., component, marker_1, marker_2, umi1, umi2, read_count).
panel: The antibody panel object containing marker metadata.
markers: Marker ids to use as the count-matrix columns. Defaults to
every marker on ``panel``. Sample calling passes the markers left
after hashing clones collapse to their base name.

Returns:
An AnnData object with counts and panel information.
Expand All @@ -73,10 +78,11 @@ def pna_edgelist_to_anndata(
.tolist()
)

marker_list = list(panel.markers if markers is None else markers)
n_components = len(components)
n_markers = len(panel.markers)
n_markers = len(marker_list)
component_to_idx = {c: i for i, c in enumerate(components)}
marker_to_idx = {m: i for i, m in enumerate(panel.markers)}
marker_to_idx = {m: i for i, m in enumerate(marker_list)}

X = np.zeros((n_components, n_markers), dtype=np.uint32)
n_umi1_arr = np.zeros(n_components, dtype=np.uint64)
Expand Down Expand Up @@ -152,7 +158,7 @@ def pna_edgelist_to_anndata(
node_counts_df = pd.DataFrame(
X,
index=component_index,
columns=pd.Index(panel.markers, name="marker_id"),
columns=pd.Index(marker_list, name="marker_id"),
)

logger.debug("Computing component metrics.")
Expand Down Expand Up @@ -182,7 +188,7 @@ def pna_edgelist_to_anndata(

logger.debug("Computing antibody metrics.")
antibody_metrics_df = calculate_antibody_metrics(counts_df=node_counts_df)
antibody_metrics_df = antibody_metrics_df.reindex(index=panel.markers, fill_value=0)
antibody_metrics_df = antibody_metrics_df.reindex(index=marker_list, fill_value=0)
antibody_metrics_df.index.name = "marker_id"
# Do a dtype conversion of the columns here since AnnData cannot handle
# a pyarrow arrays.
Expand All @@ -201,7 +207,16 @@ def pna_edgelist_to_anndata(
adata = add_panel_information(adata, panel)

total_marker_counts = node_counts_df.sum(axis=1)
isotype_markers = adata.var[adata.var["control"]].index
control = panel.df["control"]
if pd.api.types.is_bool_dtype(control):
control_mask = control.fillna(False).astype(bool)
else:
control_mask = control.astype(str).str.lower().eq("yes")
isotype_markers = [
marker
for marker in control_mask.index[control_mask]
if marker in node_counts_df.columns
]
isotype_counts = node_counts_df[isotype_markers].sum(axis=1)
adata.obs["isotype_fraction"] = isotype_counts / total_marker_counts

Expand Down
10 changes: 7 additions & 3 deletions src/pixelator/pna/cli/sample_calling.py
Original file line number Diff line number Diff line change
Expand Up @@ -118,9 +118,13 @@ def sample_calling_cli(
pool_metadata=PxlFile(Path(input_pxl_file)).metadata(),
)
return
hashing_antibodies_in_panel = set(
panel_info.df[panel_info.df["sample_hashing"] == "yes"].index.to_list()
)
if "sample_hashing" not in panel_info.df.columns:
raise ValueError(
"Sample calling requires a sample_hashing column on the panel "
"so hashing markers can be identified. This panel has no "
"sample_hashing column."
)
hashing_antibodies_in_panel = panel_info.hashing_marker_ids
samplesheet_df = pl.read_csv(samplesheet)
_reject_reserved_samplesheet_names(
samplesheet_df["sample"].to_list(),
Expand Down
80 changes: 80 additions & 0 deletions src/pixelator/pna/config/panel.py
Original file line number Diff line number Diff line change
Expand Up @@ -33,6 +33,43 @@
from pixelator.pna.config.config_class import PNAConfig
from pixelator.pna.pixeldataset.dataset import PNAPixelDataset

# Trailing ``-<digits>`` is the hash group (``B2M-1`` → ``B2M``). The same
# pattern matches ordinary names such as ``PD-1``, so it is only applied to
# rows already flagged by ``sample_hashing``.
_HASHING_MARKER_ID_RE = re.compile(r"^(?P<base>.+)-(?P<index>\d+)$")


def sample_hashing_mask(sample_hashing: pd.Series) -> pd.Series:
"""Return a boolean mask for values that flag a hashing marker."""
if pd.api.types.is_bool_dtype(sample_hashing):
return sample_hashing.fillna(False).astype(bool)
if pd.api.types.is_numeric_dtype(sample_hashing):
return sample_hashing.fillna(0).astype(bool)
normalized = sample_hashing.astype(str).str.strip().str.lower()
return normalized.isin(["yes", "true"])


def split_hashing_marker_id(marker_id: str) -> tuple[str, str] | None:
"""Return ``(base, index)`` for a hashing id such as ``B2M-1``."""
match = _HASHING_MARKER_ID_RE.fullmatch(str(marker_id))
if match is None:
return None
return match.group("base"), match.group("index")


def collapsed_hashing_marker_id(marker_id: str) -> str:
"""Return the marker id sample calling stores for a hashing antibody."""
parts = split_hashing_marker_id(marker_id)
return parts[0] if parts is not None else str(marker_id)


def _hashing_marker_ids(panel_df: pd.DataFrame) -> set[str]:
"""Return hashing marker ids, or an empty set when the column is absent."""
if "sample_hashing" not in panel_df.columns:
return set()
mask = sample_hashing_mask(panel_df["sample_hashing"])
return {str(marker_id) for marker_id in panel_df.index[mask]}


class PNAAntibodyPanel:
"""Class representing a PNA antibody panel."""
Expand Down Expand Up @@ -210,6 +247,11 @@ def archived(self) -> Optional[bool]:
"""Return whether the panel is marked as archived."""
return self.metadata.archived

@property
def hashing_marker_ids(self) -> set[str]:
"""Return marker ids flagged by the ``sample_hashing`` column."""
return _hashing_marker_ids(self.df)

@classmethod
def _parse_header(cls, file: Path) -> AntibodyPanelMetadata:
"""Parse front-matter YAML metadata from a panel file.
Expand Down Expand Up @@ -349,6 +391,13 @@ def validate_antibody_panel(
errors.append(f"`{cls._INDEX_COLUMN}` is missing or is not set as index")
return errors

if panel_df.index.duplicated().any():
duplicated = panel_df.index[panel_df.index.duplicated()].unique().tolist()
errors.append(
"All values in column: marker_id were not unique. "
f"Offending values: {duplicated}"
)

errors += cls._validate_marker_names(panel_df)

if panel_df["control"].dtype != bool:
Expand Down Expand Up @@ -380,7 +429,38 @@ def check_id(id_str):

errors += cls._validate_sequences(panel_df, "sequence_1")
errors += cls._validate_sequences(panel_df, "sequence_2")
errors += cls._validate_hashing_marker_ids(panel_df)

return errors

@staticmethod
def _validate_hashing_marker_ids(panel_df: pd.DataFrame) -> list[str]:
"""Return errors when hashing ids lack a ``-<digits>`` suffix or nest."""
hashing_ids = _hashing_marker_ids(panel_df)
if not hashing_ids:
return []
errors: list[str] = []
missing_suffix = [
marker_id
for marker_id in sorted(hashing_ids)
if split_hashing_marker_id(marker_id) is None
]
if missing_suffix:
errors.append(
"Hashing marker ids must end with -<digits> (e.g. B2M-1). "
f"Offending values: {missing_suffix}"
)
nested = sorted(
marker_id
for marker_id in hashing_ids
if (base := collapsed_hashing_marker_id(marker_id)) != marker_id
and base in hashing_ids
)
if nested:
errors.append(
"Hashing marker ids must not collapse to another hashing id "
f"(e.g. B2M-1-1 next to B2M-1). Offending values: {nested}"
)
return errors

def to_polars(self) -> pl.DataFrame:
Expand Down
41 changes: 30 additions & 11 deletions src/pixelator/pna/sample_calling/__init__.py
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,6 @@
"""

import logging
import re
import tempfile
from itertools import chain
from pathlib import Path
Expand All @@ -20,7 +19,7 @@
from pixelator import __version__
from pixelator.pna.analysis_engine import AnalysisManager, PerComponentTask
from pixelator.pna.anndata import add_missing_adata_info, pna_edgelist_to_anndata
from pixelator.pna.config.panel import PNAAntibodyPanel
from pixelator.pna.config.panel import PNAAntibodyPanel, collapsed_hashing_marker_id
from pixelator.pna.pixeldataset import PNAPixelDataset
from pixelator.pna.pixeldataset.io import PixelFileWriter
from pixelator.pna.sample_calling.hash_antibodies import HashedAntibodyMapping
Expand Down Expand Up @@ -209,6 +208,25 @@ def _add_original_hash_counts_to_obs(
old_adata.obs[f"original_hash_counts_{ab}"] = 0


def _count_markers_after_hash_collapse(
panel: PNAAntibodyPanel, hashing_antibodies: set[str]
) -> list[str]:
"""Return count-matrix markers after hashing ids collapse to their base name.

Hashing clones are left out. A collapsed base that is not already on the
panel is appended, so edgelist counts under that name are kept.
"""
hashing = {str(marker) for marker in hashing_antibodies}
kept = [str(marker) for marker in panel.markers if str(marker) not in hashing]
present = set(kept)
for marker in sorted(hashing):
base = collapsed_hashing_marker_id(marker)
if base not in present:
kept.append(base)
present.add(base)
return kept


def _build_post_sample_calling_anndata(
con: duckdb.DuckDBPyConnection,
old_adata: anndata.AnnData,
Expand All @@ -235,14 +253,15 @@ def _build_post_sample_calling_anndata(
old_adata, hashing_antibody_mapping.hashing_antibodies
)

# Create the anndata object and remove all panel hashing markers from var
new_adata = pna_edgelist_to_anndata(con, panel)
non_hashing_markers = [
marker
for marker in new_adata.var.index
if marker not in hashing_antibody_mapping.hashing_antibodies
]
new_adata = new_adata[:, non_hashing_markers].copy()
# The edgelist already uses collapsed hashing ids (B2M-1 -> B2M). Include
# a collapsed base that the panel does not define, and leave the clones out.
new_adata = pna_edgelist_to_anndata(
con,
panel,
markers=_count_markers_after_hash_collapse(
panel, hashing_antibody_mapping.hashing_antibodies
),
)

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Collapsed bases break panel reload

Medium Severity

Sample calling now keeps a collapsed hashing base in var when that id is not a panel row, so control and sequences are missing. PNAAntibodyPanel.from_adata treats every var row as a panel marker, and validation then fails. Denoise reloads the panel from the pxl file and only catches KeyError, so a dehashed file with an extra collapsed base crashes that step.

Additional Locations (1)
Fix in Cursor Fix in Web

Reviewed by Cursor Bugbot for commit e475f94. Configure here.

Copy link
Copy Markdown
Contributor

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

lets ignore this bugbot comment, wont apply once we merge the stack


new_adata = add_missing_adata_info(new_adata, old_adata)
# `sample` is a reserved, transient column added when reading a
Expand Down Expand Up @@ -390,7 +409,7 @@ def sample_calling(
{
"hashed_marker": hashed_markers,
"base_marker": [
re.sub(r"-\d+$", "", marker) for marker in hashed_markers
collapsed_hashing_marker_id(marker) for marker in hashed_markers
],
}
)
Expand Down
12 changes: 10 additions & 2 deletions tests/common/data_generator/molecules.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,12 +11,20 @@
import numpy as np
import polars as pl

from pixelator.pna.config.panel import split_hashing_marker_id
from tests.common.data_generator.topology import generate_cell_graph

if TYPE_CHECKING:
from pixelator.pna.config.panel import PNAAntibodyPanel


def _hashing_index(marker_id: str) -> int:
parts = split_hashing_marker_id(marker_id)
if parts is None:
raise ValueError(f"Hashing marker {marker_id!r} must end with -<digits>.")
return int(parts[1])


def generate_edgelist(
n_cells: int,
n_nodes: int,
Expand Down Expand Up @@ -197,7 +205,7 @@ def _hashing_indices_per_cell(
if hashing.size == 0:
return [None] * n_cells
if hashing_indices is None:
indices = np.unique([int(m.rsplit("-", 1)[-1]) for m in hashing])
indices = np.unique([_hashing_index(str(m)) for m in hashing])
else:
indices = np.unique(hashing_indices)
cell_indices = np.resize(indices, n_cells)
Expand Down Expand Up @@ -324,7 +332,7 @@ def _assign_markers(
return node_umi_map

hashing_markers = markers[is_hashing]
index = np.array([int(m.rsplit("-", 1)[-1]) for m in hashing_markers])
index = np.array([_hashing_index(str(m)) for m in hashing_markers])
chosen = hashing_markers[index == hashing_index]
if chosen.size == 0:
return node_umi_map
Expand Down
3 changes: 2 additions & 1 deletion tests/common/data_generator/test_molecules.py
Original file line number Diff line number Diff line change
Expand Up @@ -11,6 +11,7 @@
from polars.testing import assert_frame_equal

from pixelator.pna.config import pna_config
from pixelator.pna.config.panel import split_hashing_marker_id
from tests.common.data_generator.molecules import (
_assign_markers,
_assign_umis,
Expand Down Expand Up @@ -154,7 +155,7 @@ def test_assign_markers_hashing_index(marker_panel):
by_index = defaultdict(list)
for marker, hashing in zip(markers, is_hashing):
if hashing:
by_index[int(marker.rsplit("-", 1)[-1])].append(marker)
by_index[int(split_hashing_marker_id(marker)[1])].append(marker)

# the chosen index's markers share the extra ~f and sit above the base
chosen = by_index[hashing_index]
Expand Down
47 changes: 47 additions & 0 deletions tests/pna/config/test_panel.py
Original file line number Diff line number Diff line change
Expand Up @@ -234,3 +234,50 @@ def test_panel_header_non_recoverable_yaml_still_fails():

with pytest.raises(yaml.YAMLError):
PNAAntibodyPanel.from_csv(tmp_file.name)


def _marker_row(marker_id: str, sequence: str, **extra) -> dict:
row = {
"marker_id": marker_id,
"control": False,
"sequence_1": sequence,
"sequence_2": sequence,
}
row.update(extra)
return row


def test_duplicate_marker_ids_are_rejected():
"""Marker ids must be unique even when sequences differ."""
frame = pd.DataFrame(
[_marker_row("CD3", "AAAA"), _marker_row("CD3", "CCCC")]
).set_index("marker_id")
with pytest.raises(AssertionError, match="marker_id were not unique"):
PNAAntibodyPanel(
frame,
AntibodyPanelMetadata(name="base", version="1.0.0"),
)


def test_hashing_marker_ids_need_a_numeric_suffix_and_must_not_nest():
"""Hashing ids end with -<digits> and must not collapse onto each other."""
missing_suffix = pd.DataFrame(
[_marker_row("B2M", "AAAA", sample_hashing=True)]
).set_index("marker_id")
with pytest.raises(AssertionError, match="must end with -<digits>"):
PNAAntibodyPanel(
missing_suffix,
AntibodyPanelMetadata(name="base", version="1.0.0"),
)

nested = pd.DataFrame(
[
_marker_row("B2M-1", "AAAA", sample_hashing=True),
_marker_row("B2M-1-1", "CCCC", sample_hashing=True),
]
).set_index("marker_id")
with pytest.raises(AssertionError, match="must not collapse"):
PNAAntibodyPanel(
nested,
AntibodyPanelMetadata(name="base", version="1.0.0"),
)
Loading
Loading