Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 12 additions & 6 deletions src/pixelator/pna/anndata.py
Original file line number Diff line number Diff line change
Expand Up @@ -41,14 +41,19 @@ def add_panel_information(adata: AnnData, panel: PNAAntibodyPanel) -> AnnData:


def pna_edgelist_to_anndata(
pixel_connection: duckdb.DuckDBPyConnection, panel: PNAAntibodyPanel
pixel_connection: duckdb.DuckDBPyConnection,
panel: PNAAntibodyPanel,
markers: list[str] | None = None,
) -> AnnData:
"""Build an AnnData object from a DuckDB connection to a pixel file and a panel object.

Args:
pixel_connection: A DuckDB connection to a pixel file. The connection must contain an 'edgelist' table
with the required columns (e.g., component, marker_1, marker_2, umi1, umi2, read_count).
panel: The antibody panel object containing marker metadata.
markers: Marker ids to use as the count-matrix columns. Defaults to
every marker on ``panel``. Sample calling passes the markers left
after hashing clones collapse to their base name.

Returns:
An AnnData object with counts and panel information.
Expand All @@ -73,10 +78,11 @@ def pna_edgelist_to_anndata(
.tolist()
)

marker_list = list(panel.markers if markers is None else markers)
n_components = len(components)
n_markers = len(panel.markers)
n_markers = len(marker_list)
component_to_idx = {c: i for i, c in enumerate(components)}
marker_to_idx = {m: i for i, m in enumerate(panel.markers)}
marker_to_idx = {m: i for i, m in enumerate(marker_list)}

X = np.zeros((n_components, n_markers), dtype=np.uint32)
n_umi1_arr = np.zeros(n_components, dtype=np.uint64)
Expand Down Expand Up @@ -152,7 +158,7 @@ def pna_edgelist_to_anndata(
node_counts_df = pd.DataFrame(
X,
index=component_index,
columns=pd.Index(panel.markers, name="marker_id"),
columns=pd.Index(marker_list, name="marker_id"),
)

logger.debug("Computing component metrics.")
Expand Down Expand Up @@ -182,7 +188,7 @@ def pna_edgelist_to_anndata(

logger.debug("Computing antibody metrics.")
antibody_metrics_df = calculate_antibody_metrics(counts_df=node_counts_df)
antibody_metrics_df = antibody_metrics_df.reindex(index=panel.markers, fill_value=0)
antibody_metrics_df = antibody_metrics_df.reindex(index=marker_list, fill_value=0)
antibody_metrics_df.index.name = "marker_id"
# Do a dtype conversion of the columns here since AnnData cannot handle
# a pyarrow arrays.
Expand All @@ -201,7 +207,7 @@ def pna_edgelist_to_anndata(
adata = add_panel_information(adata, panel)

total_marker_counts = node_counts_df.sum(axis=1)
isotype_markers = adata.var[adata.var["control"]].index
isotype_markers = panel.df.index[panel.df["control"]]

Copy link
Copy Markdown

Choose a reason for hiding this comment

The reason will be displayed to describe this comment to others. Learn more.

Off-panel markers break panel reload

Medium Severity

pna_edgelist_to_anndata can now put collapsed hashing bases that are not on the panel into var. add_panel_information left-joins those rows, so control and sequences become missing. from_adata then treats every var row as a panel row and validation fails, so from_pxl_dataset cannot reload a dehashed file that kept an off-panel base.

Additional Locations (2)
Fix in Cursor Fix in Web

Reviewed by Cursor Bugbot for commit 3e00755. Configure here.

isotype_counts = node_counts_df[isotype_markers].sum(axis=1)
adata.obs["isotype_fraction"] = isotype_counts / total_marker_counts

Expand Down
10 changes: 7 additions & 3 deletions src/pixelator/pna/cli/sample_calling.py
Original file line number Diff line number Diff line change
Expand Up @@ -118,9 +118,13 @@ def sample_calling_cli(
pool_metadata=PxlFile(Path(input_pxl_file)).metadata(),
)
return
hashing_antibodies_in_panel = set(
panel_info.df[panel_info.df["sample_hashing"] == "yes"].index.to_list()
)
if "sample_hashing" not in panel_info.df.columns:
raise ValueError(
"Sample calling requires a sample_hashing column on the panel "
"so hashing markers can be identified. This panel has no "
"sample_hashing column."
)
hashing_antibodies_in_panel = panel_info.hashing_marker_ids
samplesheet_df = pl.read_csv(samplesheet)
_reject_reserved_samplesheet_names(
samplesheet_df["sample"].to_list(),
Expand Down
Loading
Loading