Atera breast cancer neighborhood exploration¶
Continue from the Atera Landscape–Clustergram tutorial.
Run its data preprocessing first; this notebook reuses the saved cell list,
10x cell types and tissue coordinates. Requires celldega>=0.26.0.
Two views explore the same sample at 75 µm: cell-type composition niches from hextiles, and overlap of cell-type alpha-shape footprints. Niche labels belong to hextiles; no niche attribute is added to individual cells.
from pathlib import Path
from urllib.request import Request, urlopen
import json
import celldega as dega
import numpy as np
import pandas as pd
import scanpy as sc
HEX_DIAMETER_UM = 75
ALPHA_SCALE_UM = 75
MIN_CELLS_PER_HEX = 5
N_NICHES = 8
root = next((p for p in [Path.cwd(), *Path.cwd().parents]
if (p / "pyproject.toml").is_file()), None)
data_dir = (root / "notebooks" if root else Path.cwd()) / "data/atera_preview_data_AnnDatas"
base_url = "https://raw.githubusercontent.com/cornhundred/DegaFiles_WTA_Preview_FFPE_Breast_Cancer_outs/main"
processed_path = data_dir / "WTA_Preview_FFPE_Breast_Cancer_outs_AnnData_processed.h5ad"
def cached_file(url, filename):
destination = data_dir / filename
if not destination.is_file():
with urlopen(Request(url, headers={"User-Agent": "celldega-example-notebook"})) as response:
destination.write_bytes(response.read())
return destination
# Neighborhoods need cell metadata, not the full expression matrix.
if not processed_path.is_file():
raise FileNotFoundError("Run the main Atera tutorial's preprocessing before this notebook.")
processed = sc.read_h5ad(processed_path, backed="r")
adata = processed[:, :0].to_memory()
processed.file.close()
groups_url = ("https://cf.10xgenomics.com/samples/atera/dev/WTA_Preview_FFPE_Breast_Cancer/"
"WTA_Preview_FFPE_Breast_Cancer_cell_groups.csv")
groups = pd.read_csv(cached_file(groups_url, "cell_groups.csv"), index_col="cell_id")
adata = adata[adata.obs_names.isin(groups.index)].copy()
adata.obs["cell_type"] = pd.Categorical(groups.loc[adata.obs_names, "group"])
palette = groups.groupby("group")["color"].agg(lambda colors: colors.mode()[0])
adata.uns["cell_type_colors"] = palette.loc[adata.obs["cell_type"].cat.categories].tolist()
# DegaFiles centroids are pixels; invert the recorded affine to get microns.
metadata = pd.read_parquet(cached_file(base_url + "/cell_metadata.parquet", "cell_metadata.parquet"))
transform = np.loadtxt(cached_file(base_url + "/micron_to_image_transform.csv", "micron_to_image_transform.csv"))
xy = np.vstack(metadata.set_index("name").loc[adata.obs_names, "geometry"])
adata.obsm["spatial"] = (np.c_[xy, np.ones(len(xy))] @ np.linalg.inv(transform).T)[:, :2]
print(f"{adata.n_obs:,} cells; {len(adata.obs['cell_type'].cat.categories)} cell types")
170,022 cells; 19 cell types
Hextile niches¶
Count cell types in each tile, discard tiles with fewer than five cells, and
cluster square-root proportions (Hellinger representation) with Celldega's
Matrix.cluster and cut_tree. Eight niches is an exploratory resolution.
The Composition bars pool cell counts across retained tiles of each niche.
The broad biological names describe this 75 µm, eight-niche result; the table
retains the original IDs and leading cell types so you can check them.
These are composition labels, not validated histological regions. Review the
names if you change the tile size or clustering resolution.
Click a column label to highlight every individual hextile of that niche. Click a row label to show that cell type across the tissue, irrespective of niche. Column dendrogram selections highlight groups of niches. The polygons remain separate hextiles, so you can also select one directly in the Landscape.
hextiles = dega.nbhd.NeighborhoodCollection(
gdf=dega.nbhd.generate_hextile(adata, diameter=HEX_DIAMETER_UM),
nbhd_type="hextile",
)
hextiles.calc_population(adata, category="cell_type", output="counts", min_cells=MIN_CELLS_PER_HEX)
population = hextiles.mod["population"]
counts = population.to_df()
fractions = counts.div(counts.sum(axis=1), axis=0)
niche_matrix = dega.clust.Matrix(np.sqrt(fractions).T, name="hextile_composition")
niche_matrix.cluster(dist_type="euclidean", linkage_type="ward")
population.obs["niche_id"] = pd.Categorical(
niche_matrix.cut_tree(axis="col", n_clusters=N_NICHES).map(lambda label: f"N{label}")
)
# Composition-based names for this 75 µm, eight-niche result.
niche_names = {
"N1": "High-grade DCIS",
"N2": "Mixed tumor–immune stroma",
"N3": "High-grade DCIS / stroma",
"N4": "Luminal-like DCIS",
"N5": "Myoepithelial / stromal mixture",
"N6": "Fibroblast-rich stroma",
"N7": "T cell–rich immune",
"N8": "Vascular-rich stroma",
}
assert set(population.obs["niche_id"].cat.categories) == set(niche_names), "Review niche names after changing N_NICHES."
population.obs["niche"] = pd.Categorical(
population.obs["niche_id"].map(niche_names), categories=list(niche_names.values()),
)
population.uns["niche_colors"] = sc.pl.palettes.default_20[:len(population.obs["niche"].cat.categories)]
hex_niches = dega.nbhd.NeighborhoodCollection(
gdf=dega.nbhd.hextile_niche(hextiles.gdf.rename_axis(None), population, category="niche", dissolve=False),
nbhd_type="hextile_niche",
)
niche_counts = counts.groupby(population.obs["niche"], observed=True).sum()
composition = dega.viz.Composition(
niche_counts, category="cell_type", adata=adata, normalized=True,
row_entity=json.dumps({"entity": "cell", "attr": "cell_type"}),
col_entity=json.dumps({"entity": "nbhd", "attr": "cat"}),
name="atera_niche_composition", width=500, height=450,
row_label_scale=0.3, col_label_scale=0.35,
)
print(f"{len(hextiles.obs):,} retained hextiles; {len(niche_counts)} composition niches")
# Percentages describe pooled cells, matching the Composition bars.
niche_fractions = niche_counts.div(niche_counts.sum(axis=1), axis=0)
niche_summary = pd.DataFrame([
{
"ID": niche_id,
"Niche": name,
"Leading cell types (pooled %)": "; ".join(
f"{cell_type} ({fraction:.0%})"
for cell_type, fraction in niche_fractions.loc[name].nlargest(3).items()
),
}
for niche_id, name in niche_names.items()
]).set_index("ID")
niche_summary.style.set_properties(**{"text-align": "left"})
Calculating NBP
9,640 retained hextiles; 8 composition niches
| Niche | Leading cell types (pooled %) | |
|---|---|---|
| ID | ||
| N1 | High-grade DCIS | 11q13 High Grade DCIS Cells (84%); Basal-like Structured DCIS Cells (7%); 11q13 High Grade DCIS Tumor Cells (G1/S) (2%) |
| N2 | Mixed tumor–immune stroma | 11q13 High Grade DCIS Cells (20%); CAFs, High Grade DCIS Associated (14%); Endothelial Cells (10%) |
| N3 | High-grade DCIS / stroma | 11q13 High Grade DCIS Cells (53%); Basal-like Structured DCIS Cells (8%); Endothelial Cells (6%) |
| N4 | Luminal-like DCIS | Luminal-like Amorphous DCIS Cells (85%); Myoepithelial Cells (7%); CAFs, Low Grade DCIS Associated (2%) |
| N5 | Myoepithelial / stromal mixture | Myoepithelial Cells (25%); CAFs, Low Grade DCIS Associated (25%); Luminal-like Amorphous DCIS Cells (22%) |
| N6 | Fibroblast-rich stroma | CAFs, Low Grade DCIS Associated (61%); CXCL14+ Fibroblasts (8%); T Lymphocytes (7%) |
| N7 | T cell–rich immune | T Lymphocytes (41%); Dendritic Cells (13%); CAFs, Low Grade DCIS Associated (10%) |
| N8 | Vascular-rich stroma | CAFs, Low Grade DCIS Associated (29%); Pericytes (22%); Endothelial Cells (19%) |
hex_landscape = dega.viz.Landscape(
base_url=base_url, technology="Xenium", transform=transform,
adata=adata, cell_attr=["cell_type"], cluster_attr="cell_type",
nbhd=hex_niches, landscape_state="nbhd", name="atera_hextile_niches",
)
hextile_view = dega.viz.spatial_clustergram(hex_landscape, composition)
hextile_view
Alpha-shape overlap¶
Make one footprint per cell type, retaining disconnected patches within its
MultiPolygon. ALPHA_SCALE_UM is the inverse-alpha shape scale in microns,
not a tile diameter. Empty footprints are omitted.
Calculate intersection areas with NeighborhoodCollection.calc_overlap,
then column-normalize with Matrix.norm: each column sums to one, including
self-overlap on the diagonal. A value is a share of the column's summed
intersection areas, not the fraction of its unique area covered by a cell
type; overlapping footprints can count the same area more than once.
Click any matrix tile, including a zero, to highlight both footprints and inspect their spatial overlap or separation. Row/column labels select one footprint; dendrogram selections highlight groups with similar overlap profiles. These footprints describe spatial mixing of cell types, not cell–cell contacts.
alpha_gdf = dega.nbhd.alpha_shape_cell_clusters(
adata, cat="cell_type", alphas=[ALPHA_SCALE_UM],
)
alpha_gdf = alpha_gdf.loc[~alpha_gdf.geometry.is_empty & (alpha_gdf.area > 0)].copy()
alpha_gdf["name"] = alpha_gdf["cat"].astype(str) # One neighborhood per cell type.
alpha_nbhds = dega.nbhd.NeighborhoodCollection(gdf=alpha_gdf, nbhd_type="cell_type_alpha_shape")
alpha_nbhds.obs["cell_type"] = alpha_nbhds.obs["cat"]
alpha_nbhds.calc_overlap(metric="intersection", key="intersection_area", category="cell_type")
overlap = alpha_nbhds.add_relation_modality("intersection_area", key="overlap")
overlap.var["cell_type"] = alpha_nbhds.obs["cell_type"]
overlap_matrix = dega.clust.Matrix(
collection=alpha_nbhds, color_by="overlap",
row_entity="nbhd", col_entity="nbhd",
row_attr=["cell_type"], col_attr=["cell_type"],
global_colors=palette.to_dict(), name="cell_type_neighborhood_overlap",
)
overlap_matrix.norm("col", by="total")
overlap_matrix.cluster()
overlap_clustergram = dega.viz.Clustergram(
matrix=overlap_matrix, name="atera_alpha_overlap", width=500, height=450,
row_label_scale=0.3, col_label_scale=0.3,
)
print(f"{len(alpha_nbhds.obs)} cell-type footprints; column sums: "
f"{overlap_matrix.data.sum().min():.3f}–{overlap_matrix.data.sum().max():.3f}")
Calculating NBN-O (intersection)
18 cell-type footprints; column sums: 1.000–1.000
alpha_landscape = dega.viz.Landscape(
base_url=base_url, technology="Xenium", transform=transform,
adata=adata, cell_attr=["cell_type"], cluster_attr="cell_type",
nbhd=alpha_nbhds, landscape_state="nbhd", name="atera_alpha_neighborhoods",
)
alpha_view = dega.viz.spatial_clustergram(alpha_landscape, overlap_clustergram)
alpha_view
Compare 50/75/100 µm scales and inspect the tissue to assess the niche names and overlap groups. The public annotations include mixed and spatially named groups; cell segmentation and annotation can influence both views.