Upstream Steps
This Notebook
Metacell calculation by SEACells (following this tutorial)
import os
import sys
import numpy as np
import pandas as pd
import scanpy as sc
import pickle
#Plotting
import matplotlib.pyplot as plt
import matplotlib
import seaborn as sns
#utils
#import ipynbname
from datetime import datetime
# SeaCell
import SEACells
#import custom functions
sys.path.append('../')
import functions as fn
# Some plotting aesthetics
%matplotlib inline
sns.set_style('ticks')
matplotlib.rcParams['figure.figsize'] = [3.5, 3.5]
matplotlib.rcParams['figure.dpi'] = 100
print("Scanpy version: ", sc.__version__)
print("Pandas version: ", pd.__version__)
print("SEACell version: ", SEACells.__version__)
Scanpy version: 1.10.2 Pandas version: 2.2.2 SEACell version: 0.3.3
sc.settings.verbosity = 3
sc.settings.set_figure_params(dpi=80)
input_file = '../../../../DataDir/ExternalData/SingleCellData/Wang_IIITrimester_adata.h5ad'
output_file = '../../../../DataDir/ExternalData/SingleCellData/Wang_IIITrimester_adataMetacells.h5ad'
print(datetime.now())
2026-03-11 12:45:30.693109
adata = sc.read(input_file)
adata
AnnData object with n_obs × n_vars = 22895 × 20266
obs: 'Auth_Sample_ID', 'Auth_Estimated_postconceptional_age_in_days', 'Auth_Group', 'Auth_Region', 'Auth_nCount_RNA', 'Auth_nFeature_RNA', 'Auth_ATAC_fragments_in_peaks', 'Auth_Scrublet_doublet_score', 'Auth_S.Score', 'Auth_G2M.Score', 'Auth_Class', 'Auth_Subclass', 'Auth_Type_updated', 'Auth_Cluster', 'Auth_cell_type_ontology_term_id', 'Auth_development_stage_ontology_term_id', 'Auth_donor_id', 'Auth_cell_type', 'Auth_sex', 'Auth_tissue', 'Auth_self_reported_ethnicity', 'Auth_development_stage', 'dataset_id', 'sample_id', 'age', 'stage', 'cell_label', 'brain_region', 'n_genes_by_counts', 'log1p_n_genes_by_counts', 'total_counts', 'log1p_total_counts', 'total_counts_mito', 'log1p_total_counts_mito', 'pct_counts_mito', 'total_counts_ribo', 'log1p_total_counts_ribo', 'pct_counts_ribo', 'log1p_gene_UMI_ratio', 'n_genes', 'n_counts', 'Leiden_02', 'Leiden_04', 'Leiden_06', 'Leiden_Sel'
var: 'gene_name', 'feature_is_filtered', 'feature_name', 'feature_length', 'feature_type', 'EnsembleCode', 'mito', 'ribo', 'n_cells_by_counts', 'mean_counts', 'log1p_mean_counts', 'pct_dropout_by_counts', 'total_counts', 'log1p_total_counts', 'n_cells', 'highly_variable', 'means', 'dispersions', 'dispersions_norm', 'highly_variable_nbatches', 'highly_variable_intersection'
uns: 'Auth_Group_colors', 'Leiden_02', 'Leiden_02_colors', 'Leiden_04', 'Leiden_04_colors', 'Leiden_06', 'Leiden_06_colors', 'Leiden_Sel_colors', 'age_colors', 'brain_region_colors', 'cell_label_colors', 'diffmap_evals', 'draw_graph', 'harmony', 'hvg', 'log1p', 'pca', 'sample_id_colors', 'umap'
obsm: 'X_draw_graph_fa_harmony', 'X_pca', 'X_pca_harmony', 'X_umap_harmony', 'X_umap_nocorr'
varm: 'PCs'
layers: 'counts', 'lognorm'
obsp: 'harmony_connectivities', 'harmony_distances', 'pca_connectivities', 'pca_distances'
adata.obsm
AxisArrays with keys: X_draw_graph_fa_harmony, X_pca, X_pca_harmony, X_umap_harmony, X_umap_nocorr
adata.X
<Compressed Sparse Row sparse matrix of dtype 'float32' with 61073579 stored elements and shape (22895, 20266)>
print(adata.X[40:45, 40:45])
<Compressed Sparse Row sparse matrix of dtype 'float32' with 0 stored elements and shape (5, 5)>
adata.layers['counts']
<Compressed Sparse Row sparse matrix of dtype 'uint16' with 61073579 stored elements and shape (22895, 20266)>
print(adata.layers['counts'][40:45, 40:45])
<Compressed Sparse Row sparse matrix of dtype 'uint16' with 0 stored elements and shape (5, 5)>
sc.pl.scatter(adata, basis='umap_nocorr', color=['sample_id', 'brain_region', 'cell_label'], frameon=False)
sc.pl.scatter(adata, basis='pca_harmony', color=['sample_id', 'brain_region', 'cell_label'], frameon=False)
sc.pl.scatter(adata, basis='umap_harmony', color=['sample_id', 'brain_region', 'cell_label'], frameon=False)
adata.obs.head(3)
| Auth_Sample_ID | Auth_Estimated_postconceptional_age_in_days | Auth_Group | Auth_Region | Auth_nCount_RNA | Auth_nFeature_RNA | Auth_ATAC_fragments_in_peaks | Auth_Scrublet_doublet_score | Auth_S.Score | Auth_G2M.Score | ... | total_counts_ribo | log1p_total_counts_ribo | pct_counts_ribo | log1p_gene_UMI_ratio | n_genes | n_counts | Leiden_02 | Leiden_04 | Leiden_06 | Leiden_Sel | |
|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|---|
| GW27-2-7-18-PFC_AAACAGCCAGCAATAA-1 | GW27-2-7-18-PFC | 178 | Third_trimester | PFC | 3956 | 1981 | 4426 | 0.133070 | -0.023445 | -0.034803 | ... | 62 | 4.143135 | 1.567636 | 0.405886 | 1980 | 3955 | 7 | 11 | 10 | 11 |
| GW27-2-7-18-PFC_AAACATGCAGAACCGA-1 | GW27-2-7-18-PFC | 178 | Third_trimester | PFC | 6482 | 2856 | 3843 | 0.126689 | -0.038244 | 0.000413 | ... | 731 | 6.595781 | 11.289575 | 0.364858 | 2851 | 6475 | 1 | 5 | 4 | 5 |
| GW27-2-7-18-PFC_AAACATGCATCATGTG-1 | GW27-2-7-18-PFC | 178 | Third_trimester | PFC | 3393 | 1680 | 736 | 0.089147 | 0.035548 | -0.007472 | ... | 490 | 6.196444 | 14.462810 | 0.402312 | 1678 | 3388 | 1 | 2 | 2 | 2 |
3 rows × 45 columns
Normalization, log-transformation, HVG and dimensionality reduction have been already performed.
We follow the workflow described in the SEACells tutorial
Parameters to be defined:
We then employ the SEACells.core.SEACells function to initialize the model.
## Core parameters
n_SEACells = round(adata.n_obs/75)
print(n_SEACells)
build_kernel_on = 'X_pca_harmony' # key in ad.obsm to use for computing metacells
# This would be replaced by 'X_svd' for ATAC data
## Additional parameters
n_waypoint_eigs = 10 # Number of eigenvalues to consider when initializing metacells
305
#help(SEACells.core.SEACells)
model = SEACells.core.SEACells(adata,
build_kernel_on=build_kernel_on,
n_SEACells=n_SEACells,
n_waypoint_eigs=n_waypoint_eigs,
convergence_epsilon = 1e-5,
use_gpu=True)
Welcome to SEACells GPU!
construct_kernel_matrix function constructs the kernel matrix from data matrix using PCA/SVD and nearest neighbors. Key parameters are:
model.construct_kernel_matrix(n_neighbors=20)
M = model.kernel_matrix
Computing kNN graph using scanpy NN ...
computing neighbors
finished: added to `.uns['neighbors']`
`.obsp['distances']`, distances for each pair of neighbors
`.obsp['connectivities']`, weighted adjacency matrix (0:00:27)
Computing radius for adaptive bandwidth kernel...
0%| | 0/22895 [00:00<?, ?it/s]
Making graph symmetric... Parameter graph_construction = union being used to build KNN graph... Computing RBF kernel...
0%| | 0/22895 [00:00<?, ?it/s]
Building similarity LIL matrix...
0%| | 0/22895 [00:00<?, ?it/s]
Constructing CSR matrix...
# Initialize archetypes
model.initialize_archetypes()
Building kernel on X_pca_harmony
Computing diffusion components from X_pca_harmony for waypoint initialization ...
computing neighbors
finished: added to `.uns['neighbors']`
`.obsp['distances']`, distances for each pair of neighbors
`.obsp['connectivities']`, weighted adjacency matrix (0:00:05)
Done.
Sampling waypoints ...
Done.
Selecting 280 cells from waypoint initialization.
Initializing residual matrix using greedy column selection
Initializing f and g...
100%|██████████| 35/35 [00:01<00:00, 27.46it/s]
Selecting 25 cells from greedy initialization.
adata
AnnData object with n_obs × n_vars = 22895 × 20266
obs: 'Auth_Sample_ID', 'Auth_Estimated_postconceptional_age_in_days', 'Auth_Group', 'Auth_Region', 'Auth_nCount_RNA', 'Auth_nFeature_RNA', 'Auth_ATAC_fragments_in_peaks', 'Auth_Scrublet_doublet_score', 'Auth_S.Score', 'Auth_G2M.Score', 'Auth_Class', 'Auth_Subclass', 'Auth_Type_updated', 'Auth_Cluster', 'Auth_cell_type_ontology_term_id', 'Auth_development_stage_ontology_term_id', 'Auth_donor_id', 'Auth_cell_type', 'Auth_sex', 'Auth_tissue', 'Auth_self_reported_ethnicity', 'Auth_development_stage', 'dataset_id', 'sample_id', 'age', 'stage', 'cell_label', 'brain_region', 'n_genes_by_counts', 'log1p_n_genes_by_counts', 'total_counts', 'log1p_total_counts', 'total_counts_mito', 'log1p_total_counts_mito', 'pct_counts_mito', 'total_counts_ribo', 'log1p_total_counts_ribo', 'pct_counts_ribo', 'log1p_gene_UMI_ratio', 'n_genes', 'n_counts', 'Leiden_02', 'Leiden_04', 'Leiden_06', 'Leiden_Sel'
var: 'gene_name', 'feature_is_filtered', 'feature_name', 'feature_length', 'feature_type', 'EnsembleCode', 'mito', 'ribo', 'n_cells_by_counts', 'mean_counts', 'log1p_mean_counts', 'pct_dropout_by_counts', 'total_counts', 'log1p_total_counts', 'n_cells', 'highly_variable', 'means', 'dispersions', 'dispersions_norm', 'highly_variable_nbatches', 'highly_variable_intersection'
uns: 'Auth_Group_colors', 'Leiden_02', 'Leiden_02_colors', 'Leiden_04', 'Leiden_04_colors', 'Leiden_06', 'Leiden_06_colors', 'Leiden_Sel_colors', 'age_colors', 'brain_region_colors', 'cell_label_colors', 'diffmap_evals', 'draw_graph', 'harmony', 'hvg', 'log1p', 'pca', 'sample_id_colors', 'umap', 'neighbors'
obsm: 'X_draw_graph_fa_harmony', 'X_pca', 'X_pca_harmony', 'X_umap_harmony', 'X_umap_nocorr'
varm: 'PCs'
layers: 'counts', 'lognorm'
obsp: 'harmony_connectivities', 'harmony_distances', 'pca_connectivities', 'pca_distances', 'distances', 'connectivities'
adata.obsm['X_umap'] = adata.obsm['X_umap_harmony'].copy()
# Plot the initilization to ensure they are spread across phenotypic space
SEACells.plot.plot_initialization(adata, model)
We apply the model.fit function.
model.fit(min_iter=10, max_iter=70)
model
Randomly initialized A matrix. Starting iteration 1. Completed iteration 1. Starting iteration 10. Completed iteration 10. Starting iteration 20. Completed iteration 20. Starting iteration 30. Completed iteration 30. Converged after 34 iterations.
<SEACells.gpu.SEACellsGPU at 0x155432339480>
# Plot model convergence
model.plot_convergence()
#model.A_
plt.figure(figsize=(4,3))
sns.distplot((model.A_.T > 0.1).sum(axis=1), kde=False)
plt.title(f'Non-trivial (> 0.1) assignments per cell')
plt.xlabel('# Non-trivial SEACell Assignments')
plt.ylabel('# Cells')
plt.show()
plt.figure(figsize=(4,3))
b = np.partition(model.A_.T, -5)
sns.heatmap(np.sort(b[:,-5:])[:, ::-1], cmap='viridis', vmin=0)
plt.title('Strength of top 5 strongest assignments')
plt.xlabel('$n^{th}$ strongest assignment')
plt.show()
/tmp/ipykernel_138619/2880491417.py:4: UserWarning: `distplot` is a deprecated function and will be removed in seaborn v0.14.0. Please adapt your code to use either `displot` (a figure-level function with similar flexibility) or `histplot` (an axes-level function for histograms). For a guide to updating your code to use the new functions, please see https://gist.github.com/mwaskom/de44147ed2974457ad6372750bbe5751 sns.distplot((model.A_.T > 0.1).sum(axis=1), kde=False)
labels,weights = model.get_soft_assignments()
#labels.head()
We will use hard assignment of each cell to a single metacell to define our metacells for downstream steps. Hard assingment can be retreived in two ways:
adata.obs[['SEACell']].head()
# Alternatively:
# model.get_hard_assignments().head()
| SEACell | |
|---|---|
| index | |
| GW27-2-7-18-PFC_AAACAGCCAGCAATAA-1 | SEACell-188 |
| GW27-2-7-18-PFC_AAACATGCAGAACCGA-1 | SEACell-79 |
| GW27-2-7-18-PFC_AAACATGCATCATGTG-1 | SEACell-121 |
| GW27-2-7-18-PFC_AAACCAACAACAGCCT-1 | SEACell-43 |
| GW27-2-7-18-PFC_AAACCGGCATGGTTAT-1 | SEACell-281 |
# By default, ad.raw is used for summarization. Other layers present in the anndata can be specified using the parameter summarize_layer parameter
SEA_adata = SEACells.core.summarize_by_SEACell(adata, SEACells_label='SEACell', summarize_layer='counts', celltype_label='cell_label')
SEA_adata
100%|██████████| 305/305 [00:04<00:00, 67.69it/s] /env/lib/python3.10/site-packages/anndata/_core/anndata.py:430: FutureWarning: The dtype argument is deprecated and will be removed in late 2024. warnings.warn(
AnnData object with n_obs × n_vars = 305 × 20266
obs: 'cell_label', 'cell_label_purity'
layers: 'raw'
SEA_adata.obs.head()
| cell_label | cell_label_purity | |
|---|---|---|
| SEACell-188 | Oligo | 0.944444 |
| SEACell-79 | InN | 0.529412 |
| SEACell-121 | ExN | 0.989950 |
| SEACell-43 | ExN | 0.994286 |
| SEACell-281 | ExN | 1.000000 |
print(SEA_adata.X[40:43, 40:43])
<Compressed Sparse Row sparse matrix of dtype 'float64' with 6 stored elements and shape (3, 3)> Coords Values (0, 1) 2.0 (0, 2) 2.0 (1, 1) 1.0 (1, 2) 1.0 (2, 0) 1.0 (2, 2) 1.0
adata.obs['brain_region']
index
GW27-2-7-18-PFC_AAACAGCCAGCAATAA-1 PreFrontalCortex
GW27-2-7-18-PFC_AAACATGCAGAACCGA-1 PreFrontalCortex
GW27-2-7-18-PFC_AAACATGCATCATGTG-1 PreFrontalCortex
GW27-2-7-18-PFC_AAACCAACAACAGCCT-1 PreFrontalCortex
GW27-2-7-18-PFC_AAACCGGCATGGTTAT-1 PreFrontalCortex
...
NIH-M2837-BA10-2_TTTGTGTTCATGAAGG-1 PreFrontalCortex
NIH-M2837-BA10-2_TTTGTGTTCATTACAG-1 PreFrontalCortex
NIH-M2837-BA10-2_TTTGTGTTCCCATAAA-1 PreFrontalCortex
NIH-M2837-BA10-2_TTTGTGTTCGCTAAGT-1 PreFrontalCortex
NIH-M2837-BA10-2_TTTGTTGGTTTAGTCC-1 PreFrontalCortex
Name: brain_region, Length: 22895, dtype: category
Categories (2, object): ['PreFrontalCortex', 'VisualCortex']
top_sampleid = adata.obs['sample_id'].groupby(adata.obs['SEACell']).value_counts().groupby(level=0, group_keys=False).head(1)
SEA_adata.obs['agg_sample_id'] = top_sampleid[SEA_adata.obs_names].index.get_level_values(1)
top_brain_region = adata.obs['brain_region'].groupby(adata.obs['SEACell']).value_counts().groupby(level=0, group_keys=False).head(1)
SEA_adata.obs['Aggregated_brain_region'] = top_brain_region[SEA_adata.obs_names].index.get_level_values(1)
SEA_adata.obs["Auth_Group"] = 'Third_trimester'
SEA_adata.obs.head()
| cell_label | cell_label_purity | agg_sample_id | Aggregated_brain_region | Auth_Group | |
|---|---|---|---|---|---|
| SEACell-188 | Oligo | 0.944444 | NIH-M1154-BA10-2 | PreFrontalCortex | Third_trimester |
| SEACell-79 | InN | 0.529412 | NIH-4267-BA10-2 | PreFrontalCortex | Third_trimester |
| SEACell-121 | ExN | 0.989950 | NIH-M2837-BA10-2 | PreFrontalCortex | Third_trimester |
| SEACell-43 | ExN | 0.994286 | NIH-4267-BA10-2 | PreFrontalCortex | Third_trimester |
| SEACell-281 | ExN | 1.000000 | NIH-4267-BA10-2 | PreFrontalCortex | Third_trimester |
plot_2D(): visualize metacell assignments on UMAP or any 2-dimensional embedding froma adata.obsm. Plots can also be coloured by metacell assignment. The visualization allows to assess the representativeness of the calculated metacells: ability to represent the global structure of the original dataset. Better representation corresponds to a more uniform coverage of the dataset.
plot_SEACell_sizes(): distribution of number of cells assigned to each metacell. Given to the non-uniform distribution of cell type density, metacells are expected to vary in size: larger metacells are retrieved from denser regions, and smaller metacells from sparser ones. However outlier values may require further inspection.
SEACells.plot.plot_2D(adata, key='X_umap', colour_metacells=False)
/env/lib/python3.10/site-packages/SEACells/plot.py:63: FutureWarning: The default of observed=False is deprecated and will be changed to True in a future version of pandas. Pass observed=False to retain current behavior or observed=True to adopt the future default and silence this warning.
mcs = umap.groupby("SEACell").mean().reset_index()
/env/lib/python3.10/site-packages/seaborn/relational.py:438: UserWarning: No data for colormapping provided via 'c'. Parameters 'cmap' will be ignored
points = ax.scatter(x=x, y=y, **kws)
/env/lib/python3.10/site-packages/seaborn/relational.py:438: UserWarning: No data for colormapping provided via 'c'. Parameters 'cmap' will be ignored
points = ax.scatter(x=x, y=y, **kws)
SEACells.plot.plot_2D(adata, key='X_umap', colour_metacells=True)
/env/lib/python3.10/site-packages/SEACells/plot.py:63: FutureWarning: The default of observed=False is deprecated and will be changed to True in a future version of pandas. Pass observed=False to retain current behavior or observed=True to adopt the future default and silence this warning.
mcs = umap.groupby("SEACell").mean().reset_index()
/env/lib/python3.10/site-packages/seaborn/relational.py:438: UserWarning: No data for colormapping provided via 'c'. Parameters 'cmap' will be ignored
points = ax.scatter(x=x, y=y, **kws)
/env/lib/python3.10/site-packages/seaborn/relational.py:438: UserWarning: No data for colormapping provided via 'c'. Parameters 'cmap' will be ignored
points = ax.scatter(x=x, y=y, **kws)
SEACells.plot.plot_SEACell_sizes(adata, bins=10, figsize=(10, 5))
/env/lib/python3.10/site-packages/SEACells/plot.py:130: UserWarning:
`distplot` is a deprecated function and will be removed in seaborn v0.14.0.
Please adapt your code to use either `displot` (a figure-level function with
similar flexibility) or `histplot` (an axes-level function for histograms).
For a guide to updating your code to use the new functions, please see
https://gist.github.com/mwaskom/de44147ed2974457ad6372750bbe5751
sns.distplot(label_df.groupby("SEACell").count().iloc[:, 0], bins=bins)
| size | |
|---|---|
| SEACell | |
| SEACell-0 | 124 |
| SEACell-1 | 100 |
| SEACell-10 | 138 |
| SEACell-100 | 76 |
| SEACell-101 | 60 |
| ... | ... |
| SEACell-95 | 34 |
| SEACell-96 | 43 |
| SEACell-97 | 87 |
| SEACell-98 | 55 |
| SEACell-99 | 51 |
305 rows × 1 columns
Purity: compute_celltype_purity(ad, col_name) computes the purity of different celltype labels within a SEACell metacell.
Compactness: per-SEAcell variance in diffusion components (typically 'X_pca' for RNA). Lower values of compactness suggest more compact/lower variance metacells.
Separation: distance between a SEACell and its nth_nbr. If cluster is provided as a string, e.g. 'celltype', nearest neighbors are restricted to have the same celltype value. Higher values of separation suggest better distinction between metacells.
SEACell_purity = SEACells.evaluate.compute_celltype_purity(adata, 'cell_label')
plt.figure(figsize=(6,4))
sns.boxplot(data=SEACell_purity, y='cell_label_purity')
plt.title('cell_label Purity')
sns.despine()
plt.show()
plt.close()
SEACell_purity.head()
| cell_label | cell_label_purity | |
|---|---|---|
| SEACell | ||
| SEACell-0 | InN | 0.943548 |
| SEACell-1 | InN | 0.960000 |
| SEACell-10 | OPC | 1.000000 |
| SEACell-100 | ExN | 1.000000 |
| SEACell-101 | ExN | 0.983333 |
compactness = SEACells.evaluate.compactness(adata, 'X_pca_harmony')
plt.figure(figsize=(6,4))
sns.boxplot(data=compactness, y='compactness')
plt.title('Compactness')
sns.despine()
plt.show()
plt.close()
compactness.head()
computing neighbors
finished: added to `.uns['neighbors']`
`.obsp['distances']`, distances for each pair of neighbors
`.obsp['connectivities']`, weighted adjacency matrix (0:00:06)
| compactness | |
|---|---|
| SEACell | |
| SEACell-0 | 0.002635 |
| SEACell-1 | 0.000785 |
| SEACell-10 | 0.001359 |
| SEACell-100 | 0.002013 |
| SEACell-101 | 0.008242 |
separation = SEACells.evaluate.separation(adata, 'X_pca_harmony')
plt.figure(figsize=(6,4))
sns.boxplot(data=separation, y='separation')
plt.title('Separation')
sns.despine()
plt.show()
plt.close()
separation.head()
computing neighbors
finished: added to `.uns['neighbors']`
`.obsp['distances']`, distances for each pair of neighbors
`.obsp['connectivities']`, weighted adjacency matrix (0:00:05)
| separation | |
|---|---|
| SEACell | |
| SEACell-0 | 0.093489 |
| SEACell-1 | 0.075247 |
| SEACell-10 | 0.019242 |
| SEACell-100 | 0.097006 |
| SEACell-101 | 0.225582 |
SEA_adata.layers['counts'] = SEA_adata.X.copy()
sc.pp.normalize_total(SEA_adata)
sc.pp.log1p(SEA_adata)
SEA_adata.layers['lognorm'] = SEA_adata.X.copy()
sc.pp.highly_variable_genes(SEA_adata, inplace=True, batch_key="agg_sample_id")
sc.tl.pca(SEA_adata, use_highly_variable=True)
sc.pp.neighbors(SEA_adata, use_rep='X_pca')
sc.tl.umap(SEA_adata)
normalizing counts per cell
finished (0:00:00)
extracting highly variable genes
finished (0:00:00)
--> added
'highly_variable', boolean vector (adata.var)
'means', float vector (adata.var)
'dispersions', float vector (adata.var)
'dispersions_norm', float vector (adata.var)
computing PCA
with n_comps=50
/env/lib/python3.10/site-packages/scanpy/preprocessing/_pca.py:385: FutureWarning: Argument `use_highly_variable` is deprecated, consider using the mask argument. Use_highly_variable=True can be called through mask_var="highly_variable". Use_highly_variable=False can be called through mask_var=None warn(msg, FutureWarning)
finished (0:00:00)
computing neighbors
finished: added to `.uns['neighbors']`
`.obsp['distances']`, distances for each pair of neighbors
`.obsp['connectivities']`, weighted adjacency matrix (0:00:00)
computing UMAP
finished: added
'X_umap', UMAP coordinates (adata.obsm) (0:00:01)
sc.pl.umap(SEA_adata, color=['cell_label'], s=75)
SEA_adata
AnnData object with n_obs × n_vars = 305 × 20266
obs: 'cell_label', 'cell_label_purity', 'agg_sample_id', 'Aggregated_brain_region', 'Auth_Group'
var: 'highly_variable', 'means', 'dispersions', 'dispersions_norm', 'highly_variable_nbatches', 'highly_variable_intersection'
uns: 'log1p', 'hvg', 'pca', 'neighbors', 'umap', 'cell_label_colors'
obsm: 'X_pca', 'X_umap'
varm: 'PCs'
layers: 'raw', 'counts', 'lognorm'
obsp: 'distances', 'connectivities'
sc.pl.umap(SEA_adata, color=['agg_sample_id'], s=75)
sc.pl.umap(SEA_adata, color=['Aggregated_brain_region'], s=75)
sc.pl.umap(SEA_adata, color=['cell_label_purity'], s=75)
SEA_adata.obs.columns
Index(['cell_label', 'cell_label_purity', 'agg_sample_id',
'Aggregated_brain_region', 'Auth_Group'],
dtype='object')
SEA_adata
AnnData object with n_obs × n_vars = 305 × 20266
obs: 'cell_label', 'cell_label_purity', 'agg_sample_id', 'Aggregated_brain_region', 'Auth_Group'
var: 'highly_variable', 'means', 'dispersions', 'dispersions_norm', 'highly_variable_nbatches', 'highly_variable_intersection'
uns: 'log1p', 'hvg', 'pca', 'neighbors', 'umap', 'cell_label_colors', 'agg_sample_id_colors', 'Aggregated_brain_region_colors'
obsm: 'X_pca', 'X_umap'
varm: 'PCs'
layers: 'raw', 'counts', 'lognorm'
obsp: 'distances', 'connectivities'
SEA_adata.X
<Compressed Sparse Row sparse matrix of dtype 'float64' with 4576358 stored elements and shape (305, 20266)>
SEA_adata.write(output_file)
%%bash
# save also html and python versions for git
jupyter nbconvert SEACellsWang_ThirdTrimester.ipynb --to="python" --output="SEACellsWang_ThirdTrimester"
jupyter nbconvert SEACellsWang_ThirdTrimester.ipynb --to="html" --output="SEACellsWang_ThirdTrimester"
[NbConvertApp] Converting notebook SEACellsWang_ThirdTrimester.ipynb to python [NbConvertApp] Writing 13122 bytes to SEACellsWang_ThirdTrimester.py [NbConvertApp] Converting notebook SEACellsWang_ThirdTrimester.ipynb to html [NbConvertApp] WARNING | Alternative text is missing on 4 image(s). [NbConvertApp] Writing 3203293 bytes to SEACellsWang_ThirdTrimester.html
print(datetime.now())
2026-03-11 13:12:36.702976