Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
9 changes: 8 additions & 1 deletion CHANGELOG.md
Original file line number Diff line number Diff line change
Expand Up @@ -8,14 +8,21 @@ and this project adheres to [Semantic Versioning][].
[keep a changelog]: https://keepachangelog.com/en/1.0.0/
[semantic versioning]: https://semver.org/spec/v2.0.0.html

## Unreleased
## [0.4.0] - 2026-10-07

### Added
- `pycea.datasets.colgan26` and `pycea.datasets.yu26` load mouse embryo lineage tracing datasets.
- Added a mouse embryogenesis tutorial using the `colgan26` dataset (`docs/notebooks/mouse-embryo.ipynb`).
- `pycea.tl.ancestral_linkage` now stores `tdata.uns['{key_added}_symmetrized_linkage_stats']` when `symmetrize` is not `False` and `test='permutation'`: a table with one row per unordered category pair giving the symmetrized value, permuted value, z-score, and a p-value for the symmetrized linkage.

### Changed
- `pycea.tl.ancestral_states` supports boolean data (`obs` columns or `obsm` arrays). For `method='mean'` and `'sum'`, `True` and `False` are treated as 1 and 0, giving the fraction and number of `True` leaves.
- `pycea.tl.ancestral_linkage` in single-target mode stores per-cell results in `tdata.obs['{key_added}_linkage']` when `key_added` is specified, instead of `tdata.obs['{target}_linkage']`.
- `depth_key` now defaults to `tdata.uns['default_depth']` (falling back to `'depth'`) in `pycea.tl.clades`, `n_extant`, `tree_distance`, `tree_neighbors`, `ancestral_linkage` and `fitness`, matching `pycea.pl`.

### Fixed
- `pycea.tl.tree_neighbors` with `metric='path'` and `n_neighbors` now returns the closest leaves. Previously leaves were collected on discovery rather than in distance order, so farther leaves could displace closer ones. Ties in distance are now also broken randomly.
- `pycea.tl.tree_neighbors` with a single observation in `obs` now marks the neighbors of that observation in `tdata.obs['{key_added}_neighbors']` instead of only the observation itself.

## [0.3.0] - 2026-07-08

Expand Down
2 changes: 2 additions & 0 deletions docs/api.md
Original file line number Diff line number Diff line change
Expand Up @@ -84,4 +84,6 @@
datasets.packer19
datasets.yang22
datasets.koblan25
datasets.colgan26
datasets.yu26
```
1 change: 1 addition & 0 deletions docs/index.md
Original file line number Diff line number Diff line change
Expand Up @@ -9,6 +9,7 @@
notebooks/getting-started
notebooks/plotting
notebooks/growth-dynamics
notebooks/mouse-embryo

api.md
changelog.md
Expand Down
8 changes: 2 additions & 6 deletions docs/notebooks/growth-dynamics.ipynb
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,7 @@
},
{
"cell_type": "code",
"execution_count": 2,
"execution_count": null,
"id": "df1a3755",
"metadata": {},
"outputs": [],
Expand All @@ -30,11 +30,7 @@
"plt.rcParams[\"figure.dpi\"] = 600\n",
"edit_cmap = mcolors.ListedColormap(\n",
" [\"white\", \"lightgray\", \"#CD2626\", \"#E69F00\", \"#FFE600\", \"#009E73\", \"#83A4FF\", \"#1874CD\", \"#8E0496\", \"#DB65D2\"]\n",
")\n",
"\n",
"# Auto reload\n",
"%load_ext autoreload\n",
"%autoreload 2"
")"
]
},
{
Expand Down
845 changes: 845 additions & 0 deletions docs/notebooks/mouse-embryo.ipynb

Large diffs are not rendered by default.

16 changes: 16 additions & 0 deletions docs/references.bib
Original file line number Diff line number Diff line change
Expand Up @@ -117,3 +117,19 @@ @article{Quinn_2021
author = {Quinn, Jeffrey J. and Jones, Matthew G. and Okimoto, Ross A. and Nanjo, Shigeki and Chan, Michelle M. and Yosef, Nir and Bivona, Trever G. and Weissman, Jonathan S.},

}

@misc{Colgan_2026,
title = {Comprehensive {Lineage} {Tracing} {Maps} the {Landscape} of {Cell} {Fate} {Decisions} in {Mouse} {Embryogenesis}},
doi = {10.64898/2026.05.07.722278},
publisher = {bioRxiv},
author = {Colgan, William N. and Koblan, Luke W. and Villagrana, J. and Hou, T.-C. J. and Wang, M. and Gowri, G. and Chandler, W. and Sepulveda, L. A. and Ciftci, D. and Smolyar, K. and Yost, Kathryn E. and Young, A. and Wittler, L. and Markoulaki, S. and Loh, K. M. and Zhuang, Xiaowei and Yosef, Nir and Smith, Z. D. and Weissman, Jonathan S.},
year = {2026},
}

@misc{Yu_2026,
title = {A {DNA} {Typewriter} records the cell lineage history of a mouse, from zygote to late organogenesis},
doi = {10.64898/2026.07.29.741625},
publisher = {bioRxiv},
author = {Yu, Q. and Kim, H. and Seidel, S. and Acosta-Clark, J. F. and Martin, B. K. and O'Connor, K. and Daza, R. M. and Gasperini, M. and Nathans, J. F. and Lam, M. and Gamo, E. and Vijay Kumar, S. and Kuo, L. and Lalanne, J.-B. and Simeonov, K. and Pepper, M. and Trapnell, C. and Gray, J. M. and Choi, J. and Qiu, C. and Shendure, J.},
year = {2026},
}
3 changes: 2 additions & 1 deletion pyproject.toml
Original file line number Diff line number Diff line change
Expand Up @@ -4,7 +4,7 @@ requires = [ "hatchling" ]

[project]
name = "pycea-lineage"
version = "0.3.0"
version = "0.4.0"
description = "Scverse lineage tracing toolkit"
readme = "README.md"
license = { file = "LICENSE" }
Expand Down Expand Up @@ -152,6 +152,7 @@ addopts = [
]
markers = [
"internet: tests which rely on internet resources (enable with `--internet-tests`)",
"slow: slow tests, e.g. large dataset downloads (enable with `--slow-tests`)",
]

[tool.coverage]
Expand Down
2 changes: 1 addition & 1 deletion src/pycea/datasets/__init__.py
Original file line number Diff line number Diff line change
@@ -1 +1 @@
from .datasets import koblan25, packer19, yang22
from .datasets import colgan26, koblan25, packer19, yang22, yu26
74 changes: 73 additions & 1 deletion src/pycea/datasets/datasets.py
Original file line number Diff line number Diff line change
Expand Up @@ -13,7 +13,7 @@
from os import PathLike

DATASET_DIR = "~/.treedata/datasets"
ZENODO_DOI = "15750529" # Needs to be updated if the dataset is changed
ZENODO_DOI = "23195576" # Needs to be updated if the dataset is changed


def _load_dataset(
Expand Down Expand Up @@ -151,3 +151,75 @@ def koblan25(experiment: str = "tumor", cache_dir: PathLike | str = DATASET_DIR)
backup_url=f"https://zenodo.org/records/{ZENODO_DOI}/files/koblan25_{experiment}.h5td?download=1",
)
return tdata


def colgan26(embryos: str | list[str] | None = None, cache_dir: PathLike | str = DATASET_DIR) -> td.TreeData:
"""Comprehensive lineage tracing of mouse embryogenesis from E7.5 to E10.0 :cite:p:`Colgan_2026`.

This study uses PEtracer to record early development in chimeric mouse embryos. The resulting lineage trees resolve
~75% of cell divisions across more than 1.4 million cells from 16 replicate embryos collected at half-day
intervals from E7.5 to E10.0. This dataset contains the cells from all embryos with their
annotations, a UMAP embedding, and lineage trees (one for each seeding clone, e.g., "E7.5-R1-C1").
Gene expression and character matricies are not included but can be downloaded from
[zenodo](https://doi.org/10.5281/zenodo.19892784). Nodes in each tree have a ``time`` attribute based on
molecular clock-based estimates of branch lengths.

Parameters
----------
embryos
The embryos to load, e.g. "E8.5-R1" (stage and replicate). If None, all 16 embryos are loaded.
cache_dir
The directory where the datasets are cached. Default is `~/.treedata/datasets`.

Returns
-------
TreeData object.

"""
tdata = _load_dataset(
"colgan26.h5td",
cache_dir=cache_dir,
backup_url=f"https://zenodo.org/records/{ZENODO_DOI}/files/colgan26.h5td?download=1",
)
if embryos is not None:
if isinstance(embryos, str):
embryos = [embryos]
elif not isinstance(embryos, list):
raise ValueError("embryos must be a string or a list of strings.")
print(f"Subsetting to embryos: {', '.join(embryos)}")
tdata = tdata[tdata.obs["embryo"].isin(embryos)].copy()
keys_to_delete = [key for key, value in tdata.obst.items() if value.size() == 0]
for key in keys_to_delete:
del tdata.obst[key]
return tdata


def yu26(cache_dir: PathLike | str = DATASET_DIR) -> td.TreeData:
"""DNA Typewriter lineage tracing of a mouse from zygote to late organogenesis :cite:p:`Yu_2026`.

In this study, DNA Typewriter, a sequential molecular recorder, was used to record the division history of a
mouse over nearly two weeks of development. From a single E13.5 embryo, a time-calibrated lineage tree was
reconstructed for 655,701 cells. This dataset contains the cell annotations, a UMAP embedding,
and the lineage tree with a ``time`` attribute on each node. Gene expression is not included but can be
downloaded from [GEO](https://www.ncbi.nlm.nih.gov/geo/query/acc.cgi?acc=GSE341627).

This dataset does not include the low-detection cells scaffolded onto the lineage tree due to the
high error rate of the distance based placement method used by :cite:p:`Yu_2026`. The inferred node times
should be used with caution as the editing rate was not constant over time violating the molecular clock
assumption.

Parameters
----------
cache_dir
The directory where the datasets are cached. Default is `~/.treedata/datasets`.

Returns
-------
TreeData object.

"""
return _load_dataset(
"yu26.h5td",
cache_dir=cache_dir,
backup_url=f"https://zenodo.org/records/{ZENODO_DOI}/files/yu26.h5td?download=1",
)
9 changes: 8 additions & 1 deletion src/pycea/pl/plot_tree.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,13 @@
from matplotlib.axes import Axes
from matplotlib.collections import LineCollection

from pycea.utils import _check_tree_overlap, get_keyed_edge_data, get_keyed_node_data, get_keyed_obs_data, get_trees
from pycea.utils import (
_check_tree_overlap,
get_keyed_edge_data,
get_keyed_node_data,
get_keyed_obs_data,
get_trees,
)

from ._docs import _doc_params, doc_common_plot_args
from ._legend import _categorical_legend, _cbar_legend, _render_legends
Expand Down Expand Up @@ -424,6 +430,7 @@ def nodes(
raise ValueError("Invalid style value. Must be a marker name, or an str specifying an attribute of the nodes.")
# Apply outline
if outline_width is not None:

def _outline_edgecolors(face_colors):
if isinstance(face_colors, str):
return "black"
Expand Down
37 changes: 22 additions & 15 deletions src/pycea/tl/ancestral_linkage.py
Original file line number Diff line number Diff line change
Expand Up @@ -14,7 +14,7 @@
import treedata as td
from tqdm import tqdm

from pycea.utils import _check_tree_overlap, check_tree_has_key, get_leaves, get_trees
from pycea.utils import _check_tree_overlap, check_tree_has_key, get_depth_key, get_leaves, get_trees

from ._aggregators import _get_aggregator
from ._metrics import _TreeMetric
Expand Down Expand Up @@ -631,7 +631,7 @@ def ancestral_linkage(
n_permutations: int = 100,
n_threads: int | None = None,
by_tree: bool = False,
depth_key: str = "depth",
depth_key: str | None = None,
random_state: int | None = None,
key_added: str | None = None,
tree: str | Sequence[str] | None = None,
Expand All @@ -655,7 +655,7 @@ def ancestral_linkage(
n_permutations: int = 100,
n_threads: int | None = None,
by_tree: bool = False,
depth_key: str = "depth",
depth_key: str | None = None,
random_state: int | None = None,
key_added: str | None = None,
tree: str | Sequence[str] | None = None,
Expand All @@ -678,7 +678,7 @@ def ancestral_linkage(
n_permutations: int = 100,
n_threads: int | None = None,
by_tree: bool = False,
depth_key: str = "depth",
depth_key: str | None = None,
random_state: int | None = None,
key_added: str | None = None,
tree: str | Sequence[str] | None = None,
Expand All @@ -697,7 +697,7 @@ def ancestral_linkage(

**Single-target mode** (``target=<category>``): computes the per-cell distance to
the nearest cell of the given category and stores it in
``tdata.obs['{target}_linkage']``.
``tdata.obs['{target}_linkage']`` (or ``tdata.obs['{key_added}_linkage']`` if ``key_added`` is given).

Parameters
----------
Expand All @@ -707,7 +707,8 @@ def ancestral_linkage(
Column in ``tdata.obs`` that defines cell categories.
target
If specified, compute the per-cell distance to the nearest cell of this
category and store the result in ``tdata.obs['{target}_linkage']``.
category and store the result in ``tdata.obs['{target}_linkage']``
(or ``tdata.obs['{key_added}_linkage']`` if ``key_added`` is given).
``aggregate`` is ignored in this mode.
If ``None`` (default), compute the full pairwise category × category matrix.
aggregate
Expand Down Expand Up @@ -739,7 +740,7 @@ def ancestral_linkage(
normalize
If ``True`` (default), subtract the permuted mean from the observed values: pairwise
linkage matrix becomes ``observed - permuted_mean``; single-target
``tdata.obs['{target}_linkage']`` becomes ``cell_score - category_permuted_mean``.
obs column becomes ``cell_score - category_permuted_mean``.
This works regardless of ``test``: a single permutation is run to estimate the
permuted mean when ``test=None`` (see ``n_permutations``).
min_size
Expand Down Expand Up @@ -788,10 +789,13 @@ def ancestral_linkage(
serialisation overhead. On other platforms this argument is ignored.
depth_key
Node attribute in ``tdata.obst[tree]`` that stores each node's depth.
If `None`, uses `tdata.uns['default_depth']` if present, otherwise 'depth'.
random_state
Random seed for reproducibility of permutation tests.
key_added
Base key for output storage. Defaults to ``groupby``.
Base key for output storage. Defaults to ``groupby`` for ``tdata.uns`` outputs. In single-target mode,
the per-cell result is stored in ``tdata.obs['{key_added}_linkage']`` if given, otherwise
``tdata.obs['{target}_linkage']``.
tree
The ``obst`` key or keys of the trees to use. If ``None``, all trees are used.
copy
Expand All @@ -803,7 +807,7 @@ def ancestral_linkage(

Sets the following fields:

* ``tdata.obs['{target}_linkage']`` : :class:`Series <pandas.Series>` (dtype ``float``) – single-target mode only.
* ``tdata.obs['{key_added}_linkage']`` or ``tdata.obs['{target}_linkage']`` : :class:`Series <pandas.Series>` (dtype ``float``) – single-target mode only.
Per-cell distance to the nearest cell of the target category. When
``normalize=True``, replaced by ``cell_score - category_permuted_mean``.
* ``tdata.uns['{key_added}_linkage']`` : :class:`DataFrame <pandas.DataFrame>` – pairwise mode only.
Expand Down Expand Up @@ -833,8 +837,11 @@ def ancestral_linkage(

>>> py.tl.ancestral_linkage(tdata, groupby="celltype", target="B", test="permutation")
"""
depth_key = get_depth_key(tdata, depth_key)
# ── setup ─────────────────────────────────────────────────────────────────
_set_random_state(random_state)
# Single-target results go in obs under the target name unless key_added is given
obs_key = f"{key_added or target}_linkage"
key_added = key_added or groupby
tree_keys = tree
_check_tree_overlap(tdata, tree_keys)
Expand Down Expand Up @@ -1029,9 +1036,9 @@ def _run_single_perm(single_tree, tree_lc, tree_sm, tree_cl, extra_row_fields=No
perm_val = cat_null_mean.get(cat, np.nan) if cat is not None else np.nan
merged_norm_map[leaf] = (score - perm_val) if not np.isnan(score) else np.nan

tdata.obs[f"{target}_linkage"] = tdata.obs.index.map(pd.Series(merged_score_map, dtype=float).to_dict())
tdata.obs[obs_key] = tdata.obs.index.map(pd.Series(merged_score_map, dtype=float).to_dict())
if normalize:
tdata.obs[f"{target}_linkage"] = tdata.obs.index.map(pd.Series(merged_norm_map, dtype=float).to_dict())
tdata.obs[obs_key] = tdata.obs.index.map(pd.Series(merged_norm_map, dtype=float).to_dict())
# Return per-category means from whichever map was written to obs.
linkage_map = merged_norm_map if normalize else merged_score_map
if test == "permutation":
Expand All @@ -1046,15 +1053,15 @@ def _run_single_perm(single_tree, tree_lc, tree_sm, tree_cl, extra_row_fields=No
cat: float(np.nanmean([linkage_map.get(l, np.nan) for l in cat_to_leaves[cat]]))
for cat in all_cats
},
name=f"{target}_linkage",
name=obs_key,
)
return result_series.to_frame()

else:
# Global (non-by_tree) path
all_scores = _compute_scores(tdata, trees, leaf_to_cat, [target], single_agg, metric, depth_key) # type: ignore
score_map = {leaf: scores.get(target, np.nan) for leaf, scores in all_scores.items()}
tdata.obs[f"{target}_linkage"] = tdata.obs.index.map(pd.Series(score_map, dtype=float).to_dict())
tdata.obs[obs_key] = tdata.obs.index.map(pd.Series(score_map, dtype=float).to_dict())

if run_perm:
rows, cat_null_mean = _run_single_perm(trees, leaf_to_cat, score_map, cat_to_leaves)
Expand All @@ -1067,7 +1074,7 @@ def _run_single_perm(single_tree, tree_lc, tree_sm, tree_cl, extra_row_fields=No
else np.nan
for leaf, score in score_map.items()
}
tdata.obs[f"{target}_linkage"] = tdata.obs.index.map(pd.Series(score_map, dtype=float).to_dict())
tdata.obs[obs_key] = tdata.obs.index.map(pd.Series(score_map, dtype=float).to_dict())
if test == "permutation":
test_df = pd.DataFrame(rows)
tdata.uns[f"{key_added}_test"] = test_df
Expand All @@ -1080,7 +1087,7 @@ def _run_single_perm(single_tree, tree_lc, tree_sm, tree_cl, extra_row_fields=No
cat: float(np.nanmean([score_map.get(l, np.nan) for l in cat_to_leaves[cat]]))
for cat in all_cats
},
name=f"{target}_linkage",
name=obs_key,
)
return result_series.to_frame()

Expand Down
Loading
Loading