Source code for uchrom.datasets

"""Datasets: every dataset U-Chrom knows, as a ``ChromData`` — or as its original files.

``load`` gives a ``ChromData``: a store of the public atlas (opened over HTTP, backed: only what a view or
an analysis reads is fetched), or a dataset built once from its original files by a registered loader and
cached in the data directory::

    import uchrom.datasets as ds
    cd = ds.load("takei2025_cerebellum")             # the atlas store, https://uchrom-atlas-r2.u-science.org/...
    cd = ds.load("kim2020_scihic")                   # 1,931 cells, cell types, per-cell contact maps (linked)

Original files (4DN, GEO, Zenodo, GitHub, UCSC) — what the readers of ``uchrom.io`` read, the inputs of
the benchmarks — are downloaded once into the data directory (``$UCHROM_DATA``, else ``~/.cache/uchrom``)
and checked::

    path = ds.fetch("takei")                          # .../4DNFIHF3JCBY.csv (Takei 2021 FOF-CT)
    ds.list_datasets()                                # what there is, sizes, descriptions

Command line: ``python -m uchrom.datasets list | atlas | load NAME | fetch NAME ... | path NAME``.
"""

from __future__ import annotations

from pathlib import Path
from typing import List, Optional

from ._paths import ENV, data_dir

__all__ = ["data_dir", "load", "fetch", "files", "path", "atlas", "list_datasets", "list_atlas", "ENV"]


def _entry(name: str) -> dict:
    from ._sources import DATASETS

    if name not in DATASETS:
        raise KeyError(f"unknown dataset {name!r}; known: {', '.join(sorted(DATASETS))}")
    return DATASETS[name]


def _files(entry: dict) -> list:
    items = entry.get("files") or [(entry["filename"], entry["url"])]
    return list(items() if callable(items) else items)


[docs] def files(name: str, *, root: Optional[Path] = None) -> List[Path]: """The local paths of a dataset's files (downloaded or not).""" root = Path(root) if root is not None else data_dir() entry = _entry(name) out = [root / rel for rel, _ in _files(entry)] out += [root / rel for rel, *_ in entry.get("extract", [])] out += [root / rel for rel, *_ in entry.get("extract_many", [])] return out
[docs] def path(name: str, *, root: Optional[Path] = None) -> Path: """Where a dataset is: its file if it has one, else the folder its files share.""" paths = files(name, root=root) if len(paths) == 1: return paths[0] import os return Path(os.path.commonpath([str(p) for p in paths]))
[docs] def fetch(name: str, *, root: Optional[Path] = None, extract: bool = True, verbose: bool = True) -> Path: """Download a dataset from its original source into the data directory (files already there are kept) and check its md5 sums; returns :func:`path`.""" from ._fetch import download, extract_many, extract_member, md5 root = Path(root) if root is not None else data_dir() entry = _entry(name) items = _files(entry) if "build" in entry: # built from its source (e.g. a .hic read over HTTP), not downloaded from . import _builders if not all((root / rel).exists() for rel, _ in items): func, kw = entry["build"] getattr(_builders, func)(out=root / items[0][0], **kw) for rel, want in entry.get("md5", {}).items(): if md5(root / rel) != want: raise IOError(f"md5 mismatch for {root / rel} (expected {want}); delete it and fetch again") return path(name, root=root) def one(item): rel, url = item dest = download(url, root / rel, entry.get("description", name), entry.get("size_mb", 0), verbose=verbose) want = entry.get("md5", {}).get(rel) if want and md5(dest) != want: raise IOError(f"md5 mismatch for {dest} (expected {want}); delete it and fetch again") if entry.get("parallel", 1) > 1: # polite: NCBI throttles many parallel requests from concurrent.futures import ThreadPoolExecutor def safe(item): try: one(item) except Exception as exc: return f"{item[0]}: {exc}" with ThreadPoolExecutor(entry["parallel"]) as pool: failed = [f for f in pool.map(safe, items) if f] if failed: raise IOError(f"{len(failed)} files of {name} failed (fetch again to retry): {failed[:3]}") else: for item in items: one(item) if extract: for rel, url, member in entry.get("extract", []): extract_member(url, member, root / rel, verbose=verbose) for rel, url, pattern in entry.get("extract_many", []): extract_many(url, pattern, root / rel, verbose=verbose) return path(name, root=root)
[docs] def list_datasets(): """The datasets :func:`fetch` knows: name, size (MB), whether :func:`load` gives it as a ``ChromData``, whether ``fetch --default`` (the tutorials' inputs) includes it, description.""" import pandas as pd from ._loaders import loaders from ._sources import DATASETS known = loaders() rows = [{"name": k, "size_mb": v.get("size_mb"), "load": k in known, "default": v.get("default", False), "description": v.get("description", "")} for k, v in DATASETS.items()] rows += [{"name": k, "size_mb": None, "load": True, "default": False, "description": v.description} for k, v in known.items() if k not in DATASETS] return pd.DataFrame(rows).set_index("name")
[docs] def load(name: str, *, root: Optional[Path] = None, backed: Optional[bool] = None, rebuild: bool = False, **read_kw): """A dataset as a ``ChromData``. - A dataset with a loader (``list_datasets()``, column ``load``): built once from its original files (fetched with :func:`fetch`) into ``<data directory>/<dataset>/<dataset>.chromdata.zarr`` with its linked files next to it, then read from there (in memory unless ``backed=True``). ``rebuild=True`` builds it again. - Otherwise a store of the atlas (``list_atlas()``): opened over HTTP, backed unless ``backed=False``. ``read_kw`` go to :meth:`chromdata.ChromData.read` (``columns=``, ``tracks=``, ...). """ from chromdata import ChromData from ._loaders import loaders known = loaders() if name in known: loader = known[name] root = Path(root) if root is not None else data_dir() out = root / loader.store if out.exists() and not rebuild: meta = ChromData.read(out, backed=True).uns.get("dataset") or {} rebuild = meta.get("loader_version") != loader.version if rebuild or not out.exists(): cd = loader.build(root, out) cd.uns["dataset"] = {"name": name, "loader_version": loader.version, "sources": list(loader.sources)} cd.write(out) # atomic: replaces an older build return ChromData.read(out, backed=bool(backed), **read_kw) try: return atlas(name, backed=True if backed is None else backed, **read_kw) except KeyError: raise KeyError(f"no dataset {name!r}: neither a loader ({', '.join(sorted(known))}) nor an atlas " f"store (list_atlas())") from None
[docs] def list_atlas(root: Optional[str] = None): """The datasets of the atlas (``chromdata.catalog.DEFAULT_ATLAS`` unless ``root``) as a table.""" import pandas as pd from chromdata import catalog cat = catalog.fetch_catalog(root or catalog.DEFAULT_ATLAS) cols = ("id", "title", "n_cells", "modalities", "size_mb", "url") return pd.DataFrame([{c: d.get(c) for c in cols} for d in cat["datasets"]]).set_index("id")
[docs] def atlas(name: str, *, root: Optional[str] = None, backed: bool = True, **read_kw): """Open a dataset of the atlas by its id (``list_atlas()``), backed over HTTP by default.""" from chromdata import ChromData, catalog root = root or catalog.DEFAULT_ATLAS cat = catalog.fetch_catalog(root) urls = {d["id"]: d["url"] for d in cat["datasets"]} if name not in urls: raise KeyError(f"{name!r} is not in the atlas at {root}; it has: {', '.join(sorted(urls))}") return ChromData.read(urls[name], backed=backed, **read_kw)