Source code for uchrom.io.upgrade

"""Convert ChromData files to the format 2.0 container (``.chromdata.zarr``).

``ChromData.read`` reads every older file (``.h5cd`` 1.x and 2.0) in
memory; this rewrites them once, so later reads are fast and can be
*backed* (``ChromData.read(path, backed=True)``)::

    python -m uchrom.io.upgrade old.h5cd new.chromdata.zarr
    python -m uchrom.io.upgrade data/*.h5cd --out-dir converted/        # <stem>.chromdata.zarr
    python -m uchrom.io.upgrade data/*.h5cd --out-dir converted/ --format cdz

What changes (see ``docs/source/guide/chromdata_2_0_design.md``):

* 1.x → 2.0 data model: ``bins`` = the unique spot loci; ``spots`` store
  ``bin_id`` instead of ``chrom/start/end`` (still available via
  ``cd.spots`` / ``cd.to_dataframe()``); the spot-aligned 1.x ``tracks``
  split into bin-level ``tracks`` (columns constant within every bin) and
  ``spot_tracks``;
* the container: Zarr v3 + Parquet, spots sorted by (cell, trace, bin)
  with an ``index/`` of row offsets (``index/source_row`` keeps the
  original order).

The target format follows the destination suffix (``.chromdata.zarr``,
``.cdz``, or — deprecated — ``.h5cd``).  The source is never modified.
"""

from __future__ import annotations

import argparse
import json
import sys
from pathlib import Path
from typing import Optional, Union

PathLike = Union[str, Path]

_SUFFIX = {"zarr": ".chromdata.zarr", "cdz": ".cdz", "h5cd": ".h5cd"}


[docs] def file_format_version(path: PathLike) -> str: """The ``uchrom_format_version`` of a ChromData file (``"1.0"`` for an unversioned ``.h5cd``).""" from uchrom.core.zarrcd import container_kind path = Path(path) kind = container_kind(path) if kind == "zarr": attrs = json.loads((path / "zarr.json").read_text()).get("attributes", {}) return str(attrs.get("uchrom", {}).get("format_version", attrs.get("uchrom_format_version"))) if kind == "cdz": import zipfile with zipfile.ZipFile(path) as zf: attrs = json.loads(zf.read("zarr.json")).get("attributes", {}) return str(attrs.get("uchrom", {}).get("format_version", attrs.get("uchrom_format_version"))) import h5py with h5py.File(path, "r") as f: v = f.attrs.get("uchrom_format_version", "1.0") return v.decode() if isinstance(v, bytes) else str(v)
[docs] def default_target(src: PathLike, fmt: str = "zarr", out_dir: Optional[PathLike] = None) -> Path: """``<dir>/<stem>.chromdata.zarr`` (or ``.cdz`` / ``.h5cd``) for ``src``.""" src = Path(src) name = src.name for suf in (".chromdata.zarr", ".zarr", ".cdz", ".h5cd"): if name.lower().endswith(suf): name = name[: -len(suf)] break else: name = src.stem return Path(out_dir or src.parent) / (name + _SUFFIX[fmt])
[docs] def upgrade_h5cd(src: PathLike, dst: Optional[PathLike] = None, *, overwrite: bool = False, coord_dtype: str = "float64", compression: Optional[str] = "gzip", compression_opts: Optional[int] = 4, compress_floats: bool = False) -> dict: """Read ``src`` (``.h5cd`` 1.x / 2.0, or any ChromData store) and write it to ``dst`` in the current format (``.chromdata.zarr`` 2.2, ``.h5cd`` 2.0); a zarr / cdz source is read in its original row order. ``dst`` defaults to ``<src stem>.chromdata.zarr`` next to ``src``; its suffix picks the container (``.chromdata.zarr``, ``.cdz``, or the deprecated ``.h5cd``). ``coord_dtype`` applies to the zarr container; ``compression`` / ``compression_opts`` / ``compress_floats`` to ``.h5cd`` targets (see :meth:`ChromData.write`). Returns a summary: source / target versions and containers, ``n_spots``, ``n_bins`` and which tracks are bin- vs spot-level. ``src`` is never modified; ``dst`` must differ from it. """ import warnings from uchrom.core.cdata import FORMAT_VERSION, ChromData from uchrom.core.zarrcd import container_kind src = Path(src) dst = Path(dst) if dst is not None else default_target(src) if src.resolve() == dst.resolve(): raise ValueError("upgrade_h5cd: dst must differ from src (the source is never modified)") if dst.exists() and not overwrite: raise FileExistsError(f"{dst} exists; pass overwrite=True") before = file_format_version(src) src_kind = container_kind(src) or "h5cd" # a zarr / cdz source: the rows of the object originally written (its # index/source_row), so the new store's original order is that one too cd = ChromData.read(src, original_order=src_kind in ("zarr", "cdz")) kind = container_kind(dst) or "h5cd" if kind == "h5cd": with warnings.catch_warnings(): warnings.simplefilter("ignore", DeprecationWarning) cd.write(dst, format="h5cd", compression=compression, compression_opts=compression_opts, compress_floats=compress_floats) else: cd.write(dst, coord_dtype=coord_dtype) names = cd.track_names() return { "src": str(src), "dst": str(dst), "from_version": before, "to_version": file_format_version(dst), "from_container": src_kind, "to_container": kind, "n_spots": int(cd.n_spots), "n_bins": int(cd.n_bins), "bin_tracks": names["bin"], "spot_tracks": names["spot"], }
#: the conversion is not specific to .h5cd sources convert = upgrade_h5cd
[docs] def main(argv: Optional[list] = None) -> int: ap = argparse.ArgumentParser( prog="python -m uchrom.io.upgrade", description="Rewrite ChromData files (.h5cd 1.x / 2.0) as format 2.0 " "(.chromdata.zarr by default).", ) ap.add_argument("paths", nargs="+", help="SRC DST, or one or more SRC with --out-dir") ap.add_argument("--out-dir", help="write each SRC to OUT_DIR/<stem>.chromdata.zarr (see --format)") ap.add_argument("--format", default="zarr", choices=["zarr", "cdz", "h5cd"], help="target container with --out-dir or a single SRC (default zarr)") ap.add_argument("--overwrite", action="store_true", help="replace existing outputs") ap.add_argument("--coord-dtype", default="float64", choices=["float64", "float32"], help="storage dtype of coords / layers in the zarr container") ap.add_argument("--compression", default="gzip", choices=["gzip", "lzf", "none"], help=".h5cd targets only: dataset filter (default gzip)") ap.add_argument("--level", type=int, default=4, help=".h5cd targets only: gzip level") ap.add_argument("--compress-floats", action="store_true", help=".h5cd targets only: also compress coords / float tracks") args = ap.parse_args(argv) if args.out_dir: out = Path(args.out_dir) out.mkdir(parents=True, exist_ok=True) pairs = [(Path(p), default_target(p, args.format, out)) for p in args.paths] elif len(args.paths) == 2: pairs = [(Path(args.paths[0]), Path(args.paths[1]))] elif len(args.paths) == 1: pairs = [(Path(args.paths[0]), default_target(args.paths[0], args.format))] else: ap.error("give SRC DST, one SRC, or several SRC with --out-dir") for src, dst in pairs: comp = None if args.compression == "none" else args.compression info = upgrade_h5cd(src, dst, overwrite=args.overwrite, coord_dtype=args.coord_dtype, compression=comp, compression_opts=args.level, compress_floats=args.compress_floats) print(f"{src} ({info['from_version']}, {info['from_container']}) -> {dst} " f"({info['to_version']}, {info['to_container']}): " f"{info['n_spots']} spots, {info['n_bins']} bins, " f"{len(info['bin_tracks'])} bin tracks, {len(info['spot_tracks'])} spot tracks") return 0
if __name__ == "__main__": sys.exit(main())