Source code for qblox_scheduler.data_handling

# SPDX-FileCopyrightText: © 2025 Qblox <https://qblox.com>
# SPDX-License-Identifier: BSD-3-Clause

"""Data handling utilities for Qblox Scheduler."""

import datetime
import sys
from collections import defaultdict
from copy import copy
from pathlib import Path
from typing import Any, ClassVar, Literal, Optional

import xarray as xr
from dateutil.parser import parse

from qblox_scheduler.data_dir import OutputDirectoryManager
from qblox_scheduler.quantify_utils import (
    TUID,
    save_json,
)
from qblox_scheduler.quantify_utils import (
    snapshot as create_snapshot,
)
from qblox_scheduler.quantify_utils import (
    write_dataset as qc_write_dataset,
)


[docs] class ExperimentDataContainer: """ Class which represents all data related to an experiment. This allows the user to run experiments and store data. The class serves as an initial interface and uses the directory paths set by OutputDirectoryManager. """ DATASET_NAME: ClassVar[str] = "dataset.hdf5" SNAPSHOT_FILENAME: ClassVar[str] = "snapshot.json" _TUID_LENGTH: ClassVar[int] = 26 # Length of "YYYYmmDD-HHMMSS-sss-******" def __init__(self, tuid: str, name: str) -> None: """ Creates an instance of the ExperimentDataContainer. Parameters ---------- tuid TUID to use name Name to append to the data directory path. """ self.tuid = tuid # Date folder works as a container of TUIDs date_folder = tuid.split("-", maxsplit=1)[0] self.day_folder = OutputDirectoryManager.get_datadir() / date_folder Path.mkdir(self.day_folder, exist_ok=True) # A TUID folder that contains data and potentially snapshot self.data_folder = ( (self.day_folder / f"{self.tuid}-{name}") if name else self.day_folder / f"{self.tuid}" ) Path.mkdir(self.data_folder, exist_ok=True) @property def experiment_name(self) -> str: """The name of the experiment.""" return self.tuid[self._TUID_LENGTH :]
[docs] @classmethod def load_dataset( cls, tuid: TUID, name: str = DATASET_NAME, ) -> xr.Dataset: """ Loads a dataset specified by a tuid. Parameters ---------- tuid A :class:`~qblox_scheduler.quantify_utils.TUID` string. It is also possible to specify only the first part of a tuid. name Name of the dataset. Returns ------- : The dataset. """ day_folder = OutputDirectoryManager.get_datadir() / Path(tuid.split("-")[0]) path = list(Path(day_folder).rglob(f"{tuid}*"))[0] / name return ExperimentDataContainer.load_dataset_from_path(path)
[docs] @classmethod def load_dataset_from_path(cls, path: Path | str) -> xr.Dataset: """ Loads a :class:`~xarray.Dataset` with a specific engine preference. Before returning the dataset :meth:`AdapterH5NetCDF.recover() <qblox_scheduler.helpers.dataset_adapters.AdapterH5NetCDF.recover>` is applied. This function tries to load the dataset until success with the following engine preference: - ``"h5netcdf"`` - ``"netcdf4"`` - No engine specified (:func:`~xarray.load_dataset` default) Parameters ---------- path Path to the dataset. Returns ------- : The loaded dataset. """ # pylint: disable=line-too-long exceptions = [] engines = ["h5netcdf", "netcdf4", None] for engine in engines: # there are three datasets that a user can load: # - "old" quantify datasets ( <2.0.0) # - "new" quantify datasets (>= 2.0.0) # - qblox-scheduler datasets try: dataset = xr.load_dataset(path, engine=engine) except Exception as exception: # noqa: BLE001, PERF203 exceptions.append(exception) else: return dataset # Do not let exceptions pass silently for exception, engine in zip(exceptions, engines[: engines.index(engine)]): # type: ignore # noqa: B020, B905 print( f"Failed loading dataset with '{engine}' engine. " f"Raised '{exception.__class__.__name__}':\n {exception}", ) # raise the last exception raise exception # type: ignore
[docs] def write_dataset(self, dataset: xr.Dataset) -> None: """ Writes the quantify dataset to the directory specified by `~.data_folder`. Parameters ---------- dataset The dataset to be written to the directory """ qc_write_dataset(self.data_folder / self.DATASET_NAME, dataset)
[docs] def save_snapshot( self, snapshot: Optional[dict[str, Any]] = None, compression: Literal["bz2", "gzip", "lzma"] | None = None, ) -> None: """ Writes the snapshot to disk as specified by `~.data_folder`. Parameters ---------- snapshot The snapshot to be written to the directory compression The compression type to use. Can be one of 'gzip', 'bz2', 'lzma'. Defaults to None, which means no compression. """ if snapshot is None: snapshot = create_snapshot() save_json( directory=self.data_folder, filename=self.SNAPSHOT_FILENAME, data=snapshot, compression=compression, )
[docs] @classmethod def get_latest_tuid(cls, contains: str = "") -> TUID: """ Returns the most recent tuid. .. tip:: This function is similar to :func:`~get_tuids_containing` but is preferred if one is only interested in the most recent :class:`~qblox_scheduler.quantify_utils.TUID` for performance reasons. Parameters ---------- contains An optional string contained in the experiment name. Returns ------- : The latest TUID. Raises ------ FileNotFoundError No data found. """ # `max_results=1, reverse=True` makes sure the tuid is found efficiently asap return ExperimentDataContainer.get_tuids_containing(contains, max_results=1, reverse=True)[ 0 ]
[docs] @classmethod # pylint: disable=too-many-locals def get_tuids_containing( cls, contains: str = "", t_start: datetime.datetime | str | None = None, t_stop: datetime.datetime | str | None = None, max_results: int = sys.maxsize, reverse: bool = False, ) -> list[TUID]: """ Returns a list of tuids containing a specific label. .. tip:: If one is only interested in the most recent :class:`~qblox_scheduler.quantify_utils.TUID`, :func:`~get_latest_tuid` is preferred for performance reasons. Parameters ---------- contains A string contained in the experiment name. t_start datetime to search from, inclusive. If a string is specified, it will be converted to a datetime object using :obj:`~dateutil.parser.parse`. If no value is specified, will use the year 1 as a reference t_start. t_stop datetime to search until, exclusive. If a string is specified, it will be converted to a datetime object using :obj:`~dateutil.parser.parse`. If no value is specified, will use the current time as a reference t_stop. max_results Maximum number of results to return. Defaults to unlimited. reverse If False, sorts tuids chronologically, if True sorts by most recent. Returns ------- list A list of :class:`~qblox_scheduler.quantify_utils.TUID`: objects. Raises ------ FileNotFoundError No data found. """ datadir = OutputDirectoryManager.get_datadir() if isinstance(t_start, str): t_start = parse(t_start) elif t_start is None: t_start = datetime.datetime(1, 1, 1) if isinstance(t_stop, str): t_stop = parse(t_stop) elif t_stop is None: t_stop = datetime.datetime.now() # date range filters, define here to make the next line more readable d_start = t_start.strftime("%Y%m%d") d_stop = t_stop.strftime("%Y%m%d") def lower_bound(dir_name: str) -> bool: return dir_name >= d_start if d_start else True def upper_bound(dir_name: str) -> bool: return dir_name <= d_stop if d_stop else True daydirs = list( filter( lambda x: ( x.name.isdigit() and len(x.name) == 8 and lower_bound(x.name) and upper_bound(x.name) ), datadir.iterdir(), ), ) daydirs.sort(reverse=reverse) if len(daydirs) == 0: err_msg = f"There are no valid day directories in the data folder '{datadir}'" if t_start or t_stop: err_msg += f", for the range {t_start or ''} to {t_stop or ''}" raise FileNotFoundError(err_msg) tuids = [] for daydir in daydirs: expdirs = list( filter( lambda x: ( len(x.name) > 25 and x.is_dir() and (contains in x.name) # label is part of exp_name and TUID.is_valid(x.name[:26]) # tuid is valid and (t_start <= TUID.datetime_seconds(x.name) < t_stop) ), Path.iterdir(datadir / daydir), ), ) expdirs.sort(reverse=reverse) for expname in expdirs: # Check for inconsistent folder structure for datasets portability if daydir != expname.name[:8]: raise FileNotFoundError( f"Experiment container '{expname}' is in wrong day directory '{daydir}'", ) tuids.append(TUID(expname.name[:26])) if len(tuids) == max_results: return tuids if len(tuids) == 0: raise FileNotFoundError(f"No experiment found containing '{contains}'") return tuids
[docs] @classmethod def locate_experiment_container(cls, tuid: str) -> Path: """Returns the experiment container for the given tuid.""" day_folder = Path(tuid.split("-", maxsplit=1)[0]) # Based on the tuid check if there is a respective folder(s) folder_list = list( Path(OutputDirectoryManager.get_datadir() / day_folder).rglob(f"{tuid}*") ) if len(folder_list) == 0: raise FileNotFoundError( f"Experiment container with given TUID {tuid}\ was not found" ) return folder_list[0]
def concat_acq_data(datasets: list[xr.Dataset]) -> xr.Dataset: """ Concatenates multiple acquisition datasets together to create one xarray Dataset. Note, it changes the acquisition index dimension values. Parameters ---------- datasets List of acquisition datasets. """ datasets = copy(datasets) # Current acquisition offset for each acquisition label's acq_index dimension, # which we adjust for each element input datasets. acq_index_offset = defaultdict(lambda: 0) for i in range(len(datasets)): # Note: in general dataset arrays can share the same acq index dimension. acq_index_dims = { datasets[i][k].attrs.get("acq_index_dim_name") for k in datasets[i].keys() } for acq_index_dim in acq_index_dims: offset = acq_index_offset[acq_index_dim] n = datasets[i].sizes[acq_index_dim] datasets[i] = datasets[i].assign_coords({acq_index_dim: range(offset, offset + n)}) acq_index_offset[acq_index_dim] += n concated_dataset = xr.Dataset() for dataset in datasets: concated_dataset = concated_dataset.merge(dataset, compat="no_conflicts", join="outer") return concated_dataset