Source code for ibl_alignment_gui.utils.parse_yaml

from collections import defaultdict
from pathlib import Path
from typing import Literal

import yaml
from pydantic import BaseModel

# -------------------------------
# Pydantic Models
# -------------------------------


[docs] class DatasetPaths(BaseModel): """ Container for resolved dataset paths for a single probe. Attributes ---------- spike_sorting : Path | None Path to spike sorting output directory processed_ephys : Path | None Path to processed electrophysiology data directory raw_ephys : Path | None Path to raw electrophysiology recordings directory task : Path | None Path to task data directory raw_task : Path | None Path to raw task data directory picks : Path | None Path to probe trajectory pick files directory histology : Path | None Path to histology volume directory output : Path | None Path to alignment output directory features : Path | None Path to a per-channel ephys-features parquet file (used by the local channel-prediction plugin so the features travel with the session config). transforms : Path | None Path to a folder of registration transforms used to warp channel locations into the Allen CCF (anatomical workflow only). When set, channel locations are additionally saved in CCF coordinates. """ spike_sorting: Path | None = None processed_ephys: Path | None = None raw_ephys: Path | None = None task: Path | None = None raw_task: Path | None = None picks: Path | None = None histology: Path | None = None histology_space: str = 'ccf' output: Path | None = None features: Path | None = None transforms: Path | None = None
[docs] class Datasets(BaseModel): """ Dataset configuration. Attributes ---------- path : Path Relative or absolute path to the dataset directory space : {'ccf', 'anatomical'} | None Only meaningful for the ``histology`` entry of the top-level ``defaults`` section: the session-level coordinate space to use for histology slice loading. 'ccf' loads via NrrdSliceLoader using the Allen CCF atlas; 'anatomical' loads via AnatomicalSliceLoader using the original image space produced by the histology registration pipeline. """ path: Path | None = None space: Literal['ccf', 'anatomical'] | None = None
[docs] class Probe(BaseModel): """ Configuration for a single probe. Attributes ---------- datasets : dict[str, Datasets] | None Dictionary mapping dataset names to their configurations path : Path | None Probe-level base path for resolving relative dataset paths """ datasets: dict[str, Datasets] | None = None path: Path | None = None # Probe-level root
[docs] class Configuration(BaseModel): """ Configuration for a single experimental configuration. Attributes ---------- probes : dict[str, Probe] Dictionary mapping probe names to their configurations path : Path | None Configuration-level base path for resolving relative probe paths """ probes: dict[str, Probe] path: Path | None = None # Config-level root
[docs] class AlignmentYAML(BaseModel): """ Root-level YAML configuration structure. Attributes ---------- defaults : dict[str, Datasets] Default dataset configurations applied to all probes configurations : dict[str, Configuration] Dictionary mapping configuration names to their configurations path : Path | None Global root path for resolving all relative paths """ defaults: dict[str, Datasets] | None = None configurations: dict[str, Configuration] path: Path | None = None # Global root
# ------------------------------- # Path resolution logic # -------------------------------
[docs] def resolve_path( dataset_path: Path | None = None, probe_path: Path | None = None, config_path: Path | None = None, global_path: Path | None = None, default_path: Path | None = None, ) -> Path | None: """ Resolve dataset path using hierarchical path resolution. If path is absolute at any stage, it is returned immediately. Otherwise path is resolved progressively through the provided paths. Resolution order: dataset probe / dataset config / probe/ dataset global / config / probe / dataset If that path is still relative at the end, an error is raised. """ # Pick value or default path = dataset_path if dataset_path is not None else default_path if path is None: return None resolved_path = Path(path).expanduser() # Absolute value wins immediately if resolved_path.is_absolute(): return resolved_path.resolve() # Helper to prepend a root if present def prepend(root: Path | None, p: Path) -> Path: return root / p if root is not None else p # Progressive buildup resolved_path = prepend(probe_path, resolved_path) if resolved_path.is_absolute(): return resolved_path.resolve() resolved_path = prepend(config_path, resolved_path) if resolved_path.is_absolute(): return resolved_path.resolve() resolved_path = prepend(global_path, resolved_path) if resolved_path.is_absolute(): return resolved_path.resolve() # Still relative → cannot resolve fully raise ValueError('No absolute root provided to resolve relative path.')
# ------------------------------- # Loader # -------------------------------
[docs] def load_alignment_yaml( yaml_file: str, ) -> tuple[list[str], list[str], dict[str, dict[str, DatasetPaths]], str]: """ Load and parse alignment configuration YAML file. Resolves all dataset paths using hierarchical path resolution and applies defaults. Parameters ---------- yaml_file : str Path to the YAML configuration file Returns ------- configs : list of str List of configuration names probes : list of str List of unique probe names across all configurations data_paths : A dict of dicts of DatasetPaths Nested dictionary of resolved paths: data_paths[config_name][probe_name] -> DatasetPaths histology_space : str Session-level histology coordinate space, read from the ``space`` field of the ``defaults`` histology entry ('ccf' if unspecified). 'ccf' loads histology via the Allen CCF atlas; 'anatomical' loads the original image space from the registration pipeline. Notes ----- - If no 'configurations' section exists, creates a 'default' configuration - Falls back to the spike_sorting path if raw_ephys is not specified - Falls back to the raw_ephys path if processed_ephys is not specified - Falls back to the spike_sorting path, and then the picks path, if output is not specified Raises ------ FileNotFoundError If the yaml file does not exist. ValueError If the yaml file is empty or does not contain a mapping. """ yaml_file = Path(yaml_file) if not yaml_file.exists(): raise FileNotFoundError(f'YAML file {yaml_file} does not exist') with open(yaml_file) as f: data = yaml.safe_load(f) # An empty file loads as None, and a file holding a bare scalar or list loads as that value; # both would fail further down with an error that says nothing useful if not isinstance(data, dict): raise ValueError( f'YAML file {yaml_file} is empty or does not contain a mapping of configuration keys' ) # Support files without explicit 'configurations' section if 'configurations' not in data: probes = data.pop('probes', {}) data['configurations'] = {'default': {'probes': probes}} alignment = AlignmentYAML(**data) global_path = alignment.path # Histology space is a session-level setting read from the ``defaults`` histology entry and # shared by every probe/config ('ccf' when unspecified). default_histology = alignment.defaults.get('histology') if alignment.defaults else None histology_space = (default_histology.space if default_histology else None) or 'ccf' data_paths = defaultdict(dict) configs = [] probes = [] for cname, config in alignment.configurations.items(): config_path = config.path configs.append(cname) for pname, probe in config.probes.items(): probe_path = probe.path probes.append(pname) datasets = probe.datasets resolved_paths = DatasetPaths() def get_path(dname: str) -> str | None: """Get path for a specific dataset from probe configuration.""" # noqa is safe: this closure is only ever called within the same loop iteration # (see the dataset_name loop below), so the late binding B023 warns about cannot # be observed. dataset = datasets.get(dname) # noqa: B023 return dataset.path if dataset else None def get_default_path(dname: str) -> str | None: """Get default path for a specific dataset from defaults section.""" if alignment.defaults: default_dataset = alignment.defaults.get(dname) return default_dataset.path if default_dataset else None return None # Resolve all paths for dataset_name in [ 'spike_sorting', 'processed_ephys', 'raw_ephys', 'picks', 'histology', 'output', 'features', 'transforms', ]: path_value = get_path(dataset_name) default_value = get_default_path(dataset_name) resolved_path = resolve_path( path_value, probe_path, config_path, global_path, default_value ) setattr(resolved_paths, dataset_name, resolved_path) # The xyz picks are read from the picks path, falling back to the spike sorting # folder, and the results are written to the output, falling back to either of them. # With neither there is nowhere to read the trajectory from or write the results to. if resolved_paths.spike_sorting is None and resolved_paths.picks is None: raise ValueError( f'No spike_sorting or picks path given for probe {pname} in configuration ' f'{cname}; at least one of the two is needed' ) # The raw ephys is filled in first so that the processed ephys can fall back to # it, rather than the other way round which discarded an explicit processed_ephys if resolved_paths.raw_ephys is None: resolved_paths.raw_ephys = resolved_paths.spike_sorting if resolved_paths.processed_ephys is None: resolved_paths.processed_ephys = resolved_paths.raw_ephys # If no output is given the alignment results are written alongside the spike sorting, # falling back to the picks for sessions that have no spike sorting path if resolved_paths.output is None: resolved_paths.output = resolved_paths.spike_sorting or resolved_paths.picks data_paths[cname][pname] = resolved_paths assert len(configs) <= 2, ( 'More than two configurations found in YAML, alignment GUI supports up to two.' ) # The probes are returned as the union across configurations, and every configuration is # then indexed with that union when the shanks are built, so a probe that is missing from # one of them has to be caught here rather than failing with a KeyError later probes_per_config = {cname: set(data_paths.get(cname, {})) for cname in configs} if len({frozenset(names) for names in probes_per_config.values()}) > 1: detail = '; '.join( f'{cname}: {sorted(names)}' for cname, names in probes_per_config.items() ) raise ValueError( f'Every configuration must contain the same probes, found {detail}' ) # dict.fromkeys rather than a set, so that duplicates across configurations are dropped # while the order the probes are given in the yaml is kept. The order reaches the shank tabs # of the GUI, so it has to be the same on every run. return configs, list(dict.fromkeys(probes)), data_paths, histology_space