"""
Utility functions for electrophysiological data processing and file management.
This module provides utility functions for managing output directories, handling
metadata, and working with Parquet files in electrophysiological data analysis
pipelines. It includes functions for directory structure creation, metadata
aggregation, and file attribute management.
The module includes:
- Output directory structure setup and management
- Metadata aggregation from multiple snippet directories
- Parquet file metadata management and updates
- File path handling and validation
Functions
---------
setup_output_directory
Set up hierarchical output directory structure for probe and snippet data
get_aggregated_snippets_df
Aggregate metadata from all snippets in a probe-level directory
add_metadata_to_parquet_files
Add metadata attributes to all Parquet files in a snippet directory
_update_parquet_metadata
Update metadata for a single Parquet file (internal helper function)
Examples
--------
>>> from ephysatlas.utils import setup_output_directory, add_metadata_to_parquet_files
>>> from pathlib import Path
>>>
>>> # Set up output directory structure
>>> params = {
... 'output_dir': '/data/output',
... 'pid': '76ed566f-59dd-47ff-8ba7-59b11d09b67c',
... 't_start': 300.0,
... 'duration': 5.0
... }
>>> probe_dir, snippet_dir = setup_output_directory(params)
>>>
>>> # Add metadata to Parquet files
>>> add_metadata_to_parquet_files(
... base_level_dir='/data/output',
... snippet_level_dir='probe_76ed566f-59dd-47ff-8ba7-59b11d09b67c_000300.0_05.0',
... pid='76ed566f-59dd-47ff-8ba7-59b11d09b67c',
... t_start=300.0,
... duration=5.0
... )
Notes
-----
This module handles the creation of hierarchical directory structures for
organizing electrophysiological data by probe ID and time snippets. It supports
both .parquet and .pqt file extensions and provides robust metadata management
capabilities for tracking data provenance and processing parameters.
See Also
--------
pandas.DataFrame.attrs : DataFrame attributes for metadata storage
pathlib.Path : Path manipulation and directory operations
"""
# TODO - Remove the requirement for .pqt and .pqt separately. Or handle it by defining a repo wide VARIABLE.
import hashlib
from pathlib import Path
from typing import Any, Dict
import pandas as pd
import logging
# Set up logger
logger = logging.getLogger(__name__)
[docs]
def setup_output_directory(params: Dict[str, Any]) -> tuple[Path, Path]:
"""Set up the output directory structure for probe and snippet data.
This function creates a hierarchical directory structure for organizing
electrophysiological data by probe ID and time snippets. It supports both
probe ID-based and file-based directory naming.
Args:
params (Dict[str, Any]): Dictionary containing configuration parameters.
Must include:
- output_dir (str, optional): Base output directory path
- pid (str, optional): Probe ID for directory naming
- filename (str, optional): AP file path for hash-based naming
- t_start (float): Start time for snippet naming
- duration (float): Duration for snippet naming
Returns:
tuple[Path, Path]: A tuple containing:
- probe_level_dir (Path): Path to the probe-level directory
- snippet_level_dir (Path): Path to the snippet-level directory
If output_dir is None, returns (None, None).
Raises:
ValueError: If neither pid nor filename is provided.
Note:
The function creates a hierarchical structure:
Probe level: Uses pid or hash of filename
Snippet level: Uses probe info, t_start, and duration
Example output structure::
|-- 76ed566f-59dd-47ff-8ba7-59b11d09b67c
| |-- probe_76ed566f-59dd-47ff-8ba7-59b11d09b67c_000300.0_05.0
| `-- probe_76ed566f-59dd-47ff-8ba7-59b11d09b67c_003000.0_05.0
`-- af0a0534-9cdc-4a29-93c0-1342891d74ec
|-- probe_af0a0534-9cdc-4a29-93c0-1342891d74ec_000300.0_05.0
`-- probe_af0a0534-9cdc-4a29-93c0-1342891d74ec_003000.0_05.0
"""
if params.get("output_dir") is None:
return None, None
# Create base output directory if specified, otherwise use current directory
base_dir = Path(params.get("output_dir"))
base_dir.mkdir(parents=True, exist_ok=True)
# Create probe level subdirectory (pid or hash of filename)
if params.get("pid") is not None:
probe_level_dir = base_dir / params["pid"]
elif params["filename"] is not None:
# For file-based processing, create a hash of just the AP filename
filename = Path(params["filename"]).name
# Create a hash of the filename
filename_hash = hashlib.md5(filename.encode()).hexdigest()[:12]
probe_level_dir = base_dir / filename_hash
else:
raise ValueError("Either pid or filename must be provided")
probe_level_dir.mkdir(parents=True, exist_ok=True)
# Create Snippet level subdirectory
# Pad t_start and duration
t_start_padded = f"{params['t_start']:08.1f}" # 8 digits with 1 decimal place
duration_ap = params.get("duration_ap", 0)
duration_lf = params.get("duration_lf", 0)
duration_ap_padded = f"{duration_ap:04.1f}" # 4 digits with 1 decimal place
duration_lf_padded = f"{duration_lf:04.1f}" # 4 digits with 1 decimal place
# TODO - Handle the case where pid is not provided properly.
snippet_level_dir = (
probe_level_dir
/ f"probe_{params.get('pid', 'unknown_pid')}_{t_start_padded}_{duration_ap_padded}_{duration_lf_padded}"
)
snippet_level_dir.mkdir(parents=True, exist_ok=True)
return probe_level_dir, snippet_level_dir
[docs]
def get_aggregated_snippets_df(probe_level_dir: Path) -> pd.DataFrame:
"""Get a dataframe of metadata info for all snippets in the probe level directory.
This function scans a probe-level directory and aggregates metadata from all
snippet subdirectories. It reads metadata from Parquet files (.parquet and .pqt)
and combines them into a single DataFrame for analysis.
Args:
probe_level_dir (Path): Path to the probe-level directory containing
snippet subdirectories.
Returns:
pd.DataFrame: DataFrame containing metadata from all snippets, with each
row representing one snippet and columns representing different metadata
attributes.
Note:
The function looks for both .parquet and .pqt file extensions in each
snippet directory. It extracts metadata from the DataFrame's attrs
dictionary and combines them across all snippets.
"""
data = []
def get_metadata_for_snippet(snippet_dir: Path) -> Dict[str, Any]:
"""Get the metadata for a snippet from a parquet file.
This helper function reads metadata from all Parquet files in a snippet
directory and combines them into a single dictionary.
Args:
snippet_dir (Path): Path to the snippet directory containing
Parquet files.
Returns:
Dict[str, Any]: Dictionary containing combined metadata from all
Parquet files in the snippet directory.
Note:
The function looks for both .parquet and .pqt file extensions and
extracts metadata from the DataFrame's attrs dictionary for each file.
"""
parquet_files = list(snippet_dir.glob("*.parquet")) + list(
snippet_dir.glob("*.pqt")
)
result = {}
for file_path in parquet_files:
df = pd.read_parquet(file_path)
result.update(df.attrs)
return result
for subdir in probe_level_dir.iterdir():
if subdir.is_dir():
snip_metadata = get_metadata_for_snippet(subdir)
if snip_metadata:
data.append(snip_metadata)
else:
logger.warning(f"No metadata found for {subdir}")
df = pd.DataFrame(data)
return df