diff --git a/pyActigraphy/io/bba/bba.py b/pyActigraphy/io/bba/bba.py index a6881ee..3717bae 100644 --- a/pyActigraphy/io/bba/bba.py +++ b/pyActigraphy/io/bba/bba.py @@ -1,5 +1,5 @@ -import importlib import json +import numpy as np import os import pandas as pd import re @@ -68,18 +68,6 @@ def __init__( use_metadata_json=True, metadata_fname=None ): - # The accelerometer package causes some dependency conflicts, - # so instead of always including it as a hard dependency, we - # try to import it as an optional one. Users with BBA files - # likely already have accelerometer installed, or should be - # able to sort out the dependency conflicts as needed. - try: - utils = importlib.import_module('accelerometer.utils') - except ModuleNotFoundError as e: - raise Exception( - 'Unable to load the required accelerometer.utils module.' - ' Install the package with "pip install accelerometer".' - ) from e # get absolute file path input_fname = os.path.abspath(input_fname) @@ -88,11 +76,12 @@ def __init__( data = pd.read_csv( input_fname, engine=engine, - index_col=['time'], - parse_dates=['time'], - date_parser=utils.date_parser + index_col=['time'] ) + # parse and set new index + data.set_index(self.__parse_acc_dates(data=data), inplace=True) + # read meta-data file (if found): if use_metadata_json: meta_data = self.__read_baa_metadata_json( @@ -150,8 +139,7 @@ def __init__( # Impute missing data (if required) if impute_missing: - s11n = importlib.import_module('accelerometer.summarization') - data = s11n.imputeMissing(data) + data = self.__imputeMissing(data) # LIGHT self.__white_light = self.__extract_baa_data( @@ -339,6 +327,100 @@ def __extract_baa_metadata(meta_data, field): 'Information ({}) not found in meta-data file.'.format(field) ) + @staticmethod + def __parse_acc_dates(data): + """ Date parser + + Parse datetime format used by the biobankaccelerometer output files: + `YYYY-MM-DD HH:mm:ss.SSS±HHmm [TimeZone]`. + + Parameters + ---------- + data : pd.DataFrame + Dataframe with the datatimes to transform as index. + + Returns + ------- + index: array of datetimes + Parsed and time-zone aware datetimes. + """ + + # Regex + dt_regex = re.compile(r'(?P
.+)\W\[(?P(?<=\[).+?(?=\]))\]') + + # Extract datetime and timezone + dts = data.index.str.extract(dt_regex, expand=True) + + # Extracted TZ + tz = dts['tz'].iloc[0] + + # Check if all extracted timezones are identical + if not all(x==tz for x in dts['tz']): + raise ValueError('Extracted timezones are not identical. Not supported.') + + return pd.to_datetime(dts['dt'],utc=True).dt.tz_convert(tz) + + @staticmethod + def __imputeMissing(data): + """ Impute missing/nonwear segments + + Copyright © 2025, University of Oxford + Adapted from [accelerometer 7.5.0](https://pypi.org/project/accelerometer/) (summarisation.py) + in order to remove the dependency of pyActigraphy on the accelerometer for the use of a single method. + This decision has been made in concertation with A. Doherty (email exchange: 17/12/2024). + Licence terms: https://github.com/OxWearables/biobankAccelerometerAnalysis/blob/95e1d790f745bbb5b44112166b522e58a47c485d/LICENSE.md + WARNING: the `accelerometer` package is distributed royalty-free for academic use only. + For commercial use, please contact the University of Oxford. + + + Impute non-wear data segments using the average of similar time-of-day values + with one minute granularity on different days of the measurement. This + imputation accounts for potential wear time diurnal bias where, for example, + if the device was systematically less worn during sleep in an individual, + the crude average vector magnitude during wear time would be a biased + overestimate of the true average. See + https://journals.plos.org/plosone/article?id=10.1371/journal.pone.0169649#sec013 + + Parameters + ---------- + data: pd.DataFrame + Pandas dataframe of epoch data + + Returns + ------- + data: pd.DataFrame + Updated DataFrame with nan values replaced with time-of-day imputation + """ + + def fillna(subframe): + # Transform will first pass the subframe column-by-column as a Series. + # After passing all columns, it will pass the entire subframe again as a DataFrame. + # Processing the entire subframe is optional (return value can be omitted). See 'Notes' in transform doc. + if isinstance(subframe, pd.Series): + x = subframe.to_numpy() + nan = np.isnan(x) + nanlen = len(x[nan]) + if 0 < nanlen < len(x): # check x contains a NaN and is not all NaN + x[nan] = np.nanmean(x) + return x # will be cast back to a Series automatically + else: + return subframe + + data = ( + data + # first attempt imputation using same day of week + .groupby([data.index.weekday, data.index.hour, data.index.minute]) + .transform(fillna) + # then try within weekday/weekend + .groupby([data.index.weekday >= 5, data.index.hour, data.index.minute]) + .transform(fillna) + # finally, use all other days + .groupby([data.index.hour, data.index.minute]) + .transform(fillna) + ) + + return data + def read_raw_bba( input_fname, diff --git a/setup.py b/setup.py index e4dd0d5..d9f7d6a 100644 --- a/setup.py +++ b/setup.py @@ -99,7 +99,7 @@ def find_version(*file_paths): # packages=['actimetry'], packages=find_packages(exclude=['docs', 'tests']), # Required - python_requires="<3.11", + python_requires="<=3.12", # This field lists other packages that your project depends on to run. # Any package you put here will be installed by pip when your project is # installed, so they must be valid existing projects. @@ -107,7 +107,7 @@ def find_version(*file_paths): # For an analysis of "install_requires" vs pip's requirements files see: # https://packaging.python.org/en/latest/requirements.html install_requires=[ - 'joblib', 'lmfit', 'pandas>=1.4.0', 'plotly', 'numba<=0.57.1', 'numpy', 'pyexcel', + 'joblib', 'lmfit', 'pandas>=1.4.0', 'plotly', 'numba', 'numpy', 'pyexcel', 'pyexcel-ods3', 'pyexcel-xlsx', 'scipy', 'spm1d', 'statsmodels>=0.10', 'stochastic>=0.6.0' ], # Optional