{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Feature Engineer Activity Summary of Actigraphy Data\n\nIn this notebook, we process actigraphy data from train and test sets using a custom class, `ActivityProcessor`. The main steps are:\n\n- **Initialize Processor**:\n    - Set up paths for train and test data, output directory, and batch size.\n    - Create a cumulative DataFrame for storing results.\n\n\n- **Batch Processing**:\n    - **Load Data**: Each batch of parquet files is loaded and concatenated, adding a `student_id` and converting categorical columns as needed.\n    - **Validate Wear Flag**: Apply thresholds on ENMO and accelerometer data to confirm wear status, generating a `valid_wear_flag`.\n    - **Summarize Activity**: For each student, calculate wear duration, no-motion periods, and overall wear status.\n\n\n- **Save Results**:\n    - After processing, save separate summary CSVs for train and test data.","metadata":{}},{"cell_type":"code","source":"import cudf\nimport glob\nimport os\n\nclass ActivityProcessor:\n    \"\"\"\n    Class to process activity data from train and test parquet files, validate wear flags,\n    and calculate activity summaries in batches.\n    \"\"\"\n\n    def __init__(self, train_path, test_path, output_dir, batch_size=100):\n        \"\"\"\n        Initializes the ActivityProcessor with paths for train and test data, output directory,\n        and batch size.\n        \n        Parameters:\n            train_path (str): Path to the directory containing train parquet files.\n            test_path (str): Path to the directory containing test parquet files.\n            output_dir (str): Path to the output directory where results will be saved.\n            batch_size (int): Number of files to process per batch.\n        \"\"\"\n        self.train_path = train_path\n        self.test_path = test_path\n        self.output_dir = output_dir\n        self.batch_size = batch_size\n        self.activity_summary_df = cudf.DataFrame()\n        os.makedirs(output_dir, exist_ok=True)\n\n    def load_and_concatenate_batch(self, batch_files):\n        \"\"\"\n        Loads a batch of parquet files, adds a student_id column to each, and concatenates them.\n        \n        Parameters:\n            batch_files (list): List of parquet file paths to load and concatenate.\n        \n        Returns:\n            cudf.DataFrame: Concatenated DataFrame containing data from all files in the batch.\n        \"\"\"\n        batch_df = cudf.DataFrame()\n        for file in batch_files:\n            student_id = os.path.basename(os.path.dirname(file))\n            df = cudf.read_parquet(file)\n            df['student_id'] = student_id\n\n            # Convert categorical columns to strings\n            for column in df.columns:\n                if df[column].dtype.name == \"category\":\n                    df[column] = df[column].astype(\"str\")\n\n            batch_df = cudf.concat([batch_df, df], ignore_index=True)\n        \n        return batch_df\n\n    def validate_non_wear_flag(self, df, enmo_threshold=0.02, std_threshold=0.05, window_size=60):\n        \"\"\"\n        Validates wear and non-wear periods by applying thresholds to ENMO and accelerometer std values.\n        \n        Parameters:\n            df (cudf.DataFrame): DataFrame containing activity data.\n            enmo_threshold (float): Threshold for ENMO values indicating no motion.\n            std_threshold (float): Threshold for standard deviation indicating no motion.\n            window_size (int): Rolling window size for calculating standard deviation.\n        \n        Returns:\n            cudf.DataFrame: DataFrame with an additional 'valid_wear_flag' column indicating wear periods.\n        \"\"\"\n        df = df.copy()\n        df['std_X'] = df['X'].rolling(window=window_size).std()\n        df['std_Y'] = df['Y'].rolling(window=window_size).std()\n        df['std_Z'] = df['Z'].rolling(window=window_size).std()\n        df['std_acc'] = (df['std_X'] + df['std_Y'] + df['std_Z']) / 3\n        df['valid_wear_flag'] = 0\n        non_wear_mask = (df['enmo'] < enmo_threshold) & (df['std_acc'] < std_threshold)\n        df['valid_wear_flag'] = df['valid_wear_flag'].mask(non_wear_mask, 1)\n        df['flag_mismatch'] = df['valid_wear_flag'] != df['non-wear_flag']\n        \n        return df\n\n    def calculate_activity_summary(self, df):\n        \"\"\"\n        Calculates activity summary statistics for each student, including wear duration, no-motion periods,\n        and overall wear status.\n        \n        Parameters:\n            df (cudf.DataFrame): DataFrame containing validated wear data.\n        \n        Returns:\n            cudf.DataFrame: DataFrame containing summary statistics for each student.\n        \"\"\"\n        wear_df = df[df['valid_wear_flag'] == 0]\n        activity_summary = df.groupby('student_id').agg({\n            'valid_wear_flag': ['count', 'sum'],\n            'enmo': 'mean'\n        })\n        \n        activity_summary.columns = ['total_duration', 'non_wear_duration', 'average_activity_level']\n        activity_summary['wear_duration'] = activity_summary['total_duration'] - activity_summary['non_wear_duration']\n        no_motion_counts = df[df['enmo'] == 0].groupby('student_id').size().rename('no_motion_count')\n        activity_summary = activity_summary.merge(no_motion_counts, on='student_id', how='left').fillna(0)\n        \n        # Convert durations to seconds\n        activity_summary['total_duration_seconds'] = activity_summary['total_duration'] * 5\n        activity_summary['non_wear_duration_seconds'] = activity_summary['non_wear_duration'] * 5\n        activity_summary['wear_duration_seconds'] = activity_summary['wear_duration'] * 5\n        activity_summary['no_motion_duration_seconds'] = activity_summary['no_motion_count'] * 5\n        \n        # Determine if the device was worn for the majority of the sampling period\n        activity_summary['overall_wear_status'] = (activity_summary['wear_duration_seconds'] > \n                                                   (activity_summary['total_duration_seconds'] / 2)).astype(int)\n        \n        # Drop sample-based columns\n        activity_summary = activity_summary.drop(columns=['total_duration', 'non_wear_duration', \n                                                          'wear_duration', 'no_motion_count']).reset_index()\n        \n        return activity_summary\n\n    def save_to_csv(self, file_name):\n        \"\"\"\n        Saves the cumulative activity summary DataFrame to a CSV file.\n        \n        Parameters:\n            file_name (str): Name of the output CSV file.\n        \"\"\"\n        output_path = os.path.join(self.output_dir, file_name)\n        self.activity_summary_df.to_csv(output_path)\n        print(f\"Data saved to {output_path}\")\n\n    def process_files(self, mode='train'):\n        \"\"\"\n        Processes all parquet files in the specified mode (train or test) in batches. For each batch, loads, validates,\n        and summarizes the data, then appends to the cumulative activity summary DataFrame.\n        \n        Parameters:\n            mode (str): Indicates which data to process ('train' or 'test').\n        \"\"\"\n        data_path = self.train_path if mode == 'train' else self.test_path\n        parquet_files = glob.glob(os.path.join(data_path, '**/*.parquet'), recursive=True)\n        \n        for i in range(0, len(parquet_files), self.batch_size):\n            batch_files = parquet_files[i:i + self.batch_size]\n            batch_df = self.load_and_concatenate_batch(batch_files)\n            validated_df = self.validate_non_wear_flag(batch_df)\n            batch_summary = self.calculate_activity_summary(validated_df)\n            self.activity_summary_df = cudf.concat([self.activity_summary_df, batch_summary], ignore_index=True)\n            print(f\"Processed {mode} batch {i // self.batch_size + 1}/{len(parquet_files) // self.batch_size + 1}\")\n\n        # Save to CSV with appropriate naming based on mode\n        self.save_to_csv(f'activity_summary_{mode}.csv')\n","metadata":{"execution":{"iopub.status.busy":"2024-10-25T18:20:21.742644Z","iopub.execute_input":"2024-10-25T18:20:21.743092Z","iopub.status.idle":"2024-10-25T18:20:27.329137Z","shell.execute_reply.started":"2024-10-25T18:20:21.743046Z","shell.execute_reply":"2024-10-25T18:20:27.328160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# paths for train and test actigraphy data\ntrain_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\ntest_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet'\noutput_dir = '/kaggle/working/batch_output'\n\n# initialize with batch size of 100\nprocessor = ActivityProcessor(train_path=train_path, test_path=test_path, output_dir=output_dir, batch_size=100)\n\n# process train data files\nprocessor.process_files(mode='train')\n\n# clear activity_summary_df\nprocessor.activity_summary_df = cudf.DataFrame()\n\n# process test data files\nprocessor.process_files(mode='test')","metadata":{"execution":{"iopub.status.busy":"2024-10-25T18:20:27.331019Z","iopub.execute_input":"2024-10-25T18:20:27.331582Z","iopub.status.idle":"2024-10-25T18:23:05.552795Z","shell.execute_reply.started":"2024-10-25T18:20:27.331538Z","shell.execute_reply":"2024-10-25T18:23:05.551681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}