{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport os\n\nfrom statsmodels.tsa.stattools import ccf\nfrom pandas.plotting import autocorrelation_plot\nfrom collections import defaultdict\nfrom sklearn.preprocessing import MinMaxScaler","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-01-19T15:28:33.911682Z","iopub.execute_input":"2025-01-19T15:28:33.912137Z","iopub.status.idle":"2025-01-19T15:28:35.503620Z","shell.execute_reply.started":"2025-01-19T15:28:33.912094Z","shell.execute_reply":"2025-01-19T15:28:35.502333Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"full_day_range = pd.timedelta_range(start='0 days', end='1 days', freq='15s')[:-1]\n\ndef clean_up_single_day(single_df):\n    single_df[\"time_of_day\"] = pd.to_timedelta(single_df[\"time_of_day\"], unit=\"ns\")\n    \n    # Reindex the DataFrame to include all time steps for the full day\n    single_df.set_index('time_of_day', inplace=True)\n    df_reindexed = single_df.reindex(full_day_range)\n    \n    # Reset index and rename the index column to 'time_of_day'\n    df_reindexed.index.name = 'time_of_day'\n    df_reindexed.reset_index(inplace=True)\n\n    df_reindexed['enmo_smooth'] = df_reindexed['enmo'].rolling(window=4, center=True, min_periods=1).mean()\n\n    # Discard intervals with a VERY low variance\n    df_reindexed['enmo_var'] = df_reindexed['enmo'].rolling(window=300, min_periods=1).var()\n    df_reindexed['enmo_smooth'] = df_reindexed['enmo_smooth'].where(df_reindexed['enmo_var'] >= 0.000001)\n\n    return df_reindexed\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T15:28:35.506571Z","iopub.execute_input":"2025-01-19T15:28:35.507230Z","iopub.status.idle":"2025-01-19T15:28:35.522860Z","shell.execute_reply.started":"2025-01-19T15:28:35.507177Z","shell.execute_reply":"2025-01-19T15:28:35.521457Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def is_day_usable(day_df):\n    day_df['is_nan'] = day_df['enmo_smooth'].isna()\n    \n    missing_count = day_df['is_nan'].sum()\n    valid_amount = 1 - missing_count / len(day_df)\n    return valid_amount >= 0.25","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T15:28:35.524310Z","iopub.execute_input":"2025-01-19T15:28:35.525134Z","iopub.status.idle":"2025-01-19T15:28:35.538763Z","shell.execute_reply.started":"2025-01-19T15:28:35.525084Z","shell.execute_reply":"2025-01-19T15:28:35.537319Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_day(day_df):\n    day_df['time_of_day_hours'] = day_df['time_of_day'].dt.total_seconds() / 3600  # Convert to hours\n\n    # Identify NaN ranges\n    day_df['is_nan'] = day_df['enmo_smooth'].isna()\n    nan_ranges = []\n    current_start = None\n    \n    for idx, is_nan in enumerate(day_df['is_nan']):\n        if is_nan and current_start is None:\n            current_start = day_df['time_of_day_hours'].iloc[idx]\n        elif not is_nan and current_start is not None:\n            nan_ranges.append((current_start, day_df['time_of_day_hours'].iloc[idx]))\n            current_start = None\n    \n    if current_start is not None:\n        nan_ranges.append((current_start, day_df['time_of_day_hours'].iloc[-1]))\n\n    \n    # Plot the data\n    plt.figure(figsize=(10, 6))\n    plt.plot(day_df[\"time_of_day_hours\"], day_df[\"enmo_smooth\"])\n    plt.xlim(0, 24)\n    plt.ylim(0, 1)\n    \n\n    # Highlight NaN ranges\n    for start, end in nan_ranges:\n        plt.axvspan(start, end, color='gray', alpha=0.3)\n    \n    # Customize the plot\n    plt.title(\"Time-of-Day Plot\", fontsize=16)\n    plt.xlabel(\"Time of Day\", fontsize=12)\n    plt.ylabel(\"ENMO\", fontsize=12)\n    plt.grid(True, linestyle='--', alpha=0.7)\n    plt.tight_layout()\n    \n    # Show the plot\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T15:28:35.540533Z","iopub.execute_input":"2025-01-19T15:28:35.541045Z","iopub.status.idle":"2025-01-19T15:28:35.555549Z","shell.execute_reply.started":"2025-01-19T15:28:35.540996Z","shell.execute_reply":"2025-01-19T15:28:35.554221Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"input_directory = \"/kaggle/input/child-mind-institute-problematic-internet-use\"\nparquet_directory = f\"{input_directory}/series_train.parquet\"\n\nall_files = list()\nfor parquet_name in os.listdir(parquet_directory):\n    person_id = parquet_name[3:]\n    parquet_file = f\"{parquet_directory}/{parquet_name}\"\n\n    all_files.append((person_id, parquet_file))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T15:28:35.558671Z","iopub.execute_input":"2025-01-19T15:28:35.559051Z","iopub.status.idle":"2025-01-19T15:28:35.622961Z","shell.execute_reply.started":"2025-01-19T15:28:35.559018Z","shell.execute_reply":"2025-01-19T15:28:35.621552Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef process_parquet_of(df):\n    # NORMALIZE ENMO TO 0..1\n    # scaler = MinMaxScaler()\n    # df[\"enmo\"] = scaler.fit_transform(df[[\"enmo\"]])\n    \n    df_filtered = df[[\"relative_date_PCIAT\", \"enmo\", \"light\", \"time_of_day\"]]\n    daily_df = dict(tuple(df_filtered.groupby(\"relative_date_PCIAT\")))\n    daily_df = {day: clean_up_single_day(df) for (day, df) in daily_df.items()}\n    \n    valid_daily_df = {day: df for (day, df) in daily_df.items() if is_day_usable(df)}\n\n    # SKIP this parquet file if data is insufficient\n    if len(valid_daily_df) == 0:\n        raise ValueError(\"Insufficient data\")\n    \n    # for day_df in list(valid_daily_df.values()):\n    #     plot_day(day_df)\n    \n    daily_activity = [day_df[['enmo_smooth', 'light']].describe().T for day_df in valid_daily_df.values()]\n\n    activity = pd.concat(daily_activity)\n    activity = activity.groupby(activity.index).agg({\n        'mean':'mean', \n        'std': 'std',\n        '25%': 'mean', \n        '50%': 'mean', \n        '75%': 'mean'\n    })\n    \n    return {\n        \"enmo_mean\": activity.at['enmo_smooth', 'mean'],\n        \"enmo_std\": activity.at['enmo_smooth', 'std'],\n        \"enmo_25%\": activity.at['enmo_smooth', '25%'],\n        \"enmo_50%\": activity.at['enmo_smooth', '50%'],\n        \"enmo_75%\": activity.at['enmo_smooth', '75%'],\n        \"light_mean\": activity.at['light', 'mean'],\n        \"light_std\": activity.at['light', 'std'],\n        \"light_25%\": activity.at['light', '25%'],\n        \"light_50%\": activity.at['light', '50%'],\n        \"light_75%\": activity.at['light', '75%'],\n    }\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T15:28:35.624405Z","iopub.execute_input":"2025-01-19T15:28:35.624804Z","iopub.status.idle":"2025-01-19T15:28:35.635251Z","shell.execute_reply.started":"2025-01-19T15:28:35.624769Z","shell.execute_reply":"2025-01-19T15:28:35.633865Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"output_rows = dict()\n\ni = 0\nfor (person_id, parquet_file) in all_files:\n    print(f\"{i + 1}/{len(all_files)}\")\n\n    try:\n        df = pd.read_parquet(parquet_file)\n        data = process_parquet_of(df)\n        \n        output_rows[person_id] = data\n    except Exception as e:\n        print(e)\n\n    i += 1\n\noutput_df = pd.DataFrame.from_dict(output_rows, orient='index')\noutput_df.index.name = 'id'\noutput_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T15:33:36.784241Z","iopub.execute_input":"2025-01-19T15:33:36.785376Z","iopub.status.idle":"2025-01-19T15:33:37.556359Z","shell.execute_reply.started":"2025-01-19T15:33:36.785331Z","shell.execute_reply":"2025-01-19T15:33:37.555158Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"output_df.to_csv('/kaggle/working/output.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-01-19T15:33:17.251081Z","iopub.execute_input":"2025-01-19T15:33:17.251468Z","iopub.status.idle":"2025-01-19T15:33:17.261828Z","shell.execute_reply.started":"2025-01-19T15:33:17.251433Z","shell.execute_reply":"2025-01-19T15:33:17.260661Z"}},"outputs":[],"execution_count":null}]}