{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30761,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-13T14:51:23.564569Z","iopub.execute_input":"2024-10-13T14:51:23.565097Z","iopub.status.idle":"2024-10-13T14:51:23.571834Z","shell.execute_reply.started":"2024-10-13T14:51:23.565037Z","shell.execute_reply":"2024-10-13T14:51:23.570528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nimport datetime\nfrom tqdm import tqdm\n","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:23.574073Z","iopub.execute_input":"2024-10-13T14:51:23.574687Z","iopub.status.idle":"2024-10-13T14:51:23.590836Z","shell.execute_reply.started":"2024-10-13T14:51:23.574515Z","shell.execute_reply":"2024-10-13T14:51:23.589461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set pandas maximum columns to 1000\npd.set_option('display.max_columns', 1000)\npd.set_option('display.max_rows', 1000)","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:23.592675Z","iopub.execute_input":"2024-10-13T14:51:23.593188Z","iopub.status.idle":"2024-10-13T14:51:23.604342Z","shell.execute_reply.started":"2024-10-13T14:51:23.593141Z","shell.execute_reply":"2024-10-13T14:51:23.603023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"directory = '/kaggle/input'\nfiles = os.listdir(directory)\nprint(files)","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:23.605950Z","iopub.execute_input":"2024-10-13T14:51:23.606919Z","iopub.status.idle":"2024-10-13T14:51:23.618819Z","shell.execute_reply.started":"2024-10-13T14:51:23.606866Z","shell.execute_reply":"2024-10-13T14:51:23.616964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Read the train file and analyse the data**","metadata":{}},{"cell_type":"code","source":"train_file = os.path.join(directory,'child-mind-institute-problematic-internet-use','train.csv' )\ntrain_df = pd.read_csv(train_file)\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:23.622989Z","iopub.execute_input":"2024-10-13T14:51:23.623446Z","iopub.status.idle":"2024-10-13T14:51:23.772508Z","shell.execute_reply.started":"2024-10-13T14:51:23.623396Z","shell.execute_reply":"2024-10-13T14:51:23.771226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'train file has {train_df.shape[0]} rows and {train_df.shape[1]} columns')","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:23.774038Z","iopub.execute_input":"2024-10-13T14:51:23.774844Z","iopub.status.idle":"2024-10-13T14:51:23.780297Z","shell.execute_reply.started":"2024-10-13T14:51:23.774803Z","shell.execute_reply":"2024-10-13T14:51:23.779155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:23.781635Z","iopub.execute_input":"2024-10-13T14:51:23.781979Z","iopub.status.idle":"2024-10-13T14:51:23.814166Z","shell.execute_reply.started":"2024-10-13T14:51:23.781926Z","shell.execute_reply":"2024-10-13T14:51:23.812838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.describe()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:23.816312Z","iopub.execute_input":"2024-10-13T14:51:23.816738Z","iopub.status.idle":"2024-10-13T14:51:24.047524Z","shell.execute_reply.started":"2024-10-13T14:51:23.816696Z","shell.execute_reply":"2024-10-13T14:51:24.046010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot the distribution of the target column\nimport matplotlib.pyplot as plt\n\n# Plot the value counts as a bar chart\ntrain_df['sii'].value_counts().plot(kind='bar')\n\n# Optional: Set title and labels\nplt.title('Distribution of Target Column - sii')\nplt.xlabel('sii')\nplt.ylabel('Counts')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:24.049439Z","iopub.execute_input":"2024-10-13T14:51:24.049945Z","iopub.status.idle":"2024-10-13T14:51:24.333823Z","shell.execute_reply.started":"2024-10-13T14:51:24.049890Z","shell.execute_reply":"2024-10-13T14:51:24.332395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# analyse the nulls\n(train_df.isnull().sum() * 100) / (train_df.shape[0])","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:24.335389Z","iopub.execute_input":"2024-10-13T14:51:24.335766Z","iopub.status.idle":"2024-10-13T14:51:24.357695Z","shell.execute_reply.started":"2024-10-13T14:51:24.335727Z","shell.execute_reply":"2024-10-13T14:51:24.356573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# find all columns where missing values exceed 50%\nmissing_cols = []\ncolumns = train_df.columns\n\nfor col in columns:\n    null_count = (train_df[col].isnull().sum() * 100) / train_df.shape[0]\n    if null_count >= 50:\n        missing_cols.append(col)\nprint(f'columns with high null counts: {missing_cols} \\n')\nprint(f'There are {len(missing_cols)} with high null count' )","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:24.359396Z","iopub.execute_input":"2024-10-13T14:51:24.359783Z","iopub.status.idle":"2024-10-13T14:51:24.388124Z","shell.execute_reply.started":"2024-10-13T14:51:24.359743Z","shell.execute_reply":"2024-10-13T14:51:24.386428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop these columns\ntrain_df.drop(columns=missing_cols, axis=1, inplace=True)\nprint(f'new shape of data: {train_df.shape}')\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:24.389962Z","iopub.execute_input":"2024-10-13T14:51:24.390504Z","iopub.status.idle":"2024-10-13T14:51:24.478188Z","shell.execute_reply.started":"2024-10-13T14:51:24.390449Z","shell.execute_reply":"2024-10-13T14:51:24.476953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop rows where the 'target' column has NaN values\ntrain_df.dropna(subset=['sii'], inplace=True)\n\n# Print the new shape of the data\nprint(f'New shape of data: {train_df.shape}')","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:24.479915Z","iopub.execute_input":"2024-10-13T14:51:24.480442Z","iopub.status.idle":"2024-10-13T14:51:24.493667Z","shell.execute_reply.started":"2024-10-13T14:51:24.480387Z","shell.execute_reply":"2024-10-13T14:51:24.492406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# find all the numeric columns\nnumeric_cats = [\n    \"Basic_Demos-Sex\",\n    \"FGC-FGC_CU_Zone\",\n    \"FGC-FGC_GSND_Zone\",\n    \"FGC-FGC_GSD_Zone\",\n    \"FGC-FGC_PU_Zone\",\n    \"FGC-FGC_SRL_Zone\",\n    \"FGC-FGC_SRR_Zone\",\n    \"FGC-FGC_TL_Zone\",\n    \"BIA-BIA_Activity_Level_num\",\n    \"BIA-BIA_Frame_num\",\n    \"PCIAT-PCIAT_01\",\n    \"PCIAT-PCIAT_02\",\n    \"PCIAT-PCIAT_03\",\n    \"PCIAT-PCIAT_04\",\n    \"PCIAT-PCIAT_05\",\n    \"PCIAT-PCIAT_06\",\n    \"PCIAT-PCIAT_07\",\n    \"PCIAT-PCIAT_08\",\n    \"PCIAT-PCIAT_09\",\n    \"PCIAT-PCIAT_10\",\n    \"PCIAT-PCIAT_11\",\n    \"PCIAT-PCIAT_12\",\n    \"PCIAT-PCIAT_13\",\n    \"PCIAT-PCIAT_14\",\n    \"PCIAT-PCIAT_15\",\n    \"PCIAT-PCIAT_16\",\n    \"PCIAT-PCIAT_17\",\n    \"PCIAT-PCIAT_18\",\n    \"PCIAT-PCIAT_19\",\n    \"PCIAT-PCIAT_20\",\n    \"PreInt_EduHx-computerinternet_hoursday\"\n]\n\nnumeric_cats = [col for col in numeric_cats if col in train_df.columns]\n\n# convert the numerica categorical data to object\ntrain_df[numeric_cats] = train_df[numeric_cats].astype('object')\ncat_cols = [col for col in train_df.columns if train_df[col].dtype == 'object']\n\n# Print the list of categorical columns\nprint(f'There are {len(cat_cols)} categorical columns')\nprint(f'Categorical columns: \\n {cat_cols}\\n\\n')\n\ncont_cols = [col for col in train_df.columns if col not in cat_cols]\n\n# Print the list of continuous columns\nprint(f'There are {len(cont_cols)} continous columns')\nprint(f'Continuous columns: \\n {cont_cols} \\n\\n')","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:24.499809Z","iopub.execute_input":"2024-10-13T14:51:24.500589Z","iopub.status.idle":"2024-10-13T14:51:24.532669Z","shell.execute_reply.started":"2024-10-13T14:51:24.500543Z","shell.execute_reply":"2024-10-13T14:51:24.531200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# categorical data\ntrain_df[cat_cols].head(5)","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:24.534281Z","iopub.execute_input":"2024-10-13T14:51:24.534663Z","iopub.status.idle":"2024-10-13T14:51:24.575212Z","shell.execute_reply.started":"2024-10-13T14:51:24.534623Z","shell.execute_reply":"2024-10-13T14:51:24.573882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Continous columns\ntrain_df[cont_cols].head(5)","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:24.576851Z","iopub.execute_input":"2024-10-13T14:51:24.577304Z","iopub.status.idle":"2024-10-13T14:51:24.624477Z","shell.execute_reply.started":"2024-10-13T14:51:24.577253Z","shell.execute_reply":"2024-10-13T14:51:24.622188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Analyse the categorical data\ndef analyse_categorical_data(col, df):\n    if col not in ['id']:\n        print(f'Analysing column: {col} \\n')\n        # extract number of unique values\n        print(f'There are {df[col].nunique()} unique categories \\n')\n        print(f'unique values: \\n {df[col].unique()} \\n')","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:24.626365Z","iopub.execute_input":"2024-10-13T14:51:24.626881Z","iopub.status.idle":"2024-10-13T14:51:24.633934Z","shell.execute_reply.started":"2024-10-13T14:51:24.626823Z","shell.execute_reply":"2024-10-13T14:51:24.632622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in cat_cols:\n    analyse_categorical_data(col, train_df)","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:24.635896Z","iopub.execute_input":"2024-10-13T14:51:24.636477Z","iopub.status.idle":"2024-10-13T14:51:24.680269Z","shell.execute_reply.started":"2024-10-13T14:51:24.636417Z","shell.execute_reply":"2024-10-13T14:51:24.678966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# encode the categorical columns\ncat_cols.remove('id')\ntrain_df = pd.get_dummies(data=train_df, columns=cat_cols,dtype=float)\n# train_df = train_df.drop(columns=cat_cols)\n# train_df = pd.concat([train_df, new_cols], axis=1)\n\ntrain_df.head(5)\n\n# del new_cols\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:24.681624Z","iopub.execute_input":"2024-10-13T14:51:24.681979Z","iopub.status.idle":"2024-10-13T14:51:24.985916Z","shell.execute_reply.started":"2024-10-13T14:51:24.681942Z","shell.execute_reply":"2024-10-13T14:51:24.984666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fill blanks with mean values\ncols = [col for col in train_df.columns if col not in ['id', 'sii']]\nfor col in cols:\n    train_df[col] = train_df[col].fillna(train_df[col].mean())","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:24.987689Z","iopub.execute_input":"2024-10-13T14:51:24.988216Z","iopub.status.idle":"2024-10-13T14:51:25.066878Z","shell.execute_reply.started":"2024-10-13T14:51:24.988159Z","shell.execute_reply":"2024-10-13T14:51:25.065659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'shape of new data: {train_df.shape[0]} rows and {train_df.shape[1]} columns ')\ntrain_df.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:25.068545Z","iopub.execute_input":"2024-10-13T14:51:25.068947Z","iopub.status.idle":"2024-10-13T14:51:25.314474Z","shell.execute_reply.started":"2024-10-13T14:51:25.068908Z","shell.execute_reply":"2024-10-13T14:51:25.313178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# process the individial files\n# open sample file\nfile_path = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet/id=00115b9f/part-0.parquet'\nsample_df = pd.read_parquet(file_path)\nsample_df.sample(20)","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:25.316260Z","iopub.execute_input":"2024-10-13T14:51:25.316734Z","iopub.status.idle":"2024-10-13T14:51:25.360256Z","shell.execute_reply.started":"2024-10-13T14:51:25.316683Z","shell.execute_reply":"2024-10-13T14:51:25.358890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nimport datetime\n\n# Function to determine if the time is day or night\ndef day_or_night(t):\n    # Time boundaries for night (6 PM to 6 AM)\n    night_start = datetime.time(18, 0, 0)  # 6:00 PM\n    night_end = datetime.time(6, 0, 0)     # 6:00 AM\n    # Check if the time is in the night period\n    if t >= night_start or t < night_end:\n        return 'night'\n    else:\n        return 'day'\n\nsample_df['acceleration'] = sample_df.apply(lambda x: math.sqrt(x['X']**2 + x['Y']**2 + x['Z']**2), axis=1)\nsample_df['time'] = sample_df['time_of_day'].apply(lambda x: (datetime.datetime(1970, 1, 1) + datetime.timedelta(seconds=x/1_000_000_000)).time())\nsample_df['day_night'] = sample_df['time'].apply(lambda x: day_or_night(x) )\nsample_df.head(5)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:25.361818Z","iopub.execute_input":"2024-10-13T14:51:25.362277Z","iopub.status.idle":"2024-10-13T14:51:26.611059Z","shell.execute_reply.started":"2024-10-13T14:51:25.362234Z","shell.execute_reply":"2024-10-13T14:51:26.609840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_agg = sample_df.agg({\n    'acceleration': ['max', 'mean'],\n    'light': ['max', 'mean'],\n})\ndf_agg.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:26.612745Z","iopub.execute_input":"2024-10-13T14:51:26.613420Z","iopub.status.idle":"2024-10-13T14:51:26.634935Z","shell.execute_reply.started":"2024-10-13T14:51:26.613371Z","shell.execute_reply":"2024-10-13T14:51:26.633679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_night = sample_df.groupby('day_night').agg({\n    'acceleration': ['max', 'mean'],\n    'light': ['max', 'mean'],\n})\n\ndf_night","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:26.636898Z","iopub.execute_input":"2024-10-13T14:51:26.637959Z","iopub.status.idle":"2024-10-13T14:51:26.665377Z","shell.execute_reply.started":"2024-10-13T14:51:26.637886Z","shell.execute_reply":"2024-10-13T14:51:26.664187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_night.columns = ['_'.join(col) for col in df_night.columns]\ndf_night","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:26.667238Z","iopub.execute_input":"2024-10-13T14:51:26.667740Z","iopub.status.idle":"2024-10-13T14:51:26.683685Z","shell.execute_reply.started":"2024-10-13T14:51:26.667684Z","shell.execute_reply":"2024-10-13T14:51:26.682334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_night = df_night.reset_index()\ndf_night","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:26.685712Z","iopub.execute_input":"2024-10-13T14:51:26.686174Z","iopub.status.idle":"2024-10-13T14:51:26.702802Z","shell.execute_reply.started":"2024-10-13T14:51:26.686121Z","shell.execute_reply":"2024-10-13T14:51:26.701436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math\nimport datetime\n\n# Function to determine if the time is day or night\ndef day_or_night(t):\n    # Time boundaries for night (6 PM to 6 AM)\n    night_start = datetime.time(18, 0, 0)  # 6:00 PM\n    night_end = datetime.time(6, 0, 0)     # 6:00 AM\n    # Check if the time is in the night period\n    if t >= night_start or t < night_end:\n        return 'night'\n    else:\n        return 'day'\n    \ndef extract_data(pq_file):\n    id = pq_file.split('/')[-2]\n    df = pd.read_parquet(pq_file)\n    df['acceleration'] = df.apply(lambda x: math.sqrt(x['X']**2 + x['Y']**2 + x['Z']**2), axis=1)\n    df['time'] = df['time_of_day'].apply(lambda x: (datetime.datetime(1970, 1, 1) + datetime.timedelta(seconds=x/1_000_000_000)).time())\n    df['day_night'] = df['time'].apply(lambda x: day_or_night(x) )\n    \n    max_acceleration = df['acceleration'].sum()\n    mean_acceleration = df['acceleration'].mean()\n    mean_light = df['light'].mean()\n    max_light = df['light'].max()\n    \n    # Group by day_night and calculate aggregation\n    day_night_agg = df.groupby('day_night').agg(\n        {\n            'acceleration': ['mean', 'max'],\n            'light': ['mean', 'max'],\n        }\n    )\n\n    \n    # Extract day and night statistics\n    mean_day_acceleration = day_night_agg.loc['day', ('acceleration', 'mean')]\n    max_day_acceleration = day_night_agg.loc['day', ('acceleration', 'max')]\n    mean_day_light = day_night_agg.loc['day', ('light', 'mean')]\n    max_day_light = day_night_agg.loc['day', ('light', 'max')]\n\n    mean_night_acceleration = day_night_agg.loc['night', ('acceleration', 'mean')]\n    max_night_acceleration = day_night_agg.loc['night', ('acceleration', 'max')]\n    mean_night_light = day_night_agg.loc['night', ('light', 'mean')]\n    max_night_light = day_night_agg.loc['night', ('light', 'max')]\n    \n    # Return results in a list\n    return [\n        id, \n        max_acceleration, \n        mean_acceleration, \n        mean_light, \n        max_light, \n        max_day_acceleration, \n        mean_day_acceleration, \n        mean_day_light, \n        max_day_light, \n        max_night_acceleration, \n        mean_night_acceleration, \n        mean_night_light, \n        max_night_light\n    ]","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:51:26.704575Z","iopub.execute_input":"2024-10-13T14:51:26.705074Z","iopub.status.idle":"2024-10-13T14:51:26.724000Z","shell.execute_reply.started":"2024-10-13T14:51:26.705018Z","shell.execute_reply":"2024-10-13T14:51:26.722573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from concurrent.futures import ProcessPoolExecutor\n\n# Function to process a single file\ndef process_file(file_path):\n    try:\n        user_id = file_path.split(os.path.sep)[-2].split('=')[-1]  # Extract user_id\n        user_data = extract_data(file_path)  # Extract data from parquet file\n        return user_data\n    except Exception as e:\n        print(f\"Error processing {file_path}: {e}\")\n        return None  # Return None in case of error\n\n# Read all the parquet files\ndef read_activity_files(base_dir):\n    all_users = []\n    \n    # Gather all parquet files first to show total progress\n    parquet_files = []\n    for root, dirs, files in os.walk(base_dir):\n        for file in files:\n            if file.endswith(\".parquet\"):\n                parquet_files.append(os.path.join(root, file))  # Store full file paths\n\n    # Use tqdm to show progress while processing each file\n    with ProcessPoolExecutor() as executor:\n        # Process files in parallel and filter out None results\n        results = list(tqdm(executor.map(process_file, parquet_files), desc=\"Processing files\", unit=\"file\"))\n    \n    # Filter out None results\n    all_users = [result for result in results if result is not None]\n\n    # Create a DataFrame from the list of user data\n    df = pd.DataFrame(all_users, columns=[\n        'id', \n        'max_acceleration', \n        'mean_acceleration', \n        'mean_light', \n        'max_light', \n        'max_day_acceleration', \n        'mean_day_acceleration', \n        'mean_day_light', \n        'max_day_light', \n        'max_night_acceleration', \n        'mean_night_acceleration', \n        'mean_night_light', \n        'max_night_light'\n    ])\n    \n    return df\n\n","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:54:34.082917Z","iopub.execute_input":"2024-10-13T14:54:34.083494Z","iopub.status.idle":"2024-10-13T14:54:34.096658Z","shell.execute_reply.started":"2024-10-13T14:54:34.083446Z","shell.execute_reply":"2024-10-13T14:54:34.095401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_dir = '/kaggle/input/child-mind-institute-problematic-internet-use/series_train.parquet'\nall_users = read_activity_files(base_dir)\nall_users","metadata":{"execution":{"iopub.status.busy":"2024-10-13T14:54:34.099581Z","iopub.execute_input":"2024-10-13T14:54:34.100298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_users.to_csv('activity_data.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('activity_data.csv')\ndf.head(5)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\ndf['id'] =  df['id'].apply(lambda x: re.sub(r'id=', '', x) )\ndf.head(5)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('train_user_activity_data.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dir = 'kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet'\ntest_users = read_activity_files(test_dir)\ntest_users","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_users.to_csv('test_activity_data.csv', index=False)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('test_activity_data.csv')\ndf.head(5)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\ndf['id'] =  df['id'].apply(lambda x: re.sub(r'id=', '', x) )\ndf.to_csv('test_user_activity_data.csv', index=False)\ndf.head(5)","metadata":{},"execution_count":null,"outputs":[]}]}