{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":81933,"databundleVersionId":9643020,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom lightgbm import LGBMRegressor, LGBMClassifier\nimport fnmatch\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\ndf = pd.read_csv('/kaggle/input/child-mind-institute-problematic-internet-use/train.csv')\n# pick up rows which have sii and ignore the rest\ndf = df[df['sii'].notna()]\nid_list = df['id'].tolist()\nmissing_id_list = []\nfull_id_list = []\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n#file_path = \"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet/\"\n#count = len(fnmatch.filter(os.listdir(file_path), '*.*'))\n#print('File Count:', count)\n#for i in id_list:\n#    if os.path.exists(f\"{file_path}/id={id}/part-0.parquet\"):\n#        full_id_list.append(i)\n#    else:\n#        missing_id_list.append(i)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n#print(f\"Full Id List {len(id_list)}\")\n#print(f\"Missing Paraquet for  Number of Ids {len(missing_id_list)}\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-10-04T16:46:50.755154Z","iopub.execute_input":"2024-10-04T16:46:50.755648Z","iopub.status.idle":"2024-10-04T16:46:51.854741Z","shell.execute_reply.started":"2024-10-04T16:46:50.755601Z","shell.execute_reply":"2024-10-04T16:46:51.85252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Check how many rows have missing values\n* List out all rows which have missing values \n* Look at the types of cols which have missing values\n* Numeric values can be imputed with interpolation and categoric values need to be derived","metadata":{}},{"cell_type":"code","source":"df.shape[0], df.shape[1]","metadata":{"execution":{"iopub.status.busy":"2024-10-04T15:39:12.345231Z","iopub.execute_input":"2024-10-04T15:39:12.345627Z","iopub.status.idle":"2024-10-04T15:39:12.353693Z","shell.execute_reply.started":"2024-10-04T15:39:12.345585Z","shell.execute_reply":"2024-10-04T15:39:12.352398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_with_nans = [col for col in df if df[col].isnull().sum() > 0]\ndf[cols_with_nans].dtypes","metadata":{"execution":{"iopub.status.busy":"2024-10-04T15:39:12.355227Z","iopub.execute_input":"2024-10-04T15:39:12.355589Z","iopub.status.idle":"2024-10-04T15:39:12.395037Z","shell.execute_reply.started":"2024-10-04T15:39:12.355549Z","shell.execute_reply":"2024-10-04T15:39:12.393683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* So, out of 82 columns, 78 have atleast 1 value missing\n* Check how many mising values per each data field","metadata":{}},{"cell_type":"code","source":"df[cols_with_nans].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2024-10-04T15:39:12.398391Z","iopub.execute_input":"2024-10-04T15:39:12.399155Z","iopub.status.idle":"2024-10-04T15:39:12.415654Z","shell.execute_reply.started":"2024-10-04T15:39:12.399079Z","shell.execute_reply":"2024-10-04T15:39:12.41426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* If the column has more than 50% missing data, we can drop it since any way of imputing it will not be accurate and might cause issues in the model training. \n* Instead of imputing, we can think of manually labelling these values using human judges or LLM prompts.","metadata":{}},{"cell_type":"code","source":"full_data = df.copy()\nfor col in num_cols:\n    if(df[col].isnull().sum() > int(len(df)/2)):\n        df.drop(col, axis=1, inplace=True)\n        print(\"Dropping column %s \" %(col))\n        \n    ","metadata":{"execution":{"iopub.status.busy":"2024-10-04T15:39:12.417018Z","iopub.execute_input":"2024-10-04T15:39:12.417466Z","iopub.status.idle":"2024-10-04T15:39:12.437069Z","shell.execute_reply.started":"2024-10-04T15:39:12.417421Z","shell.execute_reply":"2024-10-04T15:39:12.435832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape[0],df.shape[1]","metadata":{"execution":{"iopub.status.busy":"2024-10-04T15:39:12.438688Z","iopub.execute_input":"2024-10-04T15:39:12.439356Z","iopub.status.idle":"2024-10-04T15:39:12.447218Z","shell.execute_reply.started":"2024-10-04T15:39:12.43931Z","shell.execute_reply":"2024-10-04T15:39:12.445998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols = [col for col in df[cols_with_nans].select_dtypes(exclude=['O'])]\ncat_cols = [col for col in df[cols_with_nans].select_dtypes(include=['O'])]\nnum_cols, cat_cols","metadata":{"execution":{"iopub.status.busy":"2024-10-04T15:39:12.449032Z","iopub.execute_input":"2024-10-04T15:39:12.449475Z","iopub.status.idle":"2024-10-04T15:39:12.471588Z","shell.execute_reply.started":"2024-10-04T15:39:12.449427Z","shell.execute_reply":"2024-10-04T15:39:12.470026Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* We now have 72 columns out of which 68 need data to be imputed.\n* Numeric data can be imp","metadata":{}},{"cell_type":"markdown","source":"* Now lets impute the missing values ","metadata":{}},{"cell_type":"code","source":"df[num_cols]\n\nfrom sklearn.impute import KNNImputer\nX = df[num_cols]\nknn_imputer = KNNImputer(n_neighbors=5)\ndf_imputed = pd.DataFrame(knn_imputer.fit_transform(X), columns=X.columns)\ndf = df.combine_first(df_imputed)","metadata":{"execution":{"iopub.status.busy":"2024-10-04T15:39:12.472847Z","iopub.execute_input":"2024-10-04T15:39:12.473261Z","iopub.status.idle":"2024-10-04T15:39:21.156938Z","shell.execute_reply.started":"2024-10-04T15:39:12.473218Z","shell.execute_reply":"2024-10-04T15:39:21.155558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_with_nans = [col for col in df if df[col].isnull().sum() > 0]\ndf[cols_with_nans]","metadata":{"execution":{"iopub.status.busy":"2024-10-04T15:39:21.15841Z","iopub.execute_input":"2024-10-04T15:39:21.15879Z","iopub.status.idle":"2024-10-04T15:39:21.201392Z","shell.execute_reply.started":"2024-10-04T15:39:21.158751Z","shell.execute_reply":"2024-10-04T15:39:21.200015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* Since the category data seems to be all season related data, we can skip them for now and work with the numeric data fields\n* Split the training data for a 80/20 train/test split","metadata":{}},{"cell_type":"code","source":"X_T,Y_T = df_imputed.to_numpy()[:, :-1],df_imputed.to_numpy()[:, -1]\nX_T, Y_T\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score\nX_train, X_test, Y_train, Y_test = train_test_split(X_T, Y_T, test_size=0.2, random_state=42)\nprint(X_train.shape)\nprint(Y_train.shape)\nprint(X_test.shape)\nprint(Y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2024-10-04T15:43:16.273832Z","iopub.execute_input":"2024-10-04T15:43:16.274285Z","iopub.status.idle":"2024-10-04T15:43:16.28547Z","shell.execute_reply.started":"2024-10-04T15:43:16.274242Z","shell.execute_reply":"2024-10-04T15:43:16.283961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Architecture:\nThe create_model function defines a sequential model with Conv1D layers for local pattern extraction, followed by LSTM layers for capturing long-term dependencies, and Dense layers for final classification.","metadata":{}},{"cell_type":"code","source":"\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Dropout\nfrom sklearn.preprocessing import StandardScaler\n\n# Standardize the data\nscaler = StandardScaler()\nX_train = scaler.fit_transform(X_train)\nX_test = scaler.transform(X_test)\n\n# Build the model\nscmodel = Sequential()\nscmodel.add(Dense(64, input_dim=X_train.shape[1], activation='relu'))\nscmodel.add(Dropout(0.5))\nscmodel.add(Dense(32, activation='relu'))\nscmodel.add(Dropout(0.5))\nscmodel.add(Dense(1, activation='sigmoid'))\n\n# Compile the model\nscmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])\n\n# Train the model\nhistory = scmodel.fit(X_train, Y_train, epochs=50, batch_size=32, validation_split=0.2)\n\n# Evaluate the model\nloss, accuracy = scmodel.evaluate(X_test, Y_test)\nprint(f\"Accuracy: {accuracy}\")","metadata":{"execution":{"iopub.status.busy":"2024-10-04T15:43:23.15843Z","iopub.execute_input":"2024-10-04T15:43:23.159748Z","iopub.status.idle":"2024-10-04T15:43:52.913021Z","shell.execute_reply.started":"2024-10-04T15:43:23.15968Z","shell.execute_reply":"2024-10-04T15:43:52.911705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Accuracy: 0.5656565427780151","metadata":{}},{"cell_type":"markdown","source":"* Now lets add the timeseries data to see if there will be any improvement to the model accuracy","metadata":{}},{"cell_type":"code","source":"file_path=\"/kaggle/input/child-mind-institute-problematic-internet-use/series_test.parquet\"\nids = df['id']\nfor i in ids:\n    \n    dfpq = pd.read_parquet(f\"{file_path}/id={i}/part-0.parquet\")\n    # Handle missing data in time series\n    dfpq['enmo'] = dfpq['enmo'].interpolate()\n    dfpq['anglez'] = dfpq['anglez'].interpolate()\n\n    # Create time-based features\n    dfpq['hour'] = pd.to_datetime(dfpq['time_of_day']).dt.hour\n    dfpq['is_weekend'] = dfpq['weekday'].isin([6, 7]).astype(int)\n\n    # Select relevant features\n    features = ['enmo', 'anglez', 'non-wear_flag', 'hour', 'is_weekend', 'relative_date_PCIAT']\n    X_actigraphy = dfpq[features]\n    X_actigraphy","metadata":{"execution":{"iopub.status.busy":"2024-10-04T17:05:26.924826Z","iopub.execute_input":"2024-10-04T17:05:26.925342Z","iopub.status.idle":"2024-10-04T17:05:27.064548Z","shell.execute_reply.started":"2024-10-04T17:05:26.925294Z","shell.execute_reply":"2024-10-04T17:05:27.062034Z"},"trusted":true},"execution_count":null,"outputs":[]}]}