{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# This notebook uses xgboost to tackle the Parkinson's Freezing of Gait (FOG) Prediction problem","metadata":{"papermill":{"duration":0.01998,"end_time":"2023-04-18T06:10:08.136875","exception":false,"start_time":"2023-04-18T06:10:08.116895","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Import Libraries","metadata":{"papermill":{"duration":0.019142,"end_time":"2023-04-18T06:10:08.173316","exception":false,"start_time":"2023-04-18T06:10:08.154174","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nfrom sklearn.model_selection import train_test_split\n!pip install xgboost\nimport xgboost as xgb\nfrom sklearn.metrics import accuracy_score","metadata":{"papermill":{"duration":1.387843,"end_time":"2023-04-18T06:10:09.578221","exception":false,"start_time":"2023-04-18T06:10:08.190378","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:25:46.914779Z","iopub.execute_input":"2023-04-21T01:25:46.915227Z","iopub.status.idle":"2023-04-21T01:26:18.641515Z","shell.execute_reply.started":"2023-04-21T01:25:46.915189Z","shell.execute_reply":"2023-04-21T01:26:18.640089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Take All the CSV Files in the Train tdcsfog Folder","metadata":{"papermill":{"duration":0.01873,"end_time":"2023-04-18T06:10:11.384414","exception":false,"start_time":"2023-04-18T06:10:11.365684","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Set the directory path to the folder containing the CSV files.\ntdcsfog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog'\n\n# Initialize an empty list to store the dataframes.\ntdcsfog_list = []\n\nfile_paths = [os.path.join(tdcsfog_path, file_name) for file_name in os.listdir(tdcsfog_path) if file_name.endswith('.csv')]\n\ntdcsfog_list = [pd.read_csv(file_path) for file_path in file_paths]\n\n# Concatenate the dataframes vertically using pd.concat().\ntdcsfog = pd.concat(tdcsfog_list, axis = 0)","metadata":{"papermill":{"duration":19.4046,"end_time":"2023-04-18T06:10:30.807859","exception":false,"start_time":"2023-04-18T06:10:11.403259","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:20:40.804677Z","iopub.execute_input":"2023-04-21T01:20:40.805628Z","iopub.status.idle":"2023-04-21T01:20:50.163483Z","shell.execute_reply.started":"2023-04-21T01:20:40.805559Z","shell.execute_reply":"2023-04-21T01:20:50.162403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Reference: [Reducing DataFrame memory size by ~65%](https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65)","metadata":{"papermill":{"duration":0.018462,"end_time":"2023-04-18T06:10:30.84543","exception":false,"start_time":"2023-04-18T06:10:30.826968","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def reduce_memory_usage(df):\n    start_mem = df.memory_usage().sum() / (1024 ** 2)\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n\n    for col in df.columns:\n        col_type = df[col].dtype\n\n        if col_type not in ('datetime64[ns]', 'category', 'object'):\n            c_min = df[col].min()\n            c_max = df[col].max()\n\n            if np.issubdtype(col_type, np.integer):\n                dtype = pd.Int64Dtype()\n                for t in (np.int8, np.int16, np.int32, np.int64):\n                    if c_min > np.iinfo(t).min and c_max < np.iinfo(t).max:\n                        dtype = t\n                        break\n                df[col] = df[col].astype(dtype)\n            elif np.issubdtype(col_type, np.floating):\n                dtype = pd.Float64Dtype()\n                for t in (np.float16, np.float32, np.float64):\n                    if c_min > np.finfo(t).min and c_max < np.finfo(t).max:\n                        dtype = t\n                        break\n                df[col] = df[col].astype(dtype)\n        elif col_type == 'object':\n            df[col] = df[col].astype('category')\n\n    mem_usg = df.memory_usage().sum() / (1024 ** 2)\n    print(\"Memory usage after optimization is: {:.2f} MB\".format(mem_usg))\n\n    return df","metadata":{"papermill":{"duration":0.037276,"end_time":"2023-04-18T06:10:30.901582","exception":false,"start_time":"2023-04-18T06:10:30.864306","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:20:50.165038Z","iopub.execute_input":"2023-04-21T01:20:50.166243Z","iopub.status.idle":"2023-04-21T01:20:50.178938Z","shell.execute_reply.started":"2023-04-21T01:20:50.166196Z","shell.execute_reply":"2023-04-21T01:20:50.177516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog = reduce_memory_usage(tdcsfog)","metadata":{"papermill":{"duration":0.563271,"end_time":"2023-04-18T06:10:31.483756","exception":false,"start_time":"2023-04-18T06:10:30.920485","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:20:50.182069Z","iopub.execute_input":"2023-04-21T01:20:50.182485Z","iopub.status.idle":"2023-04-21T01:20:50.667228Z","shell.execute_reply.started":"2023-04-21T01:20:50.182447Z","shell.execute_reply":"2023-04-21T01:20:50.665622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Take All the CSV Files in the Train defog Folder","metadata":{"papermill":{"duration":0.025823,"end_time":"2023-04-18T06:15:16.97234","exception":false,"start_time":"2023-04-18T06:15:16.946517","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Set the directory path to the folder containing the CSV files.\ndefog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog'\n\n# Initialize an empty list to store the dataframes.\ndefog_list = []\n\n# Loop through each file in the directory and read it into a dataframe.\nfor file_name in os.listdir(defog_path):\n    if file_name.endswith('.csv'):\n        file_path = os.path.join(defog_path, file_name)\n        file = pd.read_csv(file_path)\n        defog_list.append(file)\n\n# Concatenate the dataframes vertically using pd.concat().\ndefog = pd.concat(defog_list, axis = 0)","metadata":{"papermill":{"duration":26.653252,"end_time":"2023-04-18T06:15:43.651098","exception":false,"start_time":"2023-04-18T06:15:16.997846","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:20:50.668818Z","iopub.execute_input":"2023-04-21T01:20:50.669681Z","iopub.status.idle":"2023-04-21T01:21:03.004759Z","shell.execute_reply.started":"2023-04-21T01:20:50.66964Z","shell.execute_reply":"2023-04-21T01:21:03.003622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog = reduce_memory_usage(defog)\ndefog = defog[(defog['Task'] == 1) & (defog['Valid'] == 1)]\ndefog = defog.iloc[:, :7]\n\n# Concatenate the dataframes vertically using pd.concat().\nmerged = pd.concat([tdcsfog, defog], axis = 0)","metadata":{"papermill":{"duration":1.192651,"end_time":"2023-04-18T06:15:44.870176","exception":false,"start_time":"2023-04-18T06:15:43.677525","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:21:03.00661Z","iopub.execute_input":"2023-04-21T01:21:03.007103Z","iopub.status.idle":"2023-04-21T01:21:04.308795Z","shell.execute_reply.started":"2023-04-21T01:21:03.007051Z","shell.execute_reply":"2023-04-21T01:21:04.307159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Create Dataset\n\nsplit the data into features (i.e., \"AccV\", \"AccML\", and \"AccAP\") and classes (i.e., \"StartHesitation\", \"Turn\", and \"Walking\")","metadata":{"papermill":{"duration":0.026958,"end_time":"2023-04-18T06:15:47.608995","exception":false,"start_time":"2023-04-18T06:15:47.582037","status":"completed"},"tags":[]}},{"cell_type":"code","source":"X_merged = merged.iloc[:, 1:4]  # features\nX = tdcsfog.iloc[:, 1:4]\ny1 = merged['StartHesitation']\ny2 = merged['Turn']\ny3 = tdcsfog['Walking']","metadata":{"papermill":{"duration":0.241002,"end_time":"2023-04-18T06:15:47.876788","exception":false,"start_time":"2023-04-18T06:15:47.635786","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:21:04.310626Z","iopub.execute_input":"2023-04-21T01:21:04.311466Z","iopub.status.idle":"2023-04-21T01:21:04.473409Z","shell.execute_reply.started":"2023-04-21T01:21:04.311409Z","shell.execute_reply":"2023-04-21T01:21:04.472045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def balance_labels(X, y):\n    # Find the positions of y where it equals 0 and 1.\n    y_zeros = np.where(y == 0)[0]\n    y_ones = np.where(y == 1)[0]\n\n    # Choose the same number of samples with y == 1 as there are with y == 0.\n    num_ones = (y == 1).sum()\n    np.random.seed(42)\n    y_zeros_balanced = np.random.choice(y_zeros, size=num_ones, replace=False)\n\n    # Combine the positions of y == 0 and y == 1.\n    y_balanced_idxs = np.sort(np.concatenate([y_zeros_balanced, y_ones]))\n\n    # Use the balanced indices to get the corresponding rows of X and y.\n    X_balanced = X.iloc[y_balanced_idxs]\n    y_balanced = y.iloc[y_balanced_idxs]\n    \n    return X_balanced, y_balanced\n\n# Balance the labels for each dataset\nX1_balanced, y1_balanced = balance_labels(X_merged, y1)\nX2_balanced, y2_balanced = balance_labels(X_merged, y2)\nX3_balanced, y3_balanced = balance_labels(X, y3)\n\n# train test split\nX1_train, X1_test, y1_train, y1_test = train_test_split(X1_balanced, y1_balanced, test_size = 0.1, random_state = 42)\nX2_train, X2_test, y2_train, y2_test = train_test_split(X2_balanced, y2_balanced, test_size = 0.1, random_state = 42)\nX3_train, X3_test, y3_train, y3_test = train_test_split(X3_balanced, y3_balanced, test_size = 0.1, random_state = 42)","metadata":{"papermill":{"duration":1.17317,"end_time":"2023-04-18T06:15:49.921623","exception":false,"start_time":"2023-04-18T06:15:48.748453","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:21:04.474944Z","iopub.execute_input":"2023-04-21T01:21:04.476049Z","iopub.status.idle":"2023-04-21T01:21:07.171521Z","shell.execute_reply.started":"2023-04-21T01:21:04.476007Z","shell.execute_reply":"2023-04-21T01:21:07.17021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train Model","metadata":{"papermill":{"duration":0.026371,"end_time":"2023-04-18T06:15:51.459072","exception":false,"start_time":"2023-04-18T06:15:51.432701","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Define some good xgboost parameters\nxgb_params = {\n    'n_estimators': 100,\n    'max_depth': 5,\n    'learning_rate': 0.1,\n    'subsample': 0.8,\n    'colsample_bytree': 0.8,\n    'gamma': 0,\n    'reg_alpha': 0,\n    'reg_lambda': 1\n}\n\n# Create three XGBoost models with early stopping\nmodel1 = xgb.XGBClassifier(**xgb_params)\nmodel2 = xgb.XGBClassifier(**xgb_params)\nmodel3 = xgb.XGBClassifier(**xgb_params)\n\n# Define the early stopping criteria\nearly_stopping = xgb.callback.EarlyStopping(rounds=10)\n\n# Train the models on the training data with early stopping\nmodel1.fit(X1_train, y1_train, eval_set=[(X1_test, y1_test)], eval_metric='logloss', callbacks=[early_stopping])\nmodel2.fit(X2_train, y2_train, eval_set=[(X2_test, y2_test)], eval_metric='logloss', callbacks=[early_stopping])\nmodel3.fit(X3_train, y3_train, eval_set=[(X3_test, y3_test)], eval_metric='logloss', callbacks=[early_stopping])\n\n# Evaluate the models on the test data.\ny1_pred = model1.predict(X1_test)\nacc1 = accuracy_score(y1_test, y1_pred)\nprint('Accuracy for StartHesitation:', acc1)\n\ny2_pred = model2.predict(X2_test)\nacc2 = accuracy_score(y2_test, y2_pred)\nprint('Accuracy for Turn:', acc2)\n\ny3_pred = model3.predict(X3_test)\nacc3 = accuracy_score(y3_test, y3_pred)\nprint('Accuracy for Walking:', acc3)","metadata":{"execution":{"iopub.status.busy":"2023-04-21T01:26:18.644332Z","iopub.execute_input":"2023-04-21T01:26:18.644807Z","iopub.status.idle":"2023-04-21T01:26:49.365003Z","shell.execute_reply.started":"2023-04-21T01:26:18.644764Z","shell.execute_reply":"2023-04-21T01:26:49.363841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Eval","metadata":{"papermill":{"duration":0.030008,"end_time":"2023-04-18T06:16:04.00357","exception":false,"start_time":"2023-04-18T06:16:03.973562","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Set the directory path to the folder containing the CSV files.\ntdcsfog_test_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog'\n        \ntdcsfog_test_list = [\n    pd.read_csv(os.path.join(tdcsfog_test_path, file_name)).assign(Id=lambda df: file_name[:-4] + '_' + df['Time'].astype(str))\n    for file_name in os.listdir(tdcsfog_test_path)\n    if file_name.endswith('.csv')\n]        \n\n# Concatenate the dataframes vertically using pd.concat().\ntdcsfog_test = pd.concat(tdcsfog_test_list, axis = 0)","metadata":{"papermill":{"duration":0.077388,"end_time":"2023-04-18T06:16:04.111521","exception":false,"start_time":"2023-04-18T06:16:04.034133","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:29:36.805269Z","iopub.execute_input":"2023-04-21T01:29:36.805794Z","iopub.status.idle":"2023-04-21T01:29:36.830075Z","shell.execute_reply.started":"2023-04-21T01:29:36.805749Z","shell.execute_reply":"2023-04-21T01:29:36.828966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_test = reduce_memory_usage(tdcsfog_test)\n\n# Set the directory path to the folder containing the CSV files.\ndefog_test_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog'\n\ndefog_test_list = [\n    pd.read_csv(os.path.join(defog_test_path, file_name)).assign(Id=lambda df: file_name[:-4] + '_' + df['Time'].astype(str))\n    for file_name in os.listdir(defog_test_path)\n    if file_name.endswith('.csv')\n]        \n\n# Concatenate the dataframes vertically using pd.concat().\ndefog_test = pd.concat(defog_test_list, axis = 0)\n\ndefog_test = reduce_memory_usage(defog_test)\ntest = pd.concat([tdcsfog_test, defog_test], axis = 0).reset_index(drop = True)","metadata":{"papermill":{"duration":0.649463,"end_time":"2023-04-18T06:16:04.872528","exception":false,"start_time":"2023-04-18T06:16:04.223065","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:21:39.814422Z","iopub.status.idle":"2023-04-21T01:21:39.815351Z","shell.execute_reply.started":"2023-04-21T01:21:39.815018Z","shell.execute_reply":"2023-04-21T01:21:39.815055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{"papermill":{"duration":0.032587,"end_time":"2023-04-18T06:16:05.630669","exception":false,"start_time":"2023-04-18T06:16:05.598082","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Separate the dataset for the independent variables.\nX_test = test.iloc[:, 1:4]\n\n# Get the predictions for the three models on the test data.\nY1_pred = model1.predict(X_test)\nY2_pred = model2.predict(X_test)\nY3_pred = model3.predict(X_test)\n\ntest['StartHesitation'] = Y1_pred # target variable for StartHesitation\ntest['Turn'] = Y2_pred # target variable for Turn\ntest['Walking'] = Y3_pred # target variable for Walking","metadata":{"papermill":{"duration":0.142112,"end_time":"2023-04-18T06:16:05.804926","exception":false,"start_time":"2023-04-18T06:16:05.662814","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:21:39.816806Z","iopub.status.idle":"2023-04-21T01:21:39.8174Z","shell.execute_reply.started":"2023-04-21T01:21:39.817097Z","shell.execute_reply":"2023-04-21T01:21:39.817129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{"papermill":{"duration":0.031666,"end_time":"2023-04-18T06:16:05.91617","exception":false,"start_time":"2023-04-18T06:16:05.884504","status":"completed"},"tags":[]}},{"cell_type":"code","source":"submission = test.iloc[:, 4:]\nsubmission = submission.fillna(0.0)\nsubmission.to_csv(\"submission.csv\", index = False)","metadata":{"papermill":{"duration":0.4624,"end_time":"2023-04-18T06:16:06.519182","exception":false,"start_time":"2023-04-18T06:16:06.056782","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-04-21T01:21:39.818759Z","iopub.status.idle":"2023-04-21T01:21:39.819346Z","shell.execute_reply.started":"2023-04-21T01:21:39.819043Z","shell.execute_reply":"2023-04-21T01:21:39.819075Z"},"trusted":true},"execution_count":null,"outputs":[]}]}