{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Parkinson's FOG - EDA and Model Submission\n\nThis notebook aims to achieve a reasonable result, with a glampse on the Data and quick modeling with cross validation.","metadata":{"_uuid":"5b65d14a-845e-4d70-91fa-f8bd3587b229","_cell_guid":"7de344f5-7402-4348-a853-11dacc758b62","trusted":true}},{"cell_type":"markdown","source":"# Importing necessary dependecies","metadata":{"_uuid":"07a24358-b23b-417e-a3e3-a494069f440a","_cell_guid":"4edf2e57-09c8-450b-9605-c1cc58169ac0","trusted":true}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nimport time\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport glob","metadata":{"_uuid":"9543ed61-fd77-47ca-aad6-a7259e31e914","_cell_guid":"53f498cc-600c-425c-abb5-d3227a5e9f45","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:40:37.079980Z","iopub.execute_input":"2023-06-07T23:40:37.080343Z","iopub.status.idle":"2023-06-07T23:40:37.826577Z","shell.execute_reply.started":"2023-06-07T23:40:37.080312Z","shell.execute_reply":"2023-06-07T23:40:37.825552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA\n## Reading the data","metadata":{"_uuid":"325a9ee8-aa0c-44d8-8c75-6e94699953b6","_cell_guid":"7258dc6d-55c5-4a2c-9f2a-91a5f9647fc1","trusted":true}},{"cell_type":"code","source":"import glob\ntrain_files = glob.glob(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/*.csv\")\ntrain_list = []\nfor f in tqdm(train_files):\n    train_batch = pd.read_csv(f)\n    train_list.append(train_batch)\ntrain = pd.concat(train_list)","metadata":{"_uuid":"c31d996f-1076-46bc-851a-32bee4a42319","_cell_guid":"98b08cb1-9848-4b7c-b67a-eb43ee7cc4ba","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:40:37.828409Z","iopub.execute_input":"2023-06-07T23:40:37.828787Z","iopub.status.idle":"2023-06-07T23:40:56.663962Z","shell.execute_reply.started":"2023-06-07T23:40:37.828738Z","shell.execute_reply":"2023-06-07T23:40:56.663008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualizing features","metadata":{"_uuid":"11a0e419-2d9a-4fdb-b9e1-135aac9d9b2d","_cell_guid":"deb7a7cf-d932-40ba-a54b-3dfcc7c8b5f5","trusted":true}},{"cell_type":"code","source":"fig, ax = plt.subplots(3,1, figsize=(20,8))\ntrain.AccV.plot(ax=ax[0])\ntrain.AccML.plot(ax=ax[1])\ntrain.AccAP.plot(ax=ax[2])\nplt.show()","metadata":{"_uuid":"9aeadfec-2036-4591-a1e8-55bae75dea15","_cell_guid":"fcd40271-865e-461d-bc6a-04ef501fcedb","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:40:56.665486Z","iopub.execute_input":"2023-06-07T23:40:56.666166Z","iopub.status.idle":"2023-06-07T23:41:05.862437Z","shell.execute_reply.started":"2023-06-07T23:40:56.666129Z","shell.execute_reply":"2023-06-07T23:41:05.861490Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Visualizing target vectors","metadata":{"_uuid":"cb16d9d6-96c6-4f17-b22a-b085a89a3a61","_cell_guid":"40ede67b-ba2b-411f-baf9-992247188579","trusted":true}},{"cell_type":"code","source":"fig, ax = plt.subplots(3,1, figsize=(20,8))\ntrain.StartHesitation.plot(ax=ax[0])\ntrain.Turn.plot(ax=ax[1])\ntrain.Walking.plot(ax=ax[2])\nplt.show()","metadata":{"_uuid":"7244cb34-899b-4044-852a-a1939af71241","_cell_guid":"11b4a756-63f7-43e7-83ee-5f0f673298c6","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:41:05.864862Z","iopub.execute_input":"2023-06-07T23:41:05.865989Z","iopub.status.idle":"2023-06-07T23:41:08.651808Z","shell.execute_reply.started":"2023-06-07T23:41:05.865950Z","shell.execute_reply":"2023-06-07T23:41:08.650838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.describe()","metadata":{"_uuid":"19709da5-45d9-4cfc-b991-9ea0c8346d22","_cell_guid":"dd0884d7-6c0b-4f2b-90cd-830c1942ea1b","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:41:08.653311Z","iopub.execute_input":"2023-06-07T23:41:08.653949Z","iopub.status.idle":"2023-06-07T23:41:10.276673Z","shell.execute_reply.started":"2023-06-07T23:41:08.653913Z","shell.execute_reply":"2023-06-07T23:41:10.275446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### From that we understand our mission is to classificate the time series signals between (StartHesitation, Turn and Walking)\n\n","metadata":{"_uuid":"41c5c8ee-c45b-4c11-85e9-322edf2c7ac6","_cell_guid":"7b204b05-ebe0-4e6b-b0e1-36e368420257","trusted":true}},{"cell_type":"markdown","source":"## How do our features affect and correlate to our target classess?","metadata":{"_uuid":"47152225-5c9b-4987-8209-9d71c9be78dc","_cell_guid":"8144765b-f929-4ee0-9256-f42c1ea38ade","trusted":true}},{"cell_type":"markdown","source":"### AccV relationship with StartHesitation, Turn and Walking","metadata":{"_uuid":"e957c559-e8bc-4624-b893-cd2db027e126","_cell_guid":"b54d1e9f-4109-43ea-9307-3e3b31b22d21","trusted":true}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize=(20,4));\ntrain.loc[train.StartHesitation == 1].AccV.plot(kind='hist', title = 'AccV with StartHesitation = 1', ax =ax[0], bins=100, xlim=(-16,-3));\ntrain.loc[train.StartHesitation == 0].AccV.plot(kind='hist', title = 'AccV with StartHesitation = 0', ax =ax[1], bins=100, xlim=(-16,-3), color='red');\nfig, ax = plt.subplots(1,2, figsize=(20,4));\ntrain.loc[train.Turn == 1].AccV.plot(kind='hist', title = 'AccV with Turn = 1', ax =ax[0], bins=100, xlim=(-16,-3));\ntrain.loc[train.Turn == 0].AccV.plot(kind='hist', title = 'AccV with Turn = 0', ax =ax[1], bins=100, xlim=(-16,-3), color='red');\nfig, ax = plt.subplots(1,2, figsize=(20,4));\ntrain.loc[train.Walking == 1].AccV.plot(kind='hist', title = 'AccV with Walking = 1', ax =ax[0], bins=100, xlim=(-16,-3));\ntrain.loc[train.Walking == 0].AccV.plot(kind='hist', title = 'AccV with Walking = 0', ax =ax[1], bins=100, xlim=(-16,-3), color='red');","metadata":{"_uuid":"8a93cc8c-5ae6-4ffc-9cfe-3a4138494b4b","_cell_guid":"d966e54c-5057-4ddd-aa08-55293503e720","collapsed":false,"_kg_hide-input":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:41:10.278438Z","iopub.execute_input":"2023-06-07T23:41:10.278906Z","iopub.status.idle":"2023-06-07T23:41:16.374832Z","shell.execute_reply.started":"2023-06-07T23:41:10.278865Z","shell.execute_reply":"2023-06-07T23:41:16.373553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### AccML relationship with StartHesitation, Turn and Walking","metadata":{"_uuid":"cbbace38-2a56-4ed2-bf3b-cd10b9b00565","_cell_guid":"f4211aa1-82f6-4235-a002-c9370c63694d","trusted":true}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize=(20,4));\ntrain.loc[train.StartHesitation == 1].AccML.plot(kind='hist', title = 'AccML with StartHesitation = 1', ax =ax[0],xlim=(-8,8), bins=100,);\ntrain.loc[train.StartHesitation == 0].AccML.plot(kind='hist', title = 'AccML with StartHesitation = 0', ax =ax[1],xlim=(-8,8), bins=100, color='red');\nfig, ax = plt.subplots(1,2, figsize=(20,4));\ntrain.loc[train.Turn == 1].AccML.plot(kind='hist', title = 'AccML with Turn = 1', ax =ax[0], bins=100, xlim=(-8,8));\ntrain.loc[train.Turn == 0].AccML.plot(kind='hist', title = 'AccML with Turn = 0', ax =ax[1], bins=100, xlim=(-8,8), color='red');\nfig, ax = plt.subplots(1,2, figsize=(20,4));\ntrain.loc[train.Walking == 1].AccML.plot(kind='hist', title = 'AccML with Walking = 1', ax =ax[0], bins=100, xlim=(-8,8));\ntrain.loc[train.Walking == 0].AccML.plot(kind='hist', title = 'AccML with Walking = 0', ax =ax[1], bins=100, xlim=(-8,8), color='red');","metadata":{"_uuid":"ae684121-d946-4a96-9b07-7c80260a8caf","_cell_guid":"208b9a09-846a-42da-b246-9bfeb99dc7f2","collapsed":false,"_kg_hide-input":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:41:16.376079Z","iopub.execute_input":"2023-06-07T23:41:16.376431Z","iopub.status.idle":"2023-06-07T23:41:22.278797Z","shell.execute_reply.started":"2023-06-07T23:41:16.376404Z","shell.execute_reply":"2023-06-07T23:41:22.277698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### AccAP relationship with StartHesitation,Turn and Walking","metadata":{"_uuid":"104ad714-b1fb-4e0f-9fcf-988fd87797d7","_cell_guid":"731f8be5-18b8-4a03-8c3a-d8e03216715e","trusted":true}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,2, figsize=(20,4));\ntrain.loc[train.StartHesitation == 1].AccAP.plot(kind='hist', title = 'AccAP with StartHesitation = 1', ax =ax[0],xlim=(-8,8), bins=100,);\ntrain.loc[train.StartHesitation == 0].AccAP.plot(kind='hist', title = 'AccAP with StartHesitation = 0', ax =ax[1],xlim=(-8,8), bins=100, color='red');\nfig, ax = plt.subplots(1,2, figsize=(20,4));\ntrain.loc[train.Turn == 1].AccAP.plot(kind='hist', title = 'AccAP with Turn = 1', ax =ax[0], bins=100, xlim=(-8,8));\ntrain.loc[train.Turn == 0].AccAP.plot(kind='hist', title = 'AccAP with Turn = 0', ax =ax[1], bins=100, xlim=(-8,8), color='red');\nfig, ax = plt.subplots(1,2, figsize=(20,4));\ntrain.loc[train.Walking == 1].AccAP.plot(kind='hist', title = 'AccAP with Walking = 1', ax =ax[0], bins=100, xlim=(-8,8));\ntrain.loc[train.Walking == 0].AccAP.plot(kind='hist', title = 'AccAP with Walking = 0', ax =ax[1], bins=100, xlim=(-8,8), color='red');","metadata":{"_uuid":"5b709da2-84b2-4909-b7a6-cbe077ed2617","_cell_guid":"83555b24-8889-456d-9519-a0c94646d2cb","collapsed":false,"_kg_hide-input":true,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:41:22.280601Z","iopub.execute_input":"2023-06-07T23:41:22.280998Z","iopub.status.idle":"2023-06-07T23:41:28.160614Z","shell.execute_reply.started":"2023-06-07T23:41:22.280964Z","shell.execute_reply":"2023-06-07T23:41:28.159702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature correlation with our Target Classes","metadata":{"_uuid":"350e1995-22c9-42eb-af21-cada040f6690","_cell_guid":"e8d814d4-f32d-45e7-bd42-b762ad41b614","trusted":true}},{"cell_type":"code","source":"corr_matrix = train.corr()\n\n# Plot correlation matrix as a heatmap\nplt.figure(figsize=(12,8));\nplt.title('Feature Correlation');\nmask = np.triu(np.ones_like(corr_matrix, dtype=bool))\nsns.heatmap(corr_matrix, annot=True, cmap='coolwarm', mask=mask.T);","metadata":{"_uuid":"662609b8-cb1e-4b00-89ea-caa3a26d6549","_cell_guid":"5eac1471-e1b1-468a-879d-6237d5f015d8","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:41:28.162214Z","iopub.execute_input":"2023-06-07T23:41:28.162585Z","iopub.status.idle":"2023-06-07T23:41:29.846013Z","shell.execute_reply.started":"2023-06-07T23:41:28.162551Z","shell.execute_reply":"2023-06-07T23:41:29.843194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### From the analysis of the relationship between our features and our target classes, we observe:\n*  AccML has an inverse relationship to our target classes\n*  AccAP has a positive relationship to our targets\n*  AccV presents low correlation to our classes\n* Time has a STRONG positive relationship with TURN and WALKING.","metadata":{"_uuid":"5a4b6fbc-fd3e-4bd1-b20e-8b86c1c067b4","_cell_guid":"c0cf40e2-4bdf-4a0a-b938-23d64d8d6c96","trusted":true}},{"cell_type":"markdown","source":"### Are the target vectors balanced","metadata":{"_uuid":"c245fa19-9c6f-4e94-adb1-4f7d0b2eedde","_cell_guid":"00c66426-4380-4915-a4e3-cae32b250d86","trusted":true}},{"cell_type":"code","source":"fig, ax = plt.subplots(1,3, figsize=(20,8))\ntrain[['StartHesitation', 'Turn', 'Walking']].StartHesitation.plot(kind='hist', ax=ax[0], title = 'StartHesitation Countplot')\ntrain[['StartHesitation', 'Turn', 'Walking']].Turn.plot(kind='hist', ax = ax[1], title = 'Turn Countplot')\ntrain[['StartHesitation', 'Turn', 'Walking']].Walking.plot(kind='hist', ax = ax[2], title = 'Walking Countplot')\nplt.show()","metadata":{"_uuid":"223f64e1-2c00-4007-aa34-388f78c26ced","_cell_guid":"ed85ecae-10de-40ac-8468-26b5484ccc0b","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:41:29.851414Z","iopub.execute_input":"2023-06-07T23:41:29.852251Z","iopub.status.idle":"2023-06-07T23:41:32.357067Z","shell.execute_reply.started":"2023-06-07T23:41:29.852208Z","shell.execute_reply":"2023-06-07T23:41:32.355912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### It seems they are not balanced, we should deal with this problem using undersampling or class weights.","metadata":{"_uuid":"eaf6535d-5fd3-498d-9ddc-2ac1094b0dc5","_cell_guid":"805050eb-22ee-4052-ad1b-6f6d1280fa99","trusted":true}},{"cell_type":"markdown","source":"### We can see that \"StartHesitation\",\"Turn\",\"Walking\" are actually mutually exclusive","metadata":{"_uuid":"38dbc4d5-fb4b-453c-8649-4533ecb7dfb0","_cell_guid":"28c459d5-30a5-4557-9159-e3d348413b15","trusted":true}},{"cell_type":"code","source":"print('_'*25)\nprint('When StartHesitation = 1\\n',train.loc[train['StartHesitation'] == 1][['Walking', 'Turn']].value_counts(),'\\n')\nprint('_'*25)\nprint('When Turn = 1\\n',train.loc[train['Walking'] == 1][['StartHesitation', 'Turn']].value_counts(),'\\n')\nprint('_'*25)\nprint('When Walking = 1\\n',train.loc[train['Turn'] == 1][['StartHesitation', 'Walking']].value_counts(),'\\n')","metadata":{"_uuid":"e0b5ec8f-b956-4d8a-8626-8101a901745a","_cell_guid":"988056e5-6c39-4ad7-8267-cc53dfc8dd6b","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:41:32.358615Z","iopub.execute_input":"2023-06-07T23:41:32.359079Z","iopub.status.idle":"2023-06-07T23:41:32.704699Z","shell.execute_reply.started":"2023-06-07T23:41:32.359044Z","shell.execute_reply":"2023-06-07T23:41:32.703761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### That meas we could have a multiclass objective, instead of multioutput classification objective....","metadata":{"_uuid":"6a2b28d8-9d4f-4334-974a-9f431ad7083c","_cell_guid":"15a20c84-6aee-4ddb-b186-60b1c74cc7cc","trusted":true}},{"cell_type":"markdown","source":"# Feature Engineering\n\nAs our model should be able to clasify timeseries between StartHesitation, Turn and Walking, we can take advantage of clustering algorithm in order to segment diferent time series caracteristics","metadata":{"_uuid":"2e034f11-c13f-4aaf-af84-f9948d33abb6","_cell_guid":"99f26a96-c86c-431d-b598-aae5dc9661ea","trusted":true}},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nfrom sklearn.preprocessing import OneHotEncoder\n\n\n# Step 1: Choose clustering algorithm\ncluster_algo = KMeans(n_clusters=4) #-> We want to identify 4 diferent classes (StartHesitation, Turn, Walking or None)  \n\n# Step 2: Select relevant features\nselected_features = ['Time','AccV', 'AccML','AccAP']  \n\n# Step 3: Perform clustering\ncluster_labels = cluster_algo.fit_predict(train[selected_features])  # data is your original dataset\n\n# Step 4: Encode cluster labels as features\nencoder = OneHotEncoder(sparse=False)\ncluster_features = encoder.fit_transform(cluster_labels.reshape(-1, 1))\n\n# Step 5: Combine cluster features with original features\naugmented_data = np.hstack((train, cluster_features))  # Augmented dataset with original and cluster features","metadata":{"_uuid":"e6e86436-3a09-470f-9750-e153013aca44","_cell_guid":"21df6491-cec1-4f56-945c-c067752b15f7","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:41:32.709364Z","iopub.execute_input":"2023-06-07T23:41:32.712009Z","iopub.status.idle":"2023-06-07T23:42:46.198968Z","shell.execute_reply.started":"2023-06-07T23:41:32.711971Z","shell.execute_reply":"2023-06-07T23:42:46.198008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pd.Series(cluster_labels).value_counts()","metadata":{"_uuid":"c6661761-26b6-4f69-8ee4-b898b8d22ed8","_cell_guid":"36777136-5c51-442d-a14f-12dba2694a92","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:42:46.200649Z","iopub.execute_input":"2023-06-07T23:42:46.201077Z","iopub.status.idle":"2023-06-07T23:42:46.251189Z","shell.execute_reply.started":"2023-06-07T23:42:46.201030Z","shell.execute_reply":"2023-06-07T23:42:46.250108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature1 = train[:].Time\nfeature2 = train[:].AccML\n\n# Plot the clusters\nplt.scatter(feature1, feature2, c=cluster_labels, cmap='viridis')\nplt.xlabel('Feature 1')\nplt.ylabel('Feature 2')\nplt.title('K-means Clustering')\nplt.show()","metadata":{"_uuid":"6b1c6695-7ec5-476e-bfe0-c03a6d2db7cc","_cell_guid":"a77e784e-caa2-421f-a70b-48adee17b8b4","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:42:46.252917Z","iopub.execute_input":"2023-06-07T23:42:46.253269Z","iopub.status.idle":"2023-06-07T23:45:05.011874Z","shell.execute_reply.started":"2023-06-07T23:42:46.253238Z","shell.execute_reply":"2023-06-07T23:45:05.010688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Cluster'] = cluster_labels","metadata":{"_uuid":"701b0d4e-013e-4412-80e7-7cab2790600d","_cell_guid":"77740068-13b1-49cd-980d-96526b5c33e8","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:45:05.013356Z","iopub.execute_input":"2023-06-07T23:45:05.014233Z","iopub.status.idle":"2023-06-07T23:45:05.027115Z","shell.execute_reply.started":"2023-06-07T23:45:05.014198Z","shell.execute_reply":"2023-06-07T23:45:05.025982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.loc[train.Cluster == 1][['StartHesitation', 'Turn', 'Walking']].value_counts()","metadata":{"_uuid":"230ada0f-2116-42f5-a167-d61a7d25f138","_cell_guid":"33fada65-ad02-4b06-8dc1-6c159190df52","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:45:05.028934Z","iopub.execute_input":"2023-06-07T23:45:05.029402Z","iopub.status.idle":"2023-06-07T23:45:05.070943Z","shell.execute_reply.started":"2023-06-07T23:45:05.029369Z","shell.execute_reply":"2023-06-07T23:45:05.069822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Processing pipeline","metadata":{"_uuid":"0c5d93ce-cd59-4be5-bfd4-c0e8cba576ca","_cell_guid":"db6a9f52-67e7-4d5e-b4df-2e86bc773496","trusted":true}},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler\nfrom imblearn.under_sampling import RandomUnderSampler\nfrom imblearn.pipeline import Pipeline\n\n# Define the preprocessing pipeline\npreprocessing_pipeline = Pipeline([\n    ('imputer', SimpleImputer(strategy='mean')),        # Handling missing data\n    ('scaler', StandardScaler()),                       # Standardizing the data\n    ('sampler', RandomUnderSampler(random_state=42)),    # Random under-sampling\n])","metadata":{"_uuid":"4dfe7987-5e03-43ea-a636-489fdbcec560","_cell_guid":"0479b087-1a70-4a54-9efd-12333fdbcc5f","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:45:05.072765Z","iopub.execute_input":"2023-06-07T23:45:05.073550Z","iopub.status.idle":"2023-06-07T23:45:05.363689Z","shell.execute_reply.started":"2023-06-07T23:45:05.073510Z","shell.execute_reply":"2023-06-07T23:45:05.362643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Enabling multiclass classification","metadata":{}},{"cell_type":"code","source":"train['Class'] =  (train['StartHesitation'] * 1) + (train['Turn'] * 2) + (train['Walking'] * 3)\n","metadata":{"execution":{"iopub.status.busy":"2023-06-07T23:45:05.365202Z","iopub.execute_input":"2023-06-07T23:45:05.365876Z","iopub.status.idle":"2023-06-07T23:45:05.496884Z","shell.execute_reply.started":"2023-06-07T23:45:05.365837Z","shell.execute_reply":"2023-06-07T23:45:05.495969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.Class.value_counts().plot(kind='barh')","metadata":{"execution":{"iopub.status.busy":"2023-06-07T23:45:05.498532Z","iopub.execute_input":"2023-06-07T23:45:05.499169Z","iopub.status.idle":"2023-06-07T23:45:05.785626Z","shell.execute_reply.started":"2023-06-07T23:45:05.499124Z","shell.execute_reply":"2023-06-07T23:45:05.784650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training simple model","metadata":{"_uuid":"eef80d66-c508-40c6-b8ca-8352178a5182","_cell_guid":"74ac9f04-ca4d-4a76-aca4-94246858e700","trusted":true}},{"cell_type":"code","source":"import xgboost as xgb\nimport numpy as np\nfrom sklearn.model_selection import TimeSeriesSplit, cross_val_score\nimport matplotlib.pyplot as plt\n\nimport xgboost as xgb\nimport numpy as np\nfrom sklearn.model_selection import TimeSeriesSplit, StratifiedKFold\nfrom sklearn.metrics import average_precision_score\nimport matplotlib.pyplot as plt\n\ndef xgb_params(imbalance_ratio:float) -> dict:\n    params = {\n        \"tree_method\": \"gpu_hist\",\n        \"objective\": \"binary:logistic\",\n        'eval_metric': 'auc',\n        'scale_pos_weight': imbalance_ratio,\n        'learning_rate': 0.1,\n        'max_depth': 5,\n        'min_child_weight': 1,\n        'gamma': 0,\n        'subsample': 0.8,\n        'colsample_bytree': 0.8,\n        'reg_alpha': 0,\n        'reg_lambda': 1,\n        'n_estimators': 100,\n        'random_state': 42\n    }\n    return params\n\ndef train_multioutput_classifier(X, y, num_classes):\n    tscv = StratifiedKFold(n_splits=10)  # Define time series cross-validator\n\n    models = []  # List to store models for each output\n    train_losses = []  # List to store training losses\n    val_losses = []  # List to store validation losses\n\n    for i in range(num_classes):\n        print('_'*50)\n        print(\"Training model for Class \", i)\n        imbalance_ratio = pd.Series(y[:, i]).value_counts()[0]/pd.Series(y[:, i]).value_counts()[1]\n        params = xgb_params(imbalance_ratio)\n        model = xgb.XGBClassifier(**params)\n\n        output_labels = y[:, i]  # Select labels for the current output\n        train_loss_fold = []  # Training loss for each fold\n        val_loss_fold = []  # Validation loss for each fold\n\n        for train_index, test_index in tqdm(tscv.split(X,output_labels)):\n            X_train, X_test = X[train_index], X[test_index]\n            y_train, y_test = output_labels[train_index], output_labels[test_index]\n\n            model.fit(X_train, y_train)\n\n            y_pred_proba_train = model.predict_proba(X_train)[:, 1]\n            y_pred_proba_val = model.predict_proba(X_test)[:, 1]\n\n            train_loss = average_precision_score(y_train, y_pred_proba_train)\n            print(\"Train Loss: \", train_loss)\n            val_loss = average_precision_score(y_test, y_pred_proba_val)\n            print(\"Val Loss: \", val_loss)\n\n            train_loss_fold.append(train_loss)\n            val_loss_fold.append(val_loss)\n\n        models.append(model)\n        train_losses.append(train_loss_fold)\n        val_losses.append(val_loss_fold)\n        print(\"Train Losses: \", train_losses)\n        print(\"Val Losses: \", val_losses)\n\n\n    return models, train_losses, val_losses\n\ndef xgb_multi_params(imbalance_ratio: float) -> dict:\n    params = {\n        \"tree_method\": \"gpu_hist\",\n        \"objective\": \"multi:softmax\",\n        'eval_metric': 'mlogloss',\n        'scale_pos_weight': imbalance_ratio,\n        'learning_rate': 0.1,\n        'max_depth': 5,\n        'min_child_weight': 1,\n        'gamma': 0,\n        'subsample': 0.8,\n        'colsample_bytree': 0.8,\n        'reg_alpha': 0,\n        'reg_lambda': 1,\n        'num_class': 3,  # Number of classes\n        'n_estimators': 100,\n        'random_state': 42\n    }\n    return params\ndef calculate_map(y_true, y_pred_proba):\n    num_classes = y_pred_proba.shape[1]\n    map_scores = []\n    \n    for class_index in range(num_classes):\n        y_true_class = (y_true == class_index).astype(int)\n        y_pred_class = y_pred_proba[:, class_index]\n        \n        ap_score = average_precision_score(y_true_class, y_pred_class)\n        map_scores.append(ap_score)\n    \n    mAP = np.mean(map_scores)\n    return mAP\n\ndef train_multiclass_classifier(X, y):\n    skf = StratifiedKFold(n_splits=10)  # Define stratified cross-validator\n\n    models = []  # List to store models for each class\n    train_losses = []  # List to store training losses\n    val_losses = []  # List to store validation losses\n\n    y_combined = y  # Combine the boolean columns into a single integer column\n\n    print('_' * 50)\n    print(\"Training model for Multiclass Classification\")\n\n    imbalance_ratio = pd.Series(y_combined).value_counts()[0] / pd.Series(y_combined).value_counts()[1]\n    params = xgb_multi_params(imbalance_ratio)\n    model = xgb.XGBClassifier(**params)\n\n    train_loss_fold = []  # Training loss for each fold\n    val_loss_fold = []  # Validation loss for each fold\n\n    for train_index, test_index in tqdm(skf.split(X, y_combined)):\n        X_train, X_test = X[train_index], X[test_index]\n        y_train, y_test = y_combined[train_index], y_combined[test_index]\n\n        model.fit(X_train, y_train)\n\n        y_pred_proba_train = model.predict_proba(X_train)\n        y_pred_proba_val = model.predict_proba(X_test)\n\n        train_loss = calculate_map(y_train, y_pred_proba_train)  # Assuming binary classification\n        print(\"Train Loss: \", train_loss)\n        val_loss = calculate_map(y_test, y_pred_proba_val)  # Assuming binary classification\n        print(\"Val Loss: \", val_loss)\n\n        train_loss_fold.append(train_loss)\n        val_loss_fold.append(val_loss)\n\n    models.append(model)\n    train_losses.append(train_loss_fold)\n    val_losses.append(val_loss_fold)\n    print(\"Train Losses: \", train_losses)\n    print(\"Val Losses: \", val_losses)\n\n    return models, train_losses, val_losses","metadata":{"_uuid":"33be50a0-ba96-4a59-9058-1e0676e7b48e","_cell_guid":"77a080ac-d549-49ce-8cef-bdb5df6faacf","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:48:31.275263Z","iopub.execute_input":"2023-06-07T23:48:31.275922Z","iopub.status.idle":"2023-06-07T23:48:31.311896Z","shell.execute_reply.started":"2023-06-07T23:48:31.275880Z","shell.execute_reply":"2023-06-07T23:48:31.310817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns","metadata":{"_uuid":"3085181d-744a-4b64-8500-0e93d0d7678e","_cell_guid":"fbc1ab26-56c3-4db9-a160-a79556277524","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:48:32.143577Z","iopub.execute_input":"2023-06-07T23:48:32.143995Z","iopub.status.idle":"2023-06-07T23:48:32.151668Z","shell.execute_reply.started":"2023-06-07T23:48:32.143961Z","shell.execute_reply":"2023-06-07T23:48:32.150518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train our simple model\nX = train[['AccV', 'AccML','AccAP', 'Cluster']].values\ny = train['Class'].values#['StartHesitation', 'Turn', 'Walking']].values\nnum_classes = 3\n\n# Fit and transform the training data\nX, y = preprocessing_pipeline.fit_resample(X, y)\n\nmodels, train_losses, val_losses = train_multiclass_classifier(X, y)#, num_classes)\n# Plot training and validation losses\n#fig, axes = plt.subplots(nrows=num_classes, figsize=(8, 6*num_classes))\n\n#for i, ax in enumerate(axes):\n#    ax.plot(range(len(train_losses[i])), train_losses[i], label='Train')\n#    ax.plot(range(len(val_losses[i])), val_losses[i], label='Validation')\n#    ax.set_xlabel('Fold')\n#    ax.set_ylabel('Loss')\n#    ax.set_title('Output {}'.format(i+1))\n#    ax.legend()\n\n#plt.tight_layout()\n#plt.show()","metadata":{"_uuid":"f246d8e7-fed6-4908-8a34-5671357547e2","_cell_guid":"e1064323-df4d-40ab-9712-b5fcb3a4f5f3","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:48:32.775709Z","iopub.execute_input":"2023-06-07T23:48:32.776619Z","iopub.status.idle":"2023-06-07T23:49:52.250182Z","shell.execute_reply.started":"2023-06-07T23:48:32.776572Z","shell.execute_reply":"2023-06-07T23:49:52.249093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\nprint(\"Mean Train Precision\",np.mean(train_losses))\nprint(\"Mean Validation Precision\",np.mean(val_losses))","metadata":{"_uuid":"2da608a3-2c0f-4bf8-ab97-4625236af4fa","_cell_guid":"f44c8085-2e34-4ee5-bd2f-1b507856663b","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:49:58.366674Z","iopub.execute_input":"2023-06-07T23:49:58.367177Z","iopub.status.idle":"2023-06-07T23:49:58.375670Z","shell.execute_reply.started":"2023-06-07T23:49:58.367126Z","shell.execute_reply":"2023-06-07T23:49:58.374601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### It already seems our model's prediction power is better than flipping a coin, at least at local CV 😂","metadata":{"_uuid":"3d369bf0-ae4e-4201-8188-ae6f95388fe6","_cell_guid":"3511e8b3-5ffd-40d1-8452-1c1c5532af05","trusted":true}},{"cell_type":"markdown","source":"## Sample submission","metadata":{"_uuid":"ce7b0516-3fd7-4593-ba53-5d232bdb3f4d","_cell_guid":"5a0006c9-f396-4a6d-8b8e-dfb93a30f500","trusted":true}},{"cell_type":"code","source":"test_files = glob.glob(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/*/*\")\ntest_list = []\nfor f in tqdm(test_files):\n    test_batch = pd.read_csv(f)\n    test_list.append(test_batch)\ntest = pd.concat(test_list)\ntest.reset_index(inplace=True, drop = True)","metadata":{"_uuid":"4bbb5c5b-fe3b-4f0f-be95-a304fdb1fb5c","_cell_guid":"1d359ae4-bf3d-4ab5-8385-11fd011210cb","collapsed":false,"jupyter":{"outputs_hidden":false},"execution":{"iopub.status.busy":"2023-06-07T23:50:05.539476Z","iopub.execute_input":"2023-06-07T23:50:05.539873Z","iopub.status.idle":"2023-06-07T23:50:05.991836Z","shell.execute_reply.started":"2023-06-07T23:50:05.539840Z","shell.execute_reply":"2023-06-07T23:50:05.990750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 1: Perform clustering inference on the test set\ntest_cluster_labels = cluster_algo.predict(test[selected_features])\n\n# Step 2: Encode cluster labels as features\ntest_cluster_features = encoder.transform(test_cluster_labels.reshape(-1, 1))\n\n# Step 3: Combine cluster features with original features\naugmented_test_data = np.hstack((test, test_cluster_features))  # Augmented test dataset with original and cluster features","metadata":{"execution":{"iopub.status.busy":"2023-06-07T23:50:41.254087Z","iopub.execute_input":"2023-06-07T23:50:41.255074Z","iopub.status.idle":"2023-06-07T23:50:41.310900Z","shell.execute_reply.started":"2023-06-07T23:50:41.255026Z","shell.execute_reply":"2023-06-07T23:50:41.309834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['Cluster'] = test_cluster_labels","metadata":{"execution":{"iopub.status.busy":"2023-06-07T23:50:44.391713Z","iopub.execute_input":"2023-06-07T23:50:44.392818Z","iopub.status.idle":"2023-06-07T23:50:44.399188Z","shell.execute_reply.started":"2023-06-07T23:50:44.392744Z","shell.execute_reply":"2023-06-07T23:50:44.398089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#preds = dict()\n#probs = dict()\nX_TEST = test[['AccV', 'AccML','AccAP', 'Cluster']].values\n#for Class, model in enumerate(models):\n#    print('_'*25)\n#   print('Inference for Class: ',Class)\n#    preds.update({Class:model.predict(X_TEST)})\n#    probs.update({Class:model.predict_proba(X_TEST)})\n#    print('DONE')\n\npreds = models[0].predict(X_TEST)\nprobs = models[0].predict_proba(X_TEST)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-08T00:14:21.145578Z","iopub.execute_input":"2023-06-08T00:14:21.146360Z","iopub.status.idle":"2023-06-08T00:14:24.207630Z","shell.execute_reply.started":"2023-06-08T00:14:21.146323Z","shell.execute_reply":"2023-06-08T00:14:24.206486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = pd.DataFrame(preds)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T00:14:38.558724Z","iopub.execute_input":"2023-06-08T00:14:38.559837Z","iopub.status.idle":"2023-06-08T00:14:38.564684Z","shell.execute_reply.started":"2023-06-08T00:14:38.559776Z","shell.execute_reply":"2023-06-08T00:14:38.563659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.columns = ['Class']","metadata":{"execution":{"iopub.status.busy":"2023-06-08T00:16:33.248866Z","iopub.execute_input":"2023-06-08T00:16:33.249590Z","iopub.status.idle":"2023-06-08T00:16:33.254593Z","shell.execute_reply.started":"2023-06-08T00:16:33.249554Z","shell.execute_reply":"2023-06-08T00:16:33.253544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds['StartHesitation'] = 0\npreds['Turn'] = 0 \npreds['Walking'] = 0","metadata":{"execution":{"iopub.status.busy":"2023-06-08T00:15:03.670498Z","iopub.execute_input":"2023-06-08T00:15:03.671246Z","iopub.status.idle":"2023-06-08T00:15:03.680389Z","shell.execute_reply.started":"2023-06-08T00:15:03.671211Z","shell.execute_reply":"2023-06-08T00:15:03.678759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.loc[preds['Class'] == 1, 'StartHesitation'] = 1\npreds.loc[preds['Class'] == 2, 'Turn'] = 1\npreds.loc[preds['Class'] == 3, 'Walking'] = 1","metadata":{"execution":{"iopub.status.busy":"2023-06-08T00:16:54.160065Z","iopub.execute_input":"2023-06-08T00:16:54.160672Z","iopub.status.idle":"2023-06-08T00:16:54.177465Z","shell.execute_reply.started":"2023-06-08T00:16:54.160630Z","shell.execute_reply":"2023-06-08T00:16:54.176499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T00:18:09.186068Z","iopub.execute_input":"2023-06-08T00:18:09.186564Z","iopub.status.idle":"2023-06-08T00:18:09.234047Z","shell.execute_reply.started":"2023-06-08T00:18:09.186521Z","shell.execute_reply":"2023-06-08T00:18:09.232888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"probs = pd.DataFrame(probs)\nprobs.columns = ['0', 'StartHesitation', 'Turn', 'Walking']\nprobs","metadata":{"execution":{"iopub.status.busy":"2023-06-07T23:53:23.211103Z","iopub.execute_input":"2023-06-07T23:53:23.211735Z","iopub.status.idle":"2023-06-07T23:53:23.217203Z","shell.execute_reply.started":"2023-06-07T23:53:23.211689Z","shell.execute_reply":"2023-06-07T23:53:23.216184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-08T00:18:39.446551Z","iopub.execute_input":"2023-06-08T00:18:39.446956Z","iopub.status.idle":"2023-06-08T00:18:39.625970Z","shell.execute_reply.started":"2023-06-08T00:18:39.446924Z","shell.execute_reply":"2023-06-08T00:18:39.624937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#probs[0][:,1]","metadata":{"execution":{"iopub.status.busy":"2023-06-08T00:18:40.470081Z","iopub.execute_input":"2023-06-08T00:18:40.470814Z","iopub.status.idle":"2023-06-08T00:18:40.475439Z","shell.execute_reply.started":"2023-06-08T00:18:40.470757Z","shell.execute_reply":"2023-06-08T00:18:40.474080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub['StartHesitation'] = preds['StartHesitation']#probs[0][:,1]\nsub['Turn'] = preds['Turn']#[1][:,1]\nsub['Walking'] = preds['Walking'] #[2][:,1]","metadata":{"execution":{"iopub.status.busy":"2023-06-08T00:18:40.854334Z","iopub.execute_input":"2023-06-08T00:18:40.855161Z","iopub.status.idle":"2023-06-08T00:18:40.864887Z","shell.execute_reply.started":"2023-06-08T00:18:40.855119Z","shell.execute_reply":"2023-06-08T00:18:40.863836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-08T00:18:42.147262Z","iopub.execute_input":"2023-06-08T00:18:42.147648Z","iopub.status.idle":"2023-06-08T00:18:42.159812Z","shell.execute_reply.started":"2023-06-08T00:18:42.147617Z","shell.execute_reply":"2023-06-08T00:18:42.158567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-06-08T00:18:44.449304Z","iopub.execute_input":"2023-06-08T00:18:44.449697Z","iopub.status.idle":"2023-06-08T00:18:45.308409Z","shell.execute_reply.started":"2023-06-08T00:18:44.449666Z","shell.execute_reply":"2023-06-08T00:18:45.307241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Considerations for further improvement\n#### The proposal of this notebook is to provide a quick EDA and Sample model submission with some robustness to the problem of class imbalance.\nWe assumed some premisses in order to perform this study:\n* TDSCFOG has better quality information, since it was collected in a controled environment. Hence it would be possible to predict DEFOG from TDCSFOG. Althought it would probably be better to train a sample model focused only on the DEFOG series...\n* Eventhough our features have a time-series datastructure, our model can and might outperform time-series approaches, if we consider our objective to be a multiclass/multiouput classification model.\n* It is possible to classify FOG events, just using telemetry (iot sensors) data. Therefore we are not considering events, tasks ans subjects informations. Nevertheless these data could improve our results....\n\nTODO: Possible improvements:\n* Merge events, tasks and subjects dataset to the training set\n* Perform PCA in training set, before using KMEANs, this could possibily improve clustering performance\n* Try diferent techniques for handling imbalanced data\n* Train diferent model for the DEFOG dataset\n* Add TimeSeries related Features\n* Tryout diferent algorithms and hyperparameter tunning","metadata":{}}]}