{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":41880,"databundleVersionId":5677426,"sourceType":"competition"}],"dockerImageVersionId":30408,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import library\nimport os\nimport random\nimport pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:35:33.439557Z","iopub.execute_input":"2023-12-25T19:35:33.439985Z","iopub.status.idle":"2023-12-25T19:35:33.469080Z","shell.execute_reply.started":"2023-12-25T19:35:33.439949Z","shell.execute_reply":"2023-12-25T19:35:33.468318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Pre-processing (DEFOG)🏡","metadata":{}},{"cell_type":"code","source":"DATA_ROOT_DEFOG = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/'\ndefog = pd.DataFrame()\nfor root, dirs, files in os.walk(DATA_ROOT_DEFOG):\n    for name in files:       \n        f = os.path.join(root, name)\n        df_list= pd.read_csv(f)\n        df_list['file']= name.split('.')[0]\n        defog = pd.concat([defog, df_list], axis=0)\n\ndefog","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:35:34.033818Z","iopub.execute_input":"2023-12-25T19:35:34.034569Z","iopub.status.idle":"2023-12-25T19:36:21.163157Z","shell.execute_reply.started":"2023-12-25T19:35:34.034535Z","shell.execute_reply":"2023-12-25T19:36:21.162079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog = defog[(defog['Task']==1)&(defog['Valid']==1)]","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:21.165094Z","iopub.execute_input":"2023-12-25T19:36:21.165412Z","iopub.status.idle":"2023-12-25T19:36:21.756126Z","shell.execute_reply.started":"2023-12-25T19:36:21.165383Z","shell.execute_reply":"2023-12-25T19:36:21.755248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('the shape of defog dataset is {}'.format(defog.shape))","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:21.757347Z","iopub.execute_input":"2023-12-25T19:36:21.757660Z","iopub.status.idle":"2023-12-25T19:36:21.763627Z","shell.execute_reply.started":"2023-12-25T19:36:21.757630Z","shell.execute_reply":"2023-12-25T19:36:21.762488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  Combining it with 🏡defog-metadata.","metadata":{}},{"cell_type":"code","source":"defog_metadata = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv\")\ndefog_metadata","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:21.766675Z","iopub.execute_input":"2023-12-25T19:36:21.767364Z","iopub.status.idle":"2023-12-25T19:36:21.790951Z","shell.execute_reply.started":"2023-12-25T19:36:21.767299Z","shell.execute_reply":"2023-12-25T19:36:21.789981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_m= defog_metadata.merge(defog, how = 'inner', left_on = 'Id', right_on = 'file')\ndefog_m.drop(['file','Valid','Task'], axis = 1, inplace = True)\ndefog_m","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:21.791929Z","iopub.execute_input":"2023-12-25T19:36:21.792205Z","iopub.status.idle":"2023-12-25T19:36:24.181599Z","shell.execute_reply.started":"2023-12-25T19:36:21.792178Z","shell.execute_reply":"2023-12-25T19:36:24.180566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# summary table function\ndef summary(df):\n    print(f'data shape: {df.shape}')\n    summ = pd.DataFrame(df.dtypes, columns=['data type'])\n    summ['#missing'] = df.isnull().sum().values * 100\n    summ['%missing'] = df.isnull().sum().values / len(df)\n    summ['#unique'] = df.nunique().values\n    desc = pd.DataFrame(df.describe(include='all').transpose())\n    summ['min'] = desc['min'].values\n    summ['max'] = desc['max'].values\n    summ['first value'] = df.loc[0].values\n    summ['second value'] = df.loc[1].values\n    summ['third value'] = df.loc[2].values\n    \n    return summ","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:24.182806Z","iopub.execute_input":"2023-12-25T19:36:24.183104Z","iopub.status.idle":"2023-12-25T19:36:24.191508Z","shell.execute_reply.started":"2023-12-25T19:36:24.183074Z","shell.execute_reply":"2023-12-25T19:36:24.190475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary(defog_m)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:24.192765Z","iopub.execute_input":"2023-12-25T19:36:24.193029Z","iopub.status.idle":"2023-12-25T19:36:33.126670Z","shell.execute_reply.started":"2023-12-25T19:36:24.193004Z","shell.execute_reply":"2023-12-25T19:36:33.125682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# garbage collection for memory\nimport gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:33.128156Z","iopub.execute_input":"2023-12-25T19:36:33.128569Z","iopub.status.idle":"2023-12-25T19:36:33.223925Z","shell.execute_reply.started":"2023-12-25T19:36:33.128526Z","shell.execute_reply":"2023-12-25T19:36:33.222658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature engineering and modeling (DEFOG)🏡","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:33.225377Z","iopub.execute_input":"2023-12-25T19:36:33.225707Z","iopub.status.idle":"2023-12-25T19:36:33.576290Z","shell.execute_reply.started":"2023-12-25T19:36:33.225677Z","shell.execute_reply":"2023-12-25T19:36:33.575098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"conditions = [\n    (defog_m['StartHesitation'] == 1),\n    (defog_m['Turn'] == 1),\n    (defog_m['Walking'] == 1)]\nchoices = ['StartHesitation', 'Turn', 'Walking']\ndefog_m['event'] = np.select(conditions, choices, default='Normal')","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:33.580716Z","iopub.execute_input":"2023-12-25T19:36:33.581019Z","iopub.status.idle":"2023-12-25T19:36:34.410781Z","shell.execute_reply.started":"2023-12-25T19:36:33.580990Z","shell.execute_reply":"2023-12-25T19:36:34.409927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_m['event'].value_counts().to_frame().style.background_gradient()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:34.411753Z","iopub.execute_input":"2023-12-25T19:36:34.412023Z","iopub.status.idle":"2023-12-25T19:36:35.145999Z","shell.execute_reply.started":"2023-12-25T19:36:34.411996Z","shell.execute_reply":"2023-12-25T19:36:35.144962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = defog_m[['Visit','Medication','Time','AccV','AccML','AccAP','event']]","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:35.147033Z","iopub.execute_input":"2023-12-25T19:36:35.147320Z","iopub.status.idle":"2023-12-25T19:36:35.868423Z","shell.execute_reply.started":"2023-12-25T19:36:35.147293Z","shell.execute_reply":"2023-12-25T19:36:35.867441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:35.869657Z","iopub.execute_input":"2023-12-25T19:36:35.869982Z","iopub.status.idle":"2023-12-25T19:36:35.875013Z","shell.execute_reply.started":"2023-12-25T19:36:35.869952Z","shell.execute_reply":"2023-12-25T19:36:35.873939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['event'] = le.fit_transform(train_df['event'])\ntrain_df['Medication'] = le.fit_transform(train_df['Medication'])\ntrain_df['Time'] = train_df['Time'].astype(int)\ntrain_df['Visit'] = train_df['Visit'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:35.876358Z","iopub.execute_input":"2023-12-25T19:36:35.876694Z","iopub.status.idle":"2023-12-25T19:36:38.451082Z","shell.execute_reply.started":"2023-12-25T19:36:35.876656Z","shell.execute_reply":"2023-12-25T19:36:38.450179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:38.452320Z","iopub.execute_input":"2023-12-25T19:36:38.452644Z","iopub.status.idle":"2023-12-25T19:36:38.468595Z","shell.execute_reply.started":"2023-12-25T19:36:38.452615Z","shell.execute_reply":"2023-12-25T19:36:38.467544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:38.470008Z","iopub.execute_input":"2023-12-25T19:36:38.470681Z","iopub.status.idle":"2023-12-25T19:36:38.483970Z","shell.execute_reply.started":"2023-12-25T19:36:38.470650Z","shell.execute_reply":"2023-12-25T19:36:38.482780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#pip install -U imbalanced-learn","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:38.485313Z","iopub.execute_input":"2023-12-25T19:36:38.485606Z","iopub.status.idle":"2023-12-25T19:36:38.491776Z","shell.execute_reply.started":"2023-12-25T19:36:38.485578Z","shell.execute_reply":"2023-12-25T19:36:38.490951Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the dataset into features (X) and target variable (y)\nX = train_df.drop(['event'], axis=1)\ny = train_df['event']\n\n# Split the dataset into training and testing sets\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.3, random_state=42)\n\n# Check class distribution before applying downsampling\nprint(\"Class distribution before SMOTE:\")\nprint(y_train.value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:38.492893Z","iopub.execute_input":"2023-12-25T19:36:38.493209Z","iopub.status.idle":"2023-12-25T19:36:39.243834Z","shell.execute_reply.started":"2023-12-25T19:36:38.493178Z","shell.execute_reply":"2023-12-25T19:36:39.242838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:39.245174Z","iopub.execute_input":"2023-12-25T19:36:39.245555Z","iopub.status.idle":"2023-12-25T19:36:39.261884Z","shell.execute_reply.started":"2023-12-25T19:36:39.245521Z","shell.execute_reply":"2023-12-25T19:36:39.260916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\n\n# Apply SMOTE to the training data with sampling_strategy set to the desired ratio\nsmote = SMOTE(sampling_strategy='auto', random_state=42)\nX_train_resampled, y_train_resampled = smote.fit_resample(X_train, y_train)\n\n# Check class distribution after applying SMOTE\nprint(\"\\nClass distribution after SMOTE:\")\nprint(pd.Series(y_train_resampled).value_counts())\n\n# Check the shapes of the resampled datasets and the testing set\nprint(\"\\nShapes after SMOTE:\")\nprint(\"X_train_resampled:\", X_train_resampled.shape)\nprint(\"y_train_resampled:\", y_train_resampled.shape)\nprint(\"X_test:\", X_test.shape)\nprint(\"y_test:\", y_test.shape)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:36:39.263341Z","iopub.execute_input":"2023-12-25T19:36:39.263628Z","iopub.status.idle":"2023-12-25T19:36:44.811939Z","shell.execute_reply.started":"2023-12-25T19:36:39.263600Z","shell.execute_reply":"2023-12-25T19:36:44.810760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:39:10.125264Z","iopub.execute_input":"2023-12-25T19:39:10.126025Z","iopub.status.idle":"2023-12-25T19:39:10.143140Z","shell.execute_reply.started":"2023-12-25T19:39:10.125990Z","shell.execute_reply":"2023-12-25T19:39:10.142170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Classifiers**","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom xgboost import XGBClassifier\nfrom lightgbm import LGBMClassifier\nfrom sklearn.neighbors import KNeighborsClassifier","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:39:58.101772Z","iopub.execute_input":"2023-12-25T19:39:58.102861Z","iopub.status.idle":"2023-12-25T19:40:01.118963Z","shell.execute_reply.started":"2023-12-25T19:39:58.102818Z","shell.execute_reply":"2023-12-25T19:40:01.118056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import the accuracy_score function\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report  ","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:40:01.120606Z","iopub.execute_input":"2023-12-25T19:40:01.120917Z","iopub.status.idle":"2023-12-25T19:40:01.125553Z","shell.execute_reply.started":"2023-12-25T19:40:01.120887Z","shell.execute_reply":"2023-12-25T19:40:01.124430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Evaluation matrix function\ndef evaluate_metrics(y_true, y_pred):\n    \n    # Calculate metrics\n    accuracy = accuracy_score(y_true, y_pred)\n    conf_matrix = confusion_matrix(y_true, y_pred)\n    classification_rep = classification_report(y_true, y_pred)\n\n    # Print the metrics\n    print(f'Accuracy: {accuracy}')\n    print(f'Confusion Matrix:\\n{conf_matrix}')\n    print(f'Classification Report:\\n{classification_rep}')\n","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:40:01.127134Z","iopub.execute_input":"2023-12-25T19:40:01.127637Z","iopub.status.idle":"2023-12-25T19:40:01.135962Z","shell.execute_reply.started":"2023-12-25T19:40:01.127597Z","shell.execute_reply":"2023-12-25T19:40:01.134964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Forest","metadata":{}},{"cell_type":"code","source":"# Random Forest\n\n# Initialize the Random Forest classifier\nrf_classifier = RandomForestClassifier(n_estimators=100, random_state=42)\n\n# Train the classifier\nrf_classifier.fit(X_train_resampled, y_train_resampled)\n\ny_pred_1 = rf_classifier.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T19:40:01.138160Z","iopub.execute_input":"2023-12-25T19:40:01.138420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_metrics(y_test, y_pred_1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Decision Tree","metadata":{}},{"cell_type":"code","source":"# Decision tree classifier\n\n# Initialize the Decision Tree Classifier\nclf = DecisionTreeClassifier(random_state=42)\n\n# Train the classifier\nclf.fit(X_train_resampled, y_train_resampled)\n\n# Make predictions on the test set\ny_pred_2 = clf.predict(X_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_metrics(y_test, y_pred_2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Naive Bayes","metadata":{}},{"cell_type":"code","source":"# Naive Bayes Classifier\n# Initialize the Naive Bayes Classifier\nclf = GaussianNB()\n\n# Train the classifier\nclf.fit(X_train_resampled, y_train_resampled)\n\n# Make predictions on the test set\ny_pred_3 = clf.predict(X_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_metrics(y_test, y_pred_3)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# XG Boost","metadata":{}},{"cell_type":"code","source":"# XG boost\n\n# Initialize the XGBoost classifier\nxgb_classifier = XGBClassifier(n_estimators=100, learning_rate=0.1, max_depth=3, random_state=42)\n\n# Train the classifier\nxgb_classifier.fit(X_train_resampled, y_train_resampled)\n\n# Make predictions on the test set\ny_pred_4 = xgb_classifier.predict(X_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_metrics(y_test, y_pred_4)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LGBM","metadata":{}},{"cell_type":"code","source":"# LGBM\n\n# Initialize the LightGBM classifier\nlgbm_classifier = LGBMClassifier(n_estimators=100, learning_rate=0.1, max_depth=3, random_state=42)\n\n# Train the classifier\nlgbm_classifier.fit(X_train_resampled, y_train_resampled)\n\n# Make predictions on the test set\ny_pred_5 = lgbm_classifier.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T12:25:04.463521Z","iopub.execute_input":"2023-12-17T12:25:04.464285Z","iopub.status.idle":"2023-12-17T12:26:59.967394Z","shell.execute_reply.started":"2023-12-17T12:25:04.464243Z","shell.execute_reply":"2023-12-17T12:26:59.966494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_metrics(y_test, y_pred_5)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T12:26:59.968940Z","iopub.execute_input":"2023-12-17T12:26:59.969262Z","iopub.status.idle":"2023-12-17T12:27:01.878150Z","shell.execute_reply.started":"2023-12-17T12:26:59.969230Z","shell.execute_reply":"2023-12-17T12:27:01.877087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# KNN","metadata":{}},{"cell_type":"code","source":"# KNN\n\n# Initialize the KNN classifier\nknn_classifier = KNeighborsClassifier(n_neighbors=2)  # You can adjust the number of neighbors (n_neighbors) based on your preference\n\n# Train the classifier\nknn_classifier.fit(X_train_resampled, y_train_resampled)\n\n# Make predictions on the test set\ny_pred_6 = knn_classifier.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T12:29:53.867628Z","iopub.execute_input":"2023-12-17T12:29:53.868622Z","iopub.status.idle":"2023-12-17T12:31:09.875009Z","shell.execute_reply.started":"2023-12-17T12:29:53.868579Z","shell.execute_reply":"2023-12-17T12:31:09.874049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_metrics(y_test, y_pred_6)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T12:31:09.876827Z","iopub.execute_input":"2023-12-17T12:31:09.877127Z","iopub.status.idle":"2023-12-17T12:31:11.673508Z","shell.execute_reply.started":"2023-12-17T12:31:09.877098Z","shell.execute_reply":"2023-12-17T12:31:11.672427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test dataset (for submission)","metadata":{}},{"cell_type":"code","source":"test_defog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog/02ab235146.csv'\ntest_defog = pd.read_csv(test_defog_path)\nname = os.path.basename(test_defog_path)\nid_value = name.split('.')[0]\ntest_defog['Id_value'] = id_value\ntest_defog['Id'] = test_defog['Id_value'].astype(str) + '_' + test_defog['Time'].astype(str)\ntest_defog = test_defog[['Id','AccV','AccML','AccAP']]\ntest_defog.set_index('Id',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-17T12:19:52.237986Z","iopub.execute_input":"2023-12-17T12:19:52.238407Z","iopub.status.idle":"2023-12-17T12:19:53.148510Z","shell.execute_reply.started":"2023-12-17T12:19:52.238371Z","shell.execute_reply":"2023-12-17T12:19:53.147336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict event probability\ntest_defog_pred=rf_classifier.predict(test_defog)\ntest_defog['event'] = test_defog_pred","metadata":{"execution":{"iopub.status.busy":"2023-12-17T12:31:11.674636Z","iopub.execute_input":"2023-12-17T12:31:11.674957Z","iopub.status.idle":"2023-12-17T12:31:11.682020Z","shell.execute_reply.started":"2023-12-17T12:31:11.674926Z","shell.execute_reply":"2023-12-17T12:31:11.680971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_defog","metadata":{"execution":{"iopub.status.busy":"2023-12-15T10:25:47.842868Z","iopub.execute_input":"2023-12-15T10:25:47.843307Z","iopub.status.idle":"2023-12-15T10:25:47.859188Z","shell.execute_reply.started":"2023-12-15T10:25:47.843272Z","shell.execute_reply":"2023-12-15T10:25:47.857947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# expand event column it to three columns\ntest_defog['StartHesitation'] = np.where(test_defog['event']==1, 1, 0)\ntest_defog['Turn'] = np.where(test_defog['event']==2, 1, 0)\ntest_defog['Walking'] = np.where(test_defog['event']==3, 1, 0)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T10:25:56.113531Z","iopub.execute_input":"2023-12-15T10:25:56.114720Z","iopub.status.idle":"2023-12-15T10:25:56.126693Z","shell.execute_reply.started":"2023-12-15T10:25:56.114656Z","shell.execute_reply":"2023-12-15T10:25:56.125583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_defog","metadata":{"execution":{"iopub.status.busy":"2023-12-15T10:25:57.137112Z","iopub.execute_input":"2023-12-15T10:25:57.138298Z","iopub.status.idle":"2023-12-15T10:25:57.156074Z","shell.execute_reply.started":"2023-12-15T10:25:57.138256Z","shell.execute_reply":"2023-12-15T10:25:57.154990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_defog['event'].value_counts(),'\\n')\nprint(test_defog['StartHesitation'].value_counts(),'\\n')\nprint(test_defog['Turn'].value_counts(),'\\n')\nprint(test_defog['Walking'].value_counts(),'\\n')","metadata":{"execution":{"iopub.status.busy":"2023-12-15T10:26:01.170992Z","iopub.execute_input":"2023-12-15T10:26:01.172077Z","iopub.status.idle":"2023-12-15T10:26:01.189994Z","shell.execute_reply.started":"2023-12-15T10:26:01.172034Z","shell.execute_reply":"2023-12-15T10:26:01.188899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **TDCSFOG🥼**","metadata":{}},{"cell_type":"code","source":"DATA_ROOT_TDCSFOG = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/'\ntdcsfog = pd.DataFrame()\nfor root, dirs, files in os.walk(DATA_ROOT_TDCSFOG):\n    for name in files:       \n        f = os.path.join(root, name)\n        df_list= pd.read_csv(f)\n        words = name.split('.')[0]\n        df_list['file']= name.split('.')[0]\n        tdcsfog = pd.concat([tdcsfog, df_list], axis=0)\ntdcsfog","metadata":{"execution":{"iopub.status.busy":"2023-12-16T10:52:54.233064Z","iopub.execute_input":"2023-12-16T10:52:54.234223Z","iopub.status.idle":"2023-12-16T10:55:13.686799Z","shell.execute_reply.started":"2023-12-16T10:52:54.234168Z","shell.execute_reply":"2023-12-16T10:55:13.685632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##  Combining it with 🥼tdcsfog-metadata.","metadata":{"execution":{"iopub.status.busy":"2023-12-15T08:43:53.283170Z","iopub.execute_input":"2023-12-15T08:43:53.283491Z","iopub.status.idle":"2023-12-15T08:43:53.288041Z","shell.execute_reply.started":"2023-12-15T08:43:53.283461Z","shell.execute_reply":"2023-12-15T08:43:53.287090Z"}}},{"cell_type":"code","source":"tdcsfog_metadata = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv\")\ntdcsfog_metadata","metadata":{"execution":{"iopub.status.busy":"2023-12-16T10:55:13.688971Z","iopub.execute_input":"2023-12-16T10:55:13.689784Z","iopub.status.idle":"2023-12-16T10:55:13.711837Z","shell.execute_reply.started":"2023-12-16T10:55:13.689737Z","shell.execute_reply":"2023-12-16T10:55:13.710826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_m= tdcsfog_metadata.merge(tdcsfog, how = 'inner', left_on = 'Id', right_on = 'file')\ntdcsfog_m.drop(['file'], axis = 1, inplace = True)\ntdcsfog_m","metadata":{"execution":{"iopub.status.busy":"2023-12-16T12:57:41.082803Z","iopub.execute_input":"2023-12-16T12:57:41.083592Z","iopub.status.idle":"2023-12-16T12:57:44.852078Z","shell.execute_reply.started":"2023-12-16T12:57:41.083542Z","shell.execute_reply":"2023-12-16T12:57:44.851052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"summary(tdcsfog_m)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:00:35.125977Z","iopub.execute_input":"2023-12-16T13:00:35.126932Z","iopub.status.idle":"2023-12-16T13:00:51.200025Z","shell.execute_reply.started":"2023-12-16T13:00:35.126888Z","shell.execute_reply":"2023-12-16T13:00:51.198874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# garbage collection for memory\nimport gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:00:54.100086Z","iopub.execute_input":"2023-12-16T13:00:54.100916Z","iopub.status.idle":"2023-12-16T13:00:54.436314Z","shell.execute_reply.started":"2023-12-16T13:00:54.100875Z","shell.execute_reply":"2023-12-16T13:00:54.435256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature engineering and modeling (tdcsDEFOG)🥼","metadata":{}},{"cell_type":"code","source":"conditions = [\n    (tdcsfog_m['StartHesitation'] == 1),\n    (tdcsfog_m['Turn'] == 1),\n    (tdcsfog_m['Walking'] == 1)]\nchoices = ['StartHesitation', 'Turn', 'Walking']\ntdcsfog_m['event'] = np.select(conditions, choices, default='Normal')","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:01:04.228464Z","iopub.execute_input":"2023-12-16T13:01:04.229521Z","iopub.status.idle":"2023-12-16T13:01:05.605922Z","shell.execute_reply.started":"2023-12-16T13:01:04.229465Z","shell.execute_reply":"2023-12-16T13:01:05.605012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_m['event'].value_counts().to_frame().style.background_gradient()","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:01:08.015051Z","iopub.execute_input":"2023-12-16T13:01:08.015867Z","iopub.status.idle":"2023-12-16T13:01:09.322787Z","shell.execute_reply.started":"2023-12-16T13:01:08.015826Z","shell.execute_reply":"2023-12-16T13:01:09.321628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_tdcs = tdcsfog_m[['Visit','Medication','Time','AccV','AccML','AccAP','event']]","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:03:28.256685Z","iopub.execute_input":"2023-12-16T13:03:28.257708Z","iopub.status.idle":"2023-12-16T13:03:29.409000Z","shell.execute_reply.started":"2023-12-16T13:03:28.257663Z","shell.execute_reply":"2023-12-16T13:03:29.408042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_tdcs['event'] = le.fit_transform(train_df_tdcs['event'])\ntrain_df_tdcs['Medication'] = le.fit_transform(train_df_tdcs['Medication'])\ntrain_df_tdcs['Time'] = train_df_tdcs['Time'].astype(int)\ntrain_df_tdcs['Visit'] = train_df_tdcs['Visit'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:03:36.227719Z","iopub.execute_input":"2023-12-16T13:03:36.228535Z","iopub.status.idle":"2023-12-16T13:03:38.508815Z","shell.execute_reply.started":"2023-12-16T13:03:36.228495Z","shell.execute_reply":"2023-12-16T13:03:38.507897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df_tdcs.info()","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:03:40.496408Z","iopub.execute_input":"2023-12-16T13:03:40.497480Z","iopub.status.idle":"2023-12-16T13:03:40.508220Z","shell.execute_reply.started":"2023-12-16T13:03:40.497437Z","shell.execute_reply":"2023-12-16T13:03:40.506911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Split the dataset into features (X) and target variable (y)\nX2 = train_df_tdcs.drop(['event'], axis=1)\ny2 = train_df_tdcs['event']\n\n# Split the dataset into training and testing sets\nX_train2, X_test2, y_train2, y_test2 = train_test_split(X2, y2, test_size=0.3, random_state=42)\n\n# Check class distribution before applying downsampling\nprint(\"Class distribution before SMOTE:\")\nprint(y_train2.value_counts())","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test2.value_counts()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from imblearn.over_sampling import SMOTE\n\n# Apply SMOTE to the training data with sampling_strategy set to the desired ratio\nsmote = SMOTE(sampling_strategy='auto', random_state=42)\nX_train_resampled2, y_train_resampled2 = smote.fit_resample(X_train2, y_train2)\n\n# Check class distribution after applying SMOTE\nprint(\"\\nClass distribution after SMOTE:\")\nprint(pd.Series(y_train_resampled2).value_counts())\n\n# Check the shapes of the resampled datasets and the testing set\nprint(\"\\nShapes after SMOTE:\")\nprint(\"X_train_resampled:\", X_train_resampled2.shape)\nprint(\"y_train_resampled:\", y_train_resampled2.shape)\nprint(\"X_test:\", X_test2.shape)\nprint(\"y_test:\", y_test2.shape)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Classifiers**","metadata":{}},{"cell_type":"code","source":"# Decision tree classifier\n\n# Initialize the Decision Tree Classifier\ndt_clf2 = DecisionTreeClassifier(random_state=42)\n\n# Train the classifier\ndt_clf2.fit(X_train2, y_train2)\n\n# Make predictions on the test set\ny_pred_tdcs1 = dt_clf2.predict(X_test2)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:22:02.391252Z","iopub.execute_input":"2023-12-16T13:22:02.391906Z","iopub.status.idle":"2023-12-16T13:22:08.760568Z","shell.execute_reply.started":"2023-12-16T13:22:02.391865Z","shell.execute_reply":"2023-12-16T13:22:08.759345Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_metrics(y_test2, y_pred_tdcs1)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:22:08.762632Z","iopub.execute_input":"2023-12-16T13:22:08.762953Z","iopub.status.idle":"2023-12-16T13:22:09.178524Z","shell.execute_reply.started":"2023-12-16T13:22:08.762920Z","shell.execute_reply":"2023-12-16T13:22:09.177455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Random Forest\n\n# Initialize the Random Forest classifier\nrf_clf2 = RandomForestClassifier(n_estimators=100, random_state=42)\n\n# Train the classifier\nrf_clf2.fit(X_train2, y_train2)\n\ny_pred_tdcs2 = rf_clf2.predict(X_test2)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:22:09.179913Z","iopub.execute_input":"2023-12-16T13:22:09.180312Z","iopub.status.idle":"2023-12-16T13:24:52.228744Z","shell.execute_reply.started":"2023-12-16T13:22:09.180277Z","shell.execute_reply":"2023-12-16T13:24:52.227802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_metrics(y_test2, y_pred_tdcs2)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:24:52.230727Z","iopub.execute_input":"2023-12-16T13:24:52.231043Z","iopub.status.idle":"2023-12-16T13:24:52.651559Z","shell.execute_reply.started":"2023-12-16T13:24:52.231011Z","shell.execute_reply":"2023-12-16T13:24:52.650393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Naive Bayes Classifier\n# Initialize the Naive Bayes Classifier\nnb_clf2 = GaussianNB()\n\n# Train the classifier\nnb_clf2.fit(X_train, y_train)\n\n# Make predictions on the test set\ny_pred_tdcs3 = nb_clf2.predict(X_test2)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:24:52.652969Z","iopub.execute_input":"2023-12-16T13:24:52.653369Z","iopub.status.idle":"2023-12-16T13:24:52.822965Z","shell.execute_reply.started":"2023-12-16T13:24:52.653332Z","shell.execute_reply":"2023-12-16T13:24:52.822040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_metrics(y_test2, y_pred_tdcs3)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:24:52.824085Z","iopub.execute_input":"2023-12-16T13:24:52.824397Z","iopub.status.idle":"2023-12-16T13:24:53.246922Z","shell.execute_reply.started":"2023-12-16T13:24:52.824366Z","shell.execute_reply":"2023-12-16T13:24:53.245760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# XG boost\n\n# Initialize the XGBoost classifier\nxgb_clf2 = XGBClassifier(n_estimators=100, learning_rate=0.1, max_depth=3, random_state=42)\n\n# Train the classifier\nxgb_clf2.fit(X_train2, y_train2)\n\n# Make predictions on the test set\ny_pred_tdcs4 = xgb_clf2.predict(X_test2)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:25:39.934941Z","iopub.execute_input":"2023-12-16T13:25:39.935328Z","iopub.status.idle":"2023-12-16T13:26:51.328686Z","shell.execute_reply.started":"2023-12-16T13:25:39.935293Z","shell.execute_reply":"2023-12-16T13:26:51.327644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_metrics(y_test2, y_pred_tdcs4)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:26:51.330465Z","iopub.execute_input":"2023-12-16T13:26:51.331116Z","iopub.status.idle":"2023-12-16T13:26:51.756551Z","shell.execute_reply.started":"2023-12-16T13:26:51.331067Z","shell.execute_reply":"2023-12-16T13:26:51.755325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# LGBM\n\n# Initialize the LightGBM classifier\nlgbm_clf2 = LGBMClassifier(n_estimators=100, learning_rate=0.1, max_depth=3, random_state=42)\n\n# Train the classifier\nlgbm_clf2.fit(X_train2, y_train2)\n\n# Make predictions on the test set\ny_pred_tdcs5 = lgbm_clf2.predict(X_test2)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:27:23.604237Z","iopub.execute_input":"2023-12-16T13:27:23.604649Z","iopub.status.idle":"2023-12-16T13:27:31.223590Z","shell.execute_reply.started":"2023-12-16T13:27:23.604608Z","shell.execute_reply":"2023-12-16T13:27:31.222647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_metrics(y_test2, y_pred_tdcs5)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:27:31.225147Z","iopub.execute_input":"2023-12-16T13:27:31.226022Z","iopub.status.idle":"2023-12-16T13:27:31.641975Z","shell.execute_reply.started":"2023-12-16T13:27:31.225980Z","shell.execute_reply":"2023-12-16T13:27:31.640881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# KNN\n\n# Initialize the KNN classifier\nknn_clf2 = KNeighborsClassifier(n_neighbors=5)  # You can adjust the number of neighbors (n_neighbors) based on your preference\n\n# Train the classifier\nknn_clf2.fit(X_train2, y_train2)\n\n# Make predictions on the test set\ny_pred_tdcs6 = knn_clf2.predict(X_test2)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:27:35.209714Z","iopub.execute_input":"2023-12-16T13:27:35.210110Z","iopub.status.idle":"2023-12-16T13:27:45.051515Z","shell.execute_reply.started":"2023-12-16T13:27:35.210073Z","shell.execute_reply":"2023-12-16T13:27:45.050626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaluate_metrics(y_test2, y_pred_tdcs6)","metadata":{"execution":{"iopub.status.busy":"2023-12-16T13:27:45.053181Z","iopub.execute_input":"2023-12-16T13:27:45.053500Z","iopub.status.idle":"2023-12-16T13:27:45.467355Z","shell.execute_reply.started":"2023-12-16T13:27:45.053469Z","shell.execute_reply":"2023-12-16T13:27:45.466307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test dataset (for submission)","metadata":{}},{"cell_type":"code","source":"test_tdcsfog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog/003f117e14.csv'\ntest_tdcsfog = pd.read_csv(test_tdcsfog_path)\nname = os.path.basename(test_tdcsfog_path)\nid_value = name.split('.')[0]\ntest_tdcsfog['Id_value'] = id_value\ntest_tdcsfog['Id'] = test_tdcsfog['Id_value'].astype(str) + '_' + test_tdcsfog['Time'].astype(str)\ntest_tdcsfog = test_tdcsfog[['Id','AccV','AccML','AccAP']]\ntest_tdcsfog.set_index('Id',inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T11:37:08.177830Z","iopub.execute_input":"2023-12-15T11:37:08.178249Z","iopub.status.idle":"2023-12-15T11:37:08.215941Z","shell.execute_reply.started":"2023-12-15T11:37:08.178209Z","shell.execute_reply":"2023-12-15T11:37:08.215010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog","metadata":{"execution":{"iopub.status.busy":"2023-12-15T11:37:08.217957Z","iopub.execute_input":"2023-12-15T11:37:08.218297Z","iopub.status.idle":"2023-12-15T11:37:08.233434Z","shell.execute_reply.started":"2023-12-15T11:37:08.218264Z","shell.execute_reply":"2023-12-15T11:37:08.232220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog_pred=rf_clf2.predict(test_tdcsfog)\ntest_tdcsfog['event'] = test_tdcsfog_pred","metadata":{"execution":{"iopub.status.busy":"2023-12-15T11:38:24.797908Z","iopub.execute_input":"2023-12-15T11:38:24.798925Z","iopub.status.idle":"2023-12-15T11:38:24.828858Z","shell.execute_reply.started":"2023-12-15T11:38:24.798881Z","shell.execute_reply":"2023-12-15T11:38:24.827949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog","metadata":{"execution":{"iopub.status.busy":"2023-12-15T11:38:25.230988Z","iopub.execute_input":"2023-12-15T11:38:25.231993Z","iopub.status.idle":"2023-12-15T11:38:25.246358Z","shell.execute_reply.started":"2023-12-15T11:38:25.231951Z","shell.execute_reply":"2023-12-15T11:38:25.245237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog['StartHesitation'] = np.where(test_tdcsfog['event']==1, 1, 0)\ntest_tdcsfog['Turn'] = np.where(test_tdcsfog['event']==2, 1, 0)\ntest_tdcsfog['Walking'] = np.where(test_tdcsfog['event']==3, 1, 0)\ntest_tdcsfog.reset_index('Id', inplace=True)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T11:38:28.066797Z","iopub.execute_input":"2023-12-15T11:38:28.067572Z","iopub.status.idle":"2023-12-15T11:38:28.077097Z","shell.execute_reply.started":"2023-12-15T11:38:28.067534Z","shell.execute_reply":"2023-12-15T11:38:28.076054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog","metadata":{"execution":{"iopub.status.busy":"2023-12-15T11:38:29.034352Z","iopub.execute_input":"2023-12-15T11:38:29.035452Z","iopub.status.idle":"2023-12-15T11:38:29.053375Z","shell.execute_reply.started":"2023-12-15T11:38:29.035409Z","shell.execute_reply":"2023-12-15T11:38:29.052098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tdcsfog.shape","metadata":{"execution":{"iopub.status.busy":"2023-12-15T11:38:31.129978Z","iopub.execute_input":"2023-12-15T11:38:31.130923Z","iopub.status.idle":"2023-12-15T11:38:31.137436Z","shell.execute_reply.started":"2023-12-15T11:38:31.130883Z","shell.execute_reply":"2023-12-15T11:38:31.136291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_tdcsfog['event'].value_counts(),'\\n')\nprint(test_tdcsfog['StartHesitation'].value_counts(),'\\n')\nprint(test_tdcsfog['Turn'].value_counts(),'\\n')\nprint(test_tdcsfog['Walking'].value_counts(),'\\n')","metadata":{"execution":{"iopub.status.busy":"2023-12-15T11:38:32.345550Z","iopub.execute_input":"2023-12-15T11:38:32.346339Z","iopub.status.idle":"2023-12-15T11:38:32.356435Z","shell.execute_reply.started":"2023-12-15T11:38:32.346297Z","shell.execute_reply":"2023-12-15T11:38:32.355165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.concat([test_tdcsfog,test_defog])\nsubmit = submit[['Id', 'StartHesitation', 'Turn','Walking']]","metadata":{"execution":{"iopub.status.busy":"2023-12-15T11:38:34.652619Z","iopub.execute_input":"2023-12-15T11:38:34.653594Z","iopub.status.idle":"2023-12-15T11:38:34.683213Z","shell.execute_reply.started":"2023-12-15T11:38:34.653551Z","shell.execute_reply":"2023-12-15T11:38:34.682367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-12-15T11:38:35.738551Z","iopub.execute_input":"2023-12-15T11:38:35.739742Z","iopub.status.idle":"2023-12-15T11:38:36.039381Z","shell.execute_reply.started":"2023-12-15T11:38:35.739700Z","shell.execute_reply":"2023-12-15T11:38:36.038262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv('/kaggle/working/submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-15T11:38:38.457608Z","iopub.execute_input":"2023-12-15T11:38:38.458009Z","iopub.status.idle":"2023-12-15T11:38:38.529381Z","shell.execute_reply.started":"2023-12-15T11:38:38.457971Z","shell.execute_reply":"2023-12-15T11:38:38.528500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}