{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"# import os\n# import random\n# import cv2\n# import pandas as pd\n# import numpy as np\n# import plotly.express as px\n# import matplotlib.pyplot as plt","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class color:\n#    PURPLE = '\\033[95m'\n#    CYAN = '\\033[96m'\n#    DARKCYAN = '\\033[36m'\n#    BLUE = '\\033[94m'\n#    GREEN = '\\033[92m'\n#    YELLOW = '\\033[93m'\n#    RED = '\\033[91m'\n#    BOLD = '\\033[1m'\n#    UNDERLINE = '\\033[4m'\n#    END = '\\033[0m'","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:11:44.989897Z","iopub.execute_input":"2023-05-03T13:11:44.990309Z","iopub.status.idle":"2023-05-03T13:11:44.996743Z","shell.execute_reply.started":"2023-05-03T13:11:44.990268Z","shell.execute_reply":"2023-05-03T13:11:44.995785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# temp = len(os.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog\"))\n# print(\n#     f\"Number of files in folder tdcsfog/: {color.BLUE}{temp}{color.END}\",\n# )\n# temp2 = len(os.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog\"))\n# print(\n#     f\"Number of files in folder defog/: {color.BLUE}{temp2}{color.END}\",\n# )\n# temp3 = len(os.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/notype\"))\n# print(\n#     f\"Number of files in folder notype/: {color.BLUE}{temp3}{color.END}\",\n# )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:11:48.007582Z","iopub.execute_input":"2023-05-03T13:11:48.007965Z","iopub.status.idle":"2023-05-03T13:11:48.084528Z","shell.execute_reply.started":"2023-05-03T13:11:48.007935Z","shell.execute_reply":"2023-05-03T13:11:48.083383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Overview - tdcsfog","metadata":{}},{"cell_type":"code","source":"# train_tdcsfog_example_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/003f117e14.csv\")\n# temp = len(train_tdcsfog_example_df)\n# print(\n#     f\"Length of dataframe: {color.BLUE}{temp}{color.END}\",\n# )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:12:13.609270Z","iopub.execute_input":"2023-05-03T13:12:13.609646Z","iopub.status.idle":"2023-05-03T13:12:13.646411Z","shell.execute_reply.started":"2023-05-03T13:12:13.609616Z","shell.execute_reply":"2023-05-03T13:12:13.644854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_tdcsfog_example_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-01T15:51:01.115490Z","iopub.execute_input":"2023-05-01T15:51:01.116064Z","iopub.status.idle":"2023-05-01T15:51:01.145473Z","shell.execute_reply.started":"2023-05-01T15:51:01.116017Z","shell.execute_reply":"2023-05-01T15:51:01.144301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_tdcsfog_example_df.describe()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-01T15:51:04.214202Z","iopub.execute_input":"2023-05-01T15:51:04.214640Z","iopub.status.idle":"2023-05-01T15:51:04.261352Z","shell.execute_reply.started":"2023-05-01T15:51:04.214601Z","shell.execute_reply":"2023-05-01T15:51:04.260420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for column in ['AccV','AccML','AccAP']:\n#     fig = px.line(train_tdcsfog_example_df, x=\"Time\", y=column, color_discrete_sequence=['darkslateblue'])\n#     fig.update_layout(\n#         title={\n#             'text': f\"{column} Time Siries\",\n#             'y':0.95,\n#             'x':0.5,\n#             'xanchor': 'center',\n#             'yanchor': 'top'\n#         }\n#     )\n#     fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T15:51:36.525983Z","iopub.execute_input":"2023-05-01T15:51:36.526410Z","iopub.status.idle":"2023-05-01T15:51:38.171731Z","shell.execute_reply.started":"2023-05-01T15:51:36.526371Z","shell.execute_reply":"2023-05-01T15:51:38.170657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# corr_mat = np.round(train_tdcsfog_example_df.drop(\n#     [\"StartHesitation\", \"Walking\"], axis=1\n# ).corr(), 3)\n\n# fig = px.imshow(\n#     corr_mat,\n#     x=corr_mat.columns,\n#     y=corr_mat.columns, \n#     text_auto=True\n#    )\n# fig.update_xaxes(side=\"bottom\")\n# fig.update_layout(\n#     title={\n#         'text': \"Correlation Matrix of tdcsfog train instance\",\n#         'y':0.95,\n#         'x':0.5,\n#         'xanchor': 'center',\n#         'yanchor': 'top'\n#     }\n# )\n\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T15:52:47.535489Z","iopub.execute_input":"2023-05-01T15:52:47.535936Z","iopub.status.idle":"2023-05-01T15:52:47.626703Z","shell.execute_reply.started":"2023-05-01T15:52:47.535894Z","shell.execute_reply":"2023-05-01T15:52:47.625719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Overview - unlabeled","metadata":{}},{"cell_type":"code","source":"# temp = len(os.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/unlabeled\"))\n# print(\n#     f\"Number of files in folder unlabeled/: {color.BLUE}{temp}{color.END}\",\n# )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:16:27.648027Z","iopub.execute_input":"2023-05-03T13:16:27.648474Z","iopub.status.idle":"2023-05-03T13:16:27.666654Z","shell.execute_reply.started":"2023-05-03T13:16:27.648378Z","shell.execute_reply":"2023-05-03T13:16:27.665002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# unlabeled_example_df = pd.read_parquet(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/unlabeled/00c4c9313d.parquet\")\n# unlabeled_example_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:16:44.434111Z","iopub.execute_input":"2023-05-03T13:16:44.434529Z","iopub.status.idle":"2023-05-03T13:16:53.340686Z","shell.execute_reply.started":"2023-05-03T13:16:44.434489Z","shell.execute_reply":"2023-05-03T13:16:53.338975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# unlabeled_example_df.describe()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:17:05.256878Z","iopub.execute_input":"2023-05-03T13:17:05.258024Z","iopub.status.idle":"2023-05-03T13:17:13.382977Z","shell.execute_reply.started":"2023-05-03T13:17:05.257948Z","shell.execute_reply":"2023-05-03T13:17:13.381469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# corr_mat = np.round(unlabeled_example_df.corr(), 3)\n\n# fig = px.imshow(\n#     corr_mat,\n#     x=corr_mat.columns,\n#     y=corr_mat.columns, \n#     text_auto=True\n#    )\n# fig.update_xaxes(side=\"bottom\")\n# fig.update_layout(\n#     title={\n#         'text': \"Correlation Matrix of unlabeled instance\",\n#         'y':0.95,\n#         'x':0.5,\n#         'xanchor': 'center',\n#         'yanchor': 'top'\n#     }\n# )\n\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:17:18.051634Z","iopub.execute_input":"2023-05-03T13:17:18.052054Z","iopub.status.idle":"2023-05-03T13:17:23.544424Z","shell.execute_reply.started":"2023-05-03T13:17:18.052015Z","shell.execute_reply":"2023-05-03T13:17:23.543000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Overview - tdcsfog metadata","metadata":{}},{"cell_type":"code","source":"# tdcsfog_metadata_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv\")\n# tdcsfog_metadata_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:19:38.197246Z","iopub.execute_input":"2023-05-03T13:19:38.197713Z","iopub.status.idle":"2023-05-03T13:19:38.221020Z","shell.execute_reply.started":"2023-05-03T13:19:38.197671Z","shell.execute_reply":"2023-05-03T13:19:38.219215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tdcsfog_metadata_df.describe()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:19:46.448164Z","iopub.execute_input":"2023-05-03T13:19:46.448546Z","iopub.status.idle":"2023-05-03T13:19:46.472318Z","shell.execute_reply.started":"2023-05-03T13:19:46.448515Z","shell.execute_reply":"2023-05-03T13:19:46.470764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# temp = len(tdcsfog_metadata_df)\n# print(\n#     f\"Length of the tdcsfog_metadata.csv file is: {color.BLUE}{temp}{color.END}\",\n# )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:19:58.729199Z","iopub.execute_input":"2023-05-03T13:19:58.729618Z","iopub.status.idle":"2023-05-03T13:19:58.735662Z","shell.execute_reply.started":"2023-05-03T13:19:58.729548Z","shell.execute_reply":"2023-05-03T13:19:58.734673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# temp = len(tdcsfog_metadata_df.Subject.unique())\n# print(\n#     f\"Number of unique subjects: {color.BLUE}{temp}{color.END}\",\n# )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:20:13.943929Z","iopub.execute_input":"2023-05-03T13:20:13.944329Z","iopub.status.idle":"2023-05-03T13:20:13.954275Z","shell.execute_reply.started":"2023-05-03T13:20:13.944288Z","shell.execute_reply":"2023-05-03T13:20:13.951799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# unique_subject_id = \"13abfd\"\n# tdcsfog_metadata_df[tdcsfog_metadata_df.Subject == unique_subject_id]","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:21:14.893976Z","iopub.execute_input":"2023-05-03T13:21:14.894350Z","iopub.status.idle":"2023-05-03T13:21:14.911887Z","shell.execute_reply.started":"2023-05-03T13:21:14.894321Z","shell.execute_reply":"2023-05-03T13:21:14.909820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tdcsfog_metadata_df.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:21:29.631881Z","iopub.execute_input":"2023-05-03T13:21:29.632287Z","iopub.status.idle":"2023-05-03T13:21:29.644276Z","shell.execute_reply.started":"2023-05-03T13:21:29.632246Z","shell.execute_reply":"2023-05-03T13:21:29.642675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tdcsfog_visit_counts = tdcsfog_metadata_df.Visit.value_counts()\n\n# fig = px.bar(x=tdcsfog_visit_counts.index, y=tdcsfog_visit_counts.values, color_discrete_sequence=['darkgreen'])\n# fig.update_layout(xaxis_title=\"Visit\", yaxis_title=\"Count\")\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:22:07.716644Z","iopub.execute_input":"2023-05-03T13:22:07.717148Z","iopub.status.idle":"2023-05-03T13:22:07.895575Z","shell.execute_reply.started":"2023-05-03T13:22:07.717096Z","shell.execute_reply":"2023-05-03T13:22:07.893464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tdcsfog_test_counts = tdcsfog_metadata_df.Test.value_counts()\n\n# fig = px.bar(x=tdcsfog_test_counts.index, y=tdcsfog_test_counts.values, color_discrete_sequence=['darkgreen'])\n# fig.update_layout(xaxis_title=\"Test\", yaxis_title=\"Count\")\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:22:44.273405Z","iopub.execute_input":"2023-05-03T13:22:44.273818Z","iopub.status.idle":"2023-05-03T13:22:44.343118Z","shell.execute_reply.started":"2023-05-03T13:22:44.273780Z","shell.execute_reply":"2023-05-03T13:22:44.340943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tdcsfog_medication_counts = tdcsfog_metadata_df.Medication.value_counts()\n\n# fig = px.bar(x=tdcsfog_medication_counts.index, y=tdcsfog_medication_counts.values, color_discrete_sequence=['darkgreen'])\n# fig.update_layout(xaxis_title=\"Medication\", yaxis_title=\"Count\")\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:23:06.366804Z","iopub.execute_input":"2023-05-03T13:23:06.367253Z","iopub.status.idle":"2023-05-03T13:23:06.437256Z","shell.execute_reply.started":"2023-05-03T13:23:06.367216Z","shell.execute_reply":"2023-05-03T13:23:06.436181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Overview - defog metadata","metadata":{}},{"cell_type":"code","source":"# defog_metadata_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv\")\n# defog_metadata_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:25:29.700454Z","iopub.execute_input":"2023-05-03T13:25:29.700918Z","iopub.status.idle":"2023-05-03T13:25:29.719999Z","shell.execute_reply.started":"2023-05-03T13:25:29.700881Z","shell.execute_reply":"2023-05-03T13:25:29.718412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defog_metadata_df.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:25:56.586938Z","iopub.execute_input":"2023-05-03T13:25:56.587339Z","iopub.status.idle":"2023-05-03T13:25:56.596651Z","shell.execute_reply.started":"2023-05-03T13:25:56.587299Z","shell.execute_reply":"2023-05-03T13:25:56.595575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# temp = len(defog_metadata_df)\n# print(\n#     f\"Length of the defog_metadata.csv file is: {color.BLUE}{temp}{color.END}\",\n# )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:26:03.986893Z","iopub.execute_input":"2023-05-03T13:26:03.987233Z","iopub.status.idle":"2023-05-03T13:26:03.991972Z","shell.execute_reply.started":"2023-05-03T13:26:03.987204Z","shell.execute_reply":"2023-05-03T13:26:03.991144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# temp = len(defog_metadata_df.Subject.unique())\n# print(\n#     f\"Number of unique subjects: {color.BLUE}{temp}{color.END}\",\n# )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:26:12.679248Z","iopub.execute_input":"2023-05-03T13:26:12.679668Z","iopub.status.idle":"2023-05-03T13:26:12.687908Z","shell.execute_reply.started":"2023-05-03T13:26:12.679627Z","shell.execute_reply":"2023-05-03T13:26:12.686401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# unique_subject_id = \"bf608b\"\n# defog_metadata_df[defog_metadata_df.Subject == unique_subject_id]","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:26:21.610530Z","iopub.execute_input":"2023-05-03T13:26:21.610960Z","iopub.status.idle":"2023-05-03T13:26:21.625421Z","shell.execute_reply.started":"2023-05-03T13:26:21.610922Z","shell.execute_reply":"2023-05-03T13:26:21.624075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defog_visit_counts = defog_metadata_df.Visit.value_counts()\n\n# fig = px.bar(x=defog_visit_counts.index, y=defog_visit_counts.values, color_discrete_sequence=['darkslateblue'])\n# fig.update_layout(xaxis_title=\"Visit\", yaxis_title=\"Count\")\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:26:29.786679Z","iopub.execute_input":"2023-05-03T13:26:29.787039Z","iopub.status.idle":"2023-05-03T13:26:29.852356Z","shell.execute_reply.started":"2023-05-03T13:26:29.787009Z","shell.execute_reply":"2023-05-03T13:26:29.851452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defog_medication_counts = defog_metadata_df.Medication.value_counts()\n\n# fig = px.bar(x=defog_medication_counts.index, y=defog_medication_counts.values, color_discrete_sequence=['darkslateblue'])\n# fig.update_layout(xaxis_title=\"Medication\", yaxis_title=\"Count\")\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:26:44.485124Z","iopub.execute_input":"2023-05-03T13:26:44.485542Z","iopub.status.idle":"2023-05-03T13:26:44.542801Z","shell.execute_reply.started":"2023-05-03T13:26:44.485506Z","shell.execute_reply":"2023-05-03T13:26:44.540808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Overview - daliy metadata","metadata":{}},{"cell_type":"code","source":"# daily_metadata_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/daily_metadata.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:27:10.566349Z","iopub.execute_input":"2023-05-03T13:27:10.566777Z","iopub.status.idle":"2023-05-03T13:27:10.581086Z","shell.execute_reply.started":"2023-05-03T13:27:10.566740Z","shell.execute_reply":"2023-05-03T13:27:10.579536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# daily_metadata_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:27:13.170737Z","iopub.execute_input":"2023-05-03T13:27:13.172438Z","iopub.status.idle":"2023-05-03T13:27:13.185087Z","shell.execute_reply.started":"2023-05-03T13:27:13.172378Z","shell.execute_reply":"2023-05-03T13:27:13.183318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# daily_metadata_df.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:27:20.212096Z","iopub.execute_input":"2023-05-03T13:27:20.212507Z","iopub.status.idle":"2023-05-03T13:27:20.223984Z","shell.execute_reply.started":"2023-05-03T13:27:20.212469Z","shell.execute_reply":"2023-05-03T13:27:20.221637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# temp = len(daily_metadata_df)\n# print(\n#     f\"Length of the daily_metadata.csv file is: {color.BLUE}{temp}{color.END}\",\n# )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:27:26.813905Z","iopub.execute_input":"2023-05-03T13:27:26.815897Z","iopub.status.idle":"2023-05-03T13:27:26.823653Z","shell.execute_reply.started":"2023-05-03T13:27:26.815827Z","shell.execute_reply":"2023-05-03T13:27:26.822578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# temp = len(daily_metadata_df.Subject.unique())\n# print(\n#     f\"Number of unique subjects: {color.BLUE}{temp}{color.END}\",\n# )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:27:34.893050Z","iopub.execute_input":"2023-05-03T13:27:34.893722Z","iopub.status.idle":"2023-05-03T13:27:34.905458Z","shell.execute_reply.started":"2023-05-03T13:27:34.893662Z","shell.execute_reply":"2023-05-03T13:27:34.903498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# daily_visit_counts = daily_metadata_df.Visit.value_counts()\n\n# fig = px.bar(x=daily_visit_counts.index, y=daily_visit_counts.values, color_discrete_sequence=['cornflowerblue'])\n# fig.update_layout(xaxis_title=\"Visit\", yaxis_title=\"Count\")\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:27:42.357297Z","iopub.execute_input":"2023-05-03T13:27:42.358651Z","iopub.status.idle":"2023-05-03T13:27:42.447756Z","shell.execute_reply.started":"2023-05-03T13:27:42.358570Z","shell.execute_reply":"2023-05-03T13:27:42.445197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# daily_bor_counts = daily_metadata_df[\"Beginning of recording [00:00-23:59]\"].value_counts()\n\n# fig = px.bar(x=daily_bor_counts.index, y=daily_bor_counts.values, color_discrete_sequence=['cornflowerblue'])\n# fig.update_layout(xaxis_title=\"Beginning of recording\", yaxis_title=\"Count\")\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:27:53.517033Z","iopub.execute_input":"2023-05-03T13:27:53.517469Z","iopub.status.idle":"2023-05-03T13:27:53.578101Z","shell.execute_reply.started":"2023-05-03T13:27:53.517430Z","shell.execute_reply":"2023-05-03T13:27:53.576723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Overview - subjects","metadata":{}},{"cell_type":"code","source":"# subjects_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:29:40.975635Z","iopub.execute_input":"2023-05-03T13:29:40.975988Z","iopub.status.idle":"2023-05-03T13:29:40.992387Z","shell.execute_reply.started":"2023-05-03T13:29:40.975957Z","shell.execute_reply":"2023-05-03T13:29:40.989851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subjects_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:29:52.408503Z","iopub.execute_input":"2023-05-03T13:29:52.409002Z","iopub.status.idle":"2023-05-03T13:29:52.427004Z","shell.execute_reply.started":"2023-05-03T13:29:52.408948Z","shell.execute_reply":"2023-05-03T13:29:52.424475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subjects_df.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:30:06.119163Z","iopub.execute_input":"2023-05-03T13:30:06.119579Z","iopub.status.idle":"2023-05-03T13:30:06.134523Z","shell.execute_reply.started":"2023-05-03T13:30:06.119538Z","shell.execute_reply":"2023-05-03T13:30:06.132847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# temp = len(subjects_df)\n# print(\n#     f\"Length of the subjects.csv file is: {color.BLUE}{temp}{color.END}\",\n# )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:30:13.214262Z","iopub.execute_input":"2023-05-03T13:30:13.214776Z","iopub.status.idle":"2023-05-03T13:30:13.223457Z","shell.execute_reply.started":"2023-05-03T13:30:13.214735Z","shell.execute_reply":"2023-05-03T13:30:13.221128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# temp = len(subjects_df.Subject.unique())\n# print(\n#     f\"Number of unique subjects: {color.BLUE}{temp}{color.END}\",\n# )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:30:19.712755Z","iopub.execute_input":"2023-05-03T13:30:19.713182Z","iopub.status.idle":"2023-05-03T13:30:19.720551Z","shell.execute_reply.started":"2023-05-03T13:30:19.713151Z","shell.execute_reply":"2023-05-03T13:30:19.719342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subjects_visit_counts = subjects_df.Visit.value_counts()\n\n# fig = px.bar(x=subjects_visit_counts.index, y=subjects_visit_counts.values, color_discrete_sequence=['goldenrod'])\n# fig.update_layout(xaxis_title=\"Visit\", yaxis_title=\"Count\")\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:30:26.865174Z","iopub.execute_input":"2023-05-03T13:30:26.865554Z","iopub.status.idle":"2023-05-03T13:30:26.931993Z","shell.execute_reply.started":"2023-05-03T13:30:26.865523Z","shell.execute_reply":"2023-05-03T13:30:26.929733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fig = px.histogram(subjects_df, x=\"Age\", nbins=30, color_discrete_sequence=['goldenrod'])\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:30:38.021895Z","iopub.execute_input":"2023-05-03T13:30:38.022353Z","iopub.status.idle":"2023-05-03T13:30:38.289435Z","shell.execute_reply.started":"2023-05-03T13:30:38.022308Z","shell.execute_reply":"2023-05-03T13:30:38.288350Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# subjects_sex_counts = subjects_df.Sex.value_counts()\n\n# fig = px.bar(x=subjects_sex_counts.index, y=subjects_sex_counts.values, color_discrete_sequence=['goldenrod'])\n# fig.update_layout(xaxis_title=\"Sex\", yaxis_title=\"Count\")\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:30:47.556784Z","iopub.execute_input":"2023-05-03T13:30:47.557357Z","iopub.status.idle":"2023-05-03T13:30:47.631420Z","shell.execute_reply.started":"2023-05-03T13:30:47.557325Z","shell.execute_reply":"2023-05-03T13:30:47.630436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fig = px.histogram(subjects_df, x=\"YearsSinceDx\", nbins=30, color_discrete_sequence=['goldenrod'])\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:30:55.951942Z","iopub.execute_input":"2023-05-03T13:30:55.952370Z","iopub.status.idle":"2023-05-03T13:30:56.025214Z","shell.execute_reply.started":"2023-05-03T13:30:55.952331Z","shell.execute_reply":"2023-05-03T13:30:56.023310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fig = px.histogram(subjects_df, x=\"UPDRSIII_On\", nbins=30, color_discrete_sequence=['goldenrod'])\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:31:02.935887Z","iopub.execute_input":"2023-05-03T13:31:02.936287Z","iopub.status.idle":"2023-05-03T13:31:03.005063Z","shell.execute_reply.started":"2023-05-03T13:31:02.936255Z","shell.execute_reply":"2023-05-03T13:31:03.003975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fig = px.histogram(subjects_df, x=\"NFOGQ\", nbins=20, color_discrete_sequence=['goldenrod'])\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:31:10.677293Z","iopub.execute_input":"2023-05-03T13:31:10.677881Z","iopub.status.idle":"2023-05-03T13:31:10.746231Z","shell.execute_reply.started":"2023-05-03T13:31:10.677824Z","shell.execute_reply":"2023-05-03T13:31:10.744860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# corr_mat = np.round(subjects_df.corr(), 3)\n\n# fig = px.imshow(\n#     corr_mat,\n#     x=corr_mat.columns,\n#     y=corr_mat.columns, \n#     text_auto=True\n#    )\n# fig.update_xaxes(side=\"bottom\")\n# fig.update_layout(\n#     title={\n#         'text': \"Correlation Matrix of sybjects features\",\n#         'y':0.95,\n#         'x':0.5,\n#         'xanchor': 'center',\n#         'yanchor': 'top'\n#     }\n# )\n\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:31:19.610087Z","iopub.execute_input":"2023-05-03T13:31:19.610519Z","iopub.status.idle":"2023-05-03T13:31:19.687738Z","shell.execute_reply.started":"2023-05-03T13:31:19.610474Z","shell.execute_reply":"2023-05-03T13:31:19.684560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Overview - Events","metadata":{}},{"cell_type":"code","source":"# events_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/events.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:32:23.056328Z","iopub.execute_input":"2023-05-03T13:32:23.056793Z","iopub.status.idle":"2023-05-03T13:32:23.079366Z","shell.execute_reply.started":"2023-05-03T13:32:23.056759Z","shell.execute_reply":"2023-05-03T13:32:23.077854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# events_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:32:28.556085Z","iopub.execute_input":"2023-05-03T13:32:28.556552Z","iopub.status.idle":"2023-05-03T13:32:28.572359Z","shell.execute_reply.started":"2023-05-03T13:32:28.556512Z","shell.execute_reply":"2023-05-03T13:32:28.570946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# temp = len(events_df)\n# print(\n#     f\"Length of the events.csv file is: {color.BLUE}{temp}{color.END}\",\n# )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:32:51.338254Z","iopub.execute_input":"2023-05-03T13:32:51.338657Z","iopub.status.idle":"2023-05-03T13:32:51.346065Z","shell.execute_reply.started":"2023-05-03T13:32:51.338626Z","shell.execute_reply":"2023-05-03T13:32:51.344324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# events_df.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:32:57.413517Z","iopub.execute_input":"2023-05-03T13:32:57.413965Z","iopub.status.idle":"2023-05-03T13:32:57.426290Z","shell.execute_reply.started":"2023-05-03T13:32:57.413925Z","shell.execute_reply":"2023-05-03T13:32:57.424734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fig = px.histogram(events_df, x=\"Init\", nbins=50)\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:33:05.399757Z","iopub.execute_input":"2023-05-03T13:33:05.400177Z","iopub.status.idle":"2023-05-03T13:33:05.464067Z","shell.execute_reply.started":"2023-05-03T13:33:05.400138Z","shell.execute_reply":"2023-05-03T13:33:05.462694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# event_time = events_df.Completion - events_df.Init\n# fig = px.histogram(x=event_time, nbins=30)\n# fig.update_layout(\n#     xaxis_title=\"Event duration\",\n#     title={\n#         'text': \"Distribution of event durations\",\n#         'y':0.95,\n#         'x':0.5,\n#         'xanchor': 'center',\n#         'yanchor': 'top'\n#     }\n# )\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:33:13.792439Z","iopub.execute_input":"2023-05-03T13:33:13.793155Z","iopub.status.idle":"2023-05-03T13:33:13.876947Z","shell.execute_reply.started":"2023-05-03T13:33:13.793104Z","shell.execute_reply":"2023-05-03T13:33:13.875542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# events_type_counts = events_df.Type.value_counts()\n\n# fig = px.bar(x=events_type_counts.index, y=events_type_counts.values)\n# fig.update_layout(xaxis_title=\"Type\", yaxis_title=\"Count\")\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:33:25.755844Z","iopub.execute_input":"2023-05-03T13:33:25.756216Z","iopub.status.idle":"2023-05-03T13:33:25.825697Z","shell.execute_reply.started":"2023-05-03T13:33:25.756186Z","shell.execute_reply":"2023-05-03T13:33:25.823749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# events_kinetic_counts = events_df.Kinetic.value_counts()\n\n# fig = px.bar(x=events_kinetic_counts.index, y=events_kinetic_counts.values)\n# fig.update_layout(xaxis_title=\"Kinetic\", yaxis_title=\"Count\")\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:33:36.007307Z","iopub.execute_input":"2023-05-03T13:33:36.007882Z","iopub.status.idle":"2023-05-03T13:33:36.089372Z","shell.execute_reply.started":"2023-05-03T13:33:36.007838Z","shell.execute_reply":"2023-05-03T13:33:36.086158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# corr_mat = np.round(events_df.corr(), 3)\n\n# fig = px.imshow(\n#     corr_mat,\n#     x=corr_mat.columns,\n#     y=corr_mat.columns, \n#     text_auto=True\n#    )\n# fig.update_xaxes(side=\"bottom\")\n# fig.update_layout(\n#     title={\n#         'text': \"Correlation Matrix of events features\",\n#         'y':0.95,\n#         'x':0.5,\n#         'xanchor': 'center',\n#         'yanchor': 'top'\n#     }\n# )\n\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:33:47.212535Z","iopub.execute_input":"2023-05-03T13:33:47.214527Z","iopub.status.idle":"2023-05-03T13:33:47.281260Z","shell.execute_reply.started":"2023-05-03T13:33:47.214462Z","shell.execute_reply":"2023-05-03T13:33:47.280371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Overview - Tasks","metadata":{}},{"cell_type":"code","source":"# tasks_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tasks.csv\")\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:34:35.413462Z","iopub.execute_input":"2023-05-03T13:34:35.413929Z","iopub.status.idle":"2023-05-03T13:34:35.433394Z","shell.execute_reply.started":"2023-05-03T13:34:35.413887Z","shell.execute_reply":"2023-05-03T13:34:35.431647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tasks_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:34:40.811686Z","iopub.execute_input":"2023-05-03T13:34:40.812076Z","iopub.status.idle":"2023-05-03T13:34:40.824435Z","shell.execute_reply.started":"2023-05-03T13:34:40.812045Z","shell.execute_reply":"2023-05-03T13:34:40.822713Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# temp = len(tasks_df)\n# print(\n#     f\"Length of the subjects.csv file is: {color.BLUE}{temp}{color.END}\",\n# )","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:34:47.972315Z","iopub.execute_input":"2023-05-03T13:34:47.972867Z","iopub.status.idle":"2023-05-03T13:34:47.979086Z","shell.execute_reply.started":"2023-05-03T13:34:47.972823Z","shell.execute_reply":"2023-05-03T13:34:47.977932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tasks_df.isnull().sum()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:34:54.645945Z","iopub.execute_input":"2023-05-03T13:34:54.646342Z","iopub.status.idle":"2023-05-03T13:34:54.655825Z","shell.execute_reply.started":"2023-05-03T13:34:54.646306Z","shell.execute_reply":"2023-05-03T13:34:54.654322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tasks_df[\"Duration\"] = tasks_df.End - tasks_df.Begin\n# fig = px.histogram(x=tasks_df[\"Duration\"], nbins=30, color_discrete_sequence=['lightslategrey'])\n# fig.update_layout(\n#     xaxis_title=\"Task duration\",\n#     title={\n#         'text': \"Distribution of task durations\",\n#         'y':0.95,\n#         'x':0.5,\n#         'xanchor': 'center',\n#         'yanchor': 'top'\n#     }\n# )\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:35:00.924192Z","iopub.execute_input":"2023-05-03T13:35:00.924755Z","iopub.status.idle":"2023-05-03T13:35:01.013672Z","shell.execute_reply.started":"2023-05-03T13:35:00.924714Z","shell.execute_reply":"2023-05-03T13:35:01.012445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tasks_task_counts = tasks_df.Task.value_counts()\n\n# fig = px.bar(x=tasks_task_counts.index, y=tasks_task_counts.values, color_discrete_sequence=['lightslategrey'])\n# fig.update_layout(xaxis_title=\"Task\", yaxis_title=\"Count\")\n# fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:35:12.717659Z","iopub.execute_input":"2023-05-03T13:35:12.718108Z","iopub.status.idle":"2023-05-03T13:35:12.782160Z","shell.execute_reply.started":"2023-05-03T13:35:12.718069Z","shell.execute_reply":"2023-05-03T13:35:12.780836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fig = px.box(\n#     tasks_df, \n#     x='Duration', y='Task',\n#     orientation='h', height=1000, \n#     color_discrete_sequence=['lightslategrey']\n# )\n\n# fig.update_layout(\n#     title={\n#         'text': \"Task Duration Boxplot\",\n#         'y':0.98,\n#         'x':0.5,\n#         'xanchor': 'center',\n#         'yanchor': 'top'\n#     },\n#     margin=dict(l=50, r=0, t=50, b=50)\n# )\n# fig.show(autosize=True)","metadata":{"execution":{"iopub.status.busy":"2023-05-03T13:35:25.819183Z","iopub.execute_input":"2023-05-03T13:35:25.819593Z","iopub.status.idle":"2023-05-03T13:35:25.945567Z","shell.execute_reply.started":"2023-05-03T13:35:25.819555Z","shell.execute_reply":"2023-05-03T13:35:25.944183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Split train, validiation + Navie baseline model - LGBM","metadata":{}},{"cell_type":"code","source":"# import library\nimport os\nimport random\nimport cv2\nimport pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:40:28.676231Z","iopub.execute_input":"2023-05-06T14:40:28.676921Z","iopub.status.idle":"2023-05-06T14:40:28.941009Z","shell.execute_reply.started":"2023-05-06T14:40:28.676882Z","shell.execute_reply":"2023-05-06T14:40:28.939514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prepare data","metadata":{}},{"cell_type":"code","source":"# Reduce Memory Usage\n# reference : https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65 @ARJANGROEN\n\ndef reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:40:28.943920Z","iopub.execute_input":"2023-05-06T14:40:28.944810Z","iopub.status.idle":"2023-05-06T14:40:28.960887Z","shell.execute_reply.started":"2023-05-06T14:40:28.944748Z","shell.execute_reply":"2023-05-06T14:40:28.959285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#reference: https://www.kaggle.com/code/ghrangel/read-data-and-merge\n\n# DATA_ROOT_DEFOG = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/'\n# defog = pd.DataFrame()\n# for root, dirs, files in os.walk(DATA_ROOT_DEFOG):\n#     for name in files:       \n#         f = os.path.join(root, name)\n#         df_list= pd.read_csv(f)\n#         words = name.split('.')[0]\n#         df_list['file']= name.split('.')[0]\n#         defog = pd.concat([defog, df_list], axis=0)\n\n# defog","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:40:28.962087Z","iopub.execute_input":"2023-05-06T14:40:28.962448Z","iopub.status.idle":"2023-05-06T14:41:23.946509Z","shell.execute_reply.started":"2023-05-06T14:40:28.962416Z","shell.execute_reply":"2023-05-06T14:41:23.945134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defog = reduce_memory_usage(defog)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:41:23.949644Z","iopub.execute_input":"2023-05-06T14:41:23.950035Z","iopub.status.idle":"2023-05-06T14:41:26.351846Z","shell.execute_reply.started":"2023-05-06T14:41:23.949997Z","shell.execute_reply":"2023-05-06T14:41:26.350540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defog = defog[(defog['Task']==1)&(defog['Valid']==1)]\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:41:26.353760Z","iopub.execute_input":"2023-05-06T14:41:26.354245Z","iopub.status.idle":"2023-05-06T14:41:26.845843Z","shell.execute_reply.started":"2023-05-06T14:41:26.354196Z","shell.execute_reply":"2023-05-06T14:41:26.843959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print('the shape of defog dataset is {}'.format(defog.shape))\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:41:26.847986Z","iopub.execute_input":"2023-05-06T14:41:26.848433Z","iopub.status.idle":"2023-05-06T14:41:26.857575Z","shell.execute_reply.started":"2023-05-06T14:41:26.848392Z","shell.execute_reply":"2023-05-06T14:41:26.856160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defog_metadata = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv\")\n# defog_metadata","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:41:26.859076Z","iopub.execute_input":"2023-05-06T14:41:26.859468Z","iopub.status.idle":"2023-05-06T14:41:26.892545Z","shell.execute_reply.started":"2023-05-06T14:41:26.859432Z","shell.execute_reply":"2023-05-06T14:41:26.890582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defog_m= defog_metadata.merge(defog, how = 'inner', left_on = 'Id', right_on = 'file')\n# defog_m.drop(['file','Valid','Task'], axis = 1, inplace = True)\n# defog_m","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:41:26.897081Z","iopub.execute_input":"2023-05-06T14:41:26.897570Z","iopub.status.idle":"2023-05-06T14:41:28.895946Z","shell.execute_reply.started":"2023-05-06T14:41:26.897524Z","shell.execute_reply":"2023-05-06T14:41:28.894724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# summary table function\n# def summary(df):\n#     print(f'data shape: {df.shape}')\n#     summ = pd.DataFrame(df.dtypes, columns=['data type'])\n#     summ['#missing'] = df.isnull().sum().values * 100\n#     summ['%missing'] = df.isnull().sum().values / len(df)\n#     summ['#unique'] = df.nunique().values\n#     desc = pd.DataFrame(df.describe(include='all').transpose())\n#     summ['min'] = desc['min'].values\n#     summ['max'] = desc['max'].values\n#     summ['first value'] = df.loc[0].values\n#     summ['second value'] = df.loc[1].values\n#     summ['third value'] = df.loc[2].values\n    \n#     return summ","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:41:28.897470Z","iopub.execute_input":"2023-05-06T14:41:28.897899Z","iopub.status.idle":"2023-05-06T14:41:28.906798Z","shell.execute_reply.started":"2023-05-06T14:41:28.897816Z","shell.execute_reply":"2023-05-06T14:41:28.905501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# summary(defog_m)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:41:29.539952Z","iopub.execute_input":"2023-05-06T14:41:29.541135Z","iopub.status.idle":"2023-05-06T14:41:34.712761Z","shell.execute_reply.started":"2023-05-06T14:41:29.541089Z","shell.execute_reply":"2023-05-06T14:41:34.711451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# garbage collection for memory\n# import gc\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:41:40.511898Z","iopub.execute_input":"2023-05-06T14:41:40.513555Z","iopub.status.idle":"2023-05-06T14:41:40.625721Z","shell.execute_reply.started":"2023-05-06T14:41:40.513453Z","shell.execute_reply":"2023-05-06T14:41:40.623590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DATA_ROOT_TDCSFOG = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/'\n# tdcsfog = pd.DataFrame()\n# for root, dirs, files in os.walk(DATA_ROOT_TDCSFOG):\n#     for name in files:       \n#         f = os.path.join(root, name)\n#         df_list= pd.read_csv(f)\n#         words = name.split('.')[0]\n#         df_list['file']= name.split('.')[0]\n#         tdcsfog = pd.concat([tdcsfog, df_list], axis=0)\n# tdcsfog","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:41:45.677779Z","iopub.execute_input":"2023-05-06T14:41:45.678262Z","iopub.status.idle":"2023-05-06T14:44:05.178953Z","shell.execute_reply.started":"2023-05-06T14:41:45.678215Z","shell.execute_reply":"2023-05-06T14:44:05.177567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tdcsfog = reduce_memory_usage(tdcsfog)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:44:05.180684Z","iopub.execute_input":"2023-05-06T14:44:05.181009Z","iopub.status.idle":"2023-05-06T14:44:06.356928Z","shell.execute_reply.started":"2023-05-06T14:44:05.180979Z","shell.execute_reply":"2023-05-06T14:44:06.355559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tdcsfog_metadata = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv\")\n# tdcsfog_metadata","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:44:06.358257Z","iopub.execute_input":"2023-05-06T14:44:06.358603Z","iopub.status.idle":"2023-05-06T14:44:06.381256Z","shell.execute_reply.started":"2023-05-06T14:44:06.358571Z","shell.execute_reply":"2023-05-06T14:44:06.379907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tdcsfog_m = tdcsfog_metadata.merge(tdcsfog, how = 'inner', left_on = 'Id', right_on = 'file')\n# tdcsfog_m.drop(['file'], axis = 1, inplace = True)\n# tdcsfog_m","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:44:06.383552Z","iopub.execute_input":"2023-05-06T14:44:06.383902Z","iopub.status.idle":"2023-05-06T14:44:09.928540Z","shell.execute_reply.started":"2023-05-06T14:44:06.383870Z","shell.execute_reply":"2023-05-06T14:44:09.927245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_before_split_to_valid = pd.concat([defog_m, tdcsfog_m])","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:44:09.930095Z","iopub.execute_input":"2023-05-06T14:44:09.931225Z","iopub.status.idle":"2023-05-06T14:44:11.197584Z","shell.execute_reply.started":"2023-05-06T14:44:09.931154Z","shell.execute_reply":"2023-05-06T14:44:11.196150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_before_split_to_valid = train_before_split_to_valid.drop('Test', axis=1)\n# train_before_split_to_valid","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:44:11.199155Z","iopub.execute_input":"2023-05-06T14:44:11.199522Z","iopub.status.idle":"2023-05-06T14:44:13.296620Z","shell.execute_reply.started":"2023-05-06T14:44:11.199490Z","shell.execute_reply":"2023-05-06T14:44:13.295614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# garbage collection for memory\n# import gc\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:44:13.297896Z","iopub.execute_input":"2023-05-06T14:44:13.299038Z","iopub.status.idle":"2023-05-06T14:44:13.411708Z","shell.execute_reply.started":"2023-05-06T14:44:13.298985Z","shell.execute_reply":"2023-05-06T14:44:13.410134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.model_selection import KFold, StratifiedKFold, train_test_split, GridSearchCV\n# import warnings\n# warnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:44:13.413491Z","iopub.execute_input":"2023-05-06T14:44:13.414706Z","iopub.status.idle":"2023-05-06T14:44:13.840497Z","shell.execute_reply.started":"2023-05-06T14:44:13.414588Z","shell.execute_reply":"2023-05-06T14:44:13.839043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# conditions = [\n#     (train_before_split_to_valid['StartHesitation'] == 1),\n#     (train_before_split_to_valid['Turn'] == 1),\n#     (train_before_split_to_valid['Walking'] == 1)]\n# choices = ['StartHesitation', 'Turn', 'Walking']\n# train_before_split_to_valid['event'] = np.select(conditions, choices, default='Normal')","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:44:13.842161Z","iopub.execute_input":"2023-05-06T14:44:13.843307Z","iopub.status.idle":"2023-05-06T14:44:16.610228Z","shell.execute_reply.started":"2023-05-06T14:44:13.843255Z","shell.execute_reply":"2023-05-06T14:44:16.609214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_before_split_to_valid['event'].value_counts().to_frame().style.background_gradient()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:44:16.613313Z","iopub.execute_input":"2023-05-06T14:44:16.614087Z","iopub.status.idle":"2023-05-06T14:44:17.681313Z","shell.execute_reply.started":"2023-05-06T14:44:16.614045Z","shell.execute_reply":"2023-05-06T14:44:17.680132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df = train_before_split_to_valid[['AccV','AccML','AccAP','event']]\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:44:17.683065Z","iopub.execute_input":"2023-05-06T14:44:17.684136Z","iopub.status.idle":"2023-05-06T14:44:19.726419Z","shell.execute_reply.started":"2023-05-06T14:44:17.684084Z","shell.execute_reply":"2023-05-06T14:44:19.725227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.preprocessing import LabelEncoder\n# le = LabelEncoder()\n\n# train_df['target'] = le.fit_transform(train_df['event'])","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:44:19.727680Z","iopub.execute_input":"2023-05-06T14:44:19.728047Z","iopub.status.idle":"2023-05-06T14:44:22.472442Z","shell.execute_reply.started":"2023-05-06T14:44:19.728012Z","shell.execute_reply":"2023-05-06T14:44:22.471071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df","metadata":{"execution":{"iopub.status.busy":"2023-05-05T16:27:03.889186Z","iopub.execute_input":"2023-05-05T16:27:03.889943Z","iopub.status.idle":"2023-05-05T16:27:03.906848Z","shell.execute_reply.started":"2023-05-05T16:27:03.889889Z","shell.execute_reply":"2023-05-05T16:27:03.905471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Split to train and validiation**","metadata":{}},{"cell_type":"code","source":"# train_set, val_set = train_test_split(train_df, test_size=0.2, stratify=train_df['target'], random_state=42)\n\n# # check the distribution of target values in the training set\n# train_dist = train_set['target'].value_counts(normalize=True)\n# print(\"Training set target distribution:\\n\", train_dist)\n\n# # check the distribution of target values in the validation set\n# val_dist = val_set['target'].value_counts(normalize=True)\n# print(\"Validation set target distribution:\\n\", val_dist)\n    \n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:44:22.474000Z","iopub.execute_input":"2023-05-06T14:44:22.474747Z","iopub.status.idle":"2023-05-06T14:44:29.218173Z","shell.execute_reply.started":"2023-05-06T14:44:22.474707Z","shell.execute_reply":"2023-05-06T14:44:29.216860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# val_set.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-05T16:27:21.392217Z","iopub.execute_input":"2023-05-05T16:27:21.392636Z","iopub.status.idle":"2023-05-05T16:27:21.400239Z","shell.execute_reply.started":"2023-05-05T16:27:21.392601Z","shell.execute_reply":"2023-05-05T16:27:21.399175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train = train_set.drop(['event','target'], axis=1)\n# y_train = train_set['target']\n# X_val = val_set.drop(['event','target'], axis=1)\n# y_val = val_set['target']","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:44:29.219494Z","iopub.execute_input":"2023-05-06T14:44:29.219830Z","iopub.status.idle":"2023-05-06T14:44:29.335422Z","shell.execute_reply.started":"2023-05-06T14:44:29.219792Z","shell.execute_reply":"2023-05-06T14:44:29.334339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import warnings\n# import lightgbm as lgb\n# warnings.filterwarnings('ignore')\n\n\n\n# #setting up the parameters\n# params={}\n# params['learning_rate']=0.03\n# params['boosting_type']='gbdt' #GradientBoostingDecisionTree\n# params['objective']='multiclass' #Multi-class target feature\n# params['metric']='multi_logloss' #metric for multi-class\n# params['max_depth']=8\n# params['num_class']=4 #no.of unique values in the target class not inclusive of the end value\n# params['verbose']=-1\n\n# lgb_train = lgb.Dataset(X_train, label=y_train)\n# lgb_valid = lgb.Dataset(X_val, label=y_val)\n\n# #training the model\n# model_lgb = lgb.train(\n#     params = params,\n#     train_set = lgb_train,\n#     num_boost_round = 150,\n#     valid_sets = [lgb_valid, lgb_train],\n#     early_stopping_rounds = 25,\n#     verbose_eval = 25,\n#     )\n# print(\"Finish train\")\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:44:29.336699Z","iopub.execute_input":"2023-05-06T14:44:29.337027Z","iopub.status.idle":"2023-05-06T14:50:11.400664Z","shell.execute_reply.started":"2023-05-06T14:44:29.336997Z","shell.execute_reply":"2023-05-06T14:50:11.399253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred = model_lgb.predict(X_val)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:50:11.402495Z","iopub.execute_input":"2023-05-06T14:50:11.403525Z","iopub.status.idle":"2023-05-06T14:50:43.509296Z","shell.execute_reply.started":"2023-05-06T14:50:11.403471Z","shell.execute_reply":"2023-05-06T14:50:43.507973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-05T17:07:39.084553Z","iopub.execute_input":"2023-05-05T17:07:39.085098Z","iopub.status.idle":"2023-05-05T17:07:39.093780Z","shell.execute_reply.started":"2023-05-05T17:07:39.084955Z","shell.execute_reply":"2023-05-05T17:07:39.091623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# type(y_pred)","metadata":{"execution":{"iopub.status.busy":"2023-05-05T17:21:08.971670Z","iopub.execute_input":"2023-05-05T17:21:08.972159Z","iopub.status.idle":"2023-05-05T17:21:08.980368Z","shell.execute_reply.started":"2023-05-05T17:21:08.972118Z","shell.execute_reply":"2023-05-05T17:21:08.978835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# np.argmax(y_pred[0])","metadata":{"execution":{"iopub.status.busy":"2023-05-05T17:08:24.917707Z","iopub.execute_input":"2023-05-05T17:08:24.918208Z","iopub.status.idle":"2023-05-05T17:08:24.927210Z","shell.execute_reply.started":"2023-05-05T17:08:24.918161Z","shell.execute_reply":"2023-05-05T17:08:24.925890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import accuracy_score\n# y_pred_class = [np.argmax(line) for line in y_pred]\n# accuracy = accuracy_score(y_val, y_pred_class)\n# print(\"Validation accuracy:\", accuracy)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:50:43.510756Z","iopub.execute_input":"2023-05-06T14:50:43.511254Z","iopub.status.idle":"2023-05-06T14:50:48.135811Z","shell.execute_reply.started":"2023-05-06T14:50:43.511204Z","shell.execute_reply":"2023-05-06T14:50:48.134381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np\n# from sklearn.metrics import precision_recall_curve\n# from sklearn.metrics import average_precision_score\n\n# def calculate_ap_for_class(y_true, y_pred, class_label):\n#     \"\"\"\n#     Calculate the average precision score for a given class.\n\n#     Parameters:\n#     - y_true (array-like): True labels.\n#     - y_pred (array-like): Predicted labels.\n#     - class_label (int): Class label for which to calculate the AP score.\n\n#     Returns:\n#     - ap_score (float): Average precision score for the given class.\n#     \"\"\"\n\n#     # Convert y_true to a numpy array\n#     y_true = np.array(y_true)\n\n#     # Get the indices of samples that belong to the given class\n#     class_indices = np.where(y_true == class_label)[0]\n\n#     # Create a new array with true labels where 1 indicates samples that belong to the given class\n#     y_true_class = np.zeros_like(y_true)\n#     y_true_class[class_indices] = 1\n\n#     # Extract the predicted probabilities for the given class\n#     y_pred_class = y_pred[:, class_label]\n\n#     # Calculate precision and recall values at different thresholds\n#     ap_score = average_precision_score(y_true_class, y_pred_class)\n\n\n#     return ap_score\n\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:51:56.672274Z","iopub.execute_input":"2023-05-06T14:51:56.672692Z","iopub.status.idle":"2023-05-06T14:51:56.682579Z","shell.execute_reply.started":"2023-05-06T14:51:56.672661Z","shell.execute_reply":"2023-05-06T14:51:56.681035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ap_scores = []\n# ap_scores.append(calculate_ap_for_class(y_val, y_pred, 0))\n# ap_scores.append(calculate_ap_for_class(y_val, y_pred, 1))\n# ap_scores.append(calculate_ap_for_class(y_val, y_pred, 2))\n# ap_scores.append(calculate_ap_for_class(y_val, y_pred, 3))\n# print(\"MAP score:\", np.mean(ap_scores))\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:52:01.161010Z","iopub.execute_input":"2023-05-06T14:52:01.162300Z","iopub.status.idle":"2023-05-06T14:52:04.405598Z","shell.execute_reply.started":"2023-05-06T14:52:01.162254Z","shell.execute_reply":"2023-05-06T14:52:04.404284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(ap_scores)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:52:15.072497Z","iopub.execute_input":"2023-05-06T14:52:15.072987Z","iopub.status.idle":"2023-05-06T14:52:15.080651Z","shell.execute_reply.started":"2023-05-06T14:52:15.072947Z","shell.execute_reply":"2023-05-06T14:52:15.078808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pickle\n# # Save the model as a pickle file\n# output_file = \"/kaggle/working/lgb.pkl\"\n# with open(output_file, 'wb') as f:\n#     pickle.dump(model_lgb, f)\n\n# print(\"Model saved as\", output_file)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:52:22.047484Z","iopub.execute_input":"2023-05-06T14:52:22.047962Z","iopub.status.idle":"2023-05-06T14:52:22.128648Z","shell.execute_reply.started":"2023-05-06T14:52:22.047923Z","shell.execute_reply":"2023-05-06T14:52:22.127123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Feature extractor ","metadata":{}},{"cell_type":"code","source":"# import gc\n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T08:18:00.663210Z","iopub.execute_input":"2023-05-06T08:18:00.664329Z","iopub.status.idle":"2023-05-06T08:18:00.811424Z","shell.execute_reply.started":"2023-05-06T08:18:00.664271Z","shell.execute_reply":"2023-05-06T08:18:00.809892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip download tsflex -d ./tsflex/\n# !pip download seglearn -d ./seglearn/","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:17:21.148533Z","iopub.execute_input":"2023-05-06T13:17:21.149467Z","iopub.status.idle":"2023-05-06T13:17:32.346249Z","shell.execute_reply.started":"2023-05-06T13:17:21.149427Z","shell.execute_reply":"2023-05-06T13:17:32.344342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import os\n# from zipfile import ZipFile\n\n# dirName = \"./\"\n# zipName = \"packages.zip\"\n\n# # Create a ZipFile Object\n# with ZipFile(zipName, 'w') as zipObj:\n#     # Iterate over all the files in directory\n#     for folderName, subfolders, filenames in os.walk(dirName):\n#         for filename in filenames:\n#             if (filename != zipName):\n#                 # create complete filepath of file in directory\n#                 filePath = os.path.join(folderName, filename)\n#                 # Add file to zip\n#                 zipObj.write(filePath)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:23:29.322119Z","iopub.execute_input":"2023-05-06T13:23:29.322717Z","iopub.status.idle":"2023-05-06T13:23:29.741194Z","shell.execute_reply.started":"2023-05-06T13:23:29.322668Z","shell.execute_reply":"2023-05-06T13:23:29.739507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ! ls ../input/imp-packages/tsflex\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:42:40.223100Z","iopub.execute_input":"2023-05-06T13:42:40.223586Z","iopub.status.idle":"2023-05-06T13:42:41.307027Z","shell.execute_reply.started":"2023-05-06T13:42:40.223526Z","shell.execute_reply":"2023-05-06T13:42:41.305517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tsflex --no-index --find-links=file:///kaggle/input/imp-packages/tsflex\n!pip install seglearn --no-index --find-links=file:///kaggle/input/imp-packages/seglearn","metadata":{"execution":{"iopub.status.busy":"2023-05-07T07:49:11.626775Z","iopub.execute_input":"2023-05-07T07:49:11.627279Z","iopub.status.idle":"2023-05-07T07:49:40.407741Z","shell.execute_reply.started":"2023-05-07T07:49:11.627233Z","shell.execute_reply":"2023-05-07T07:49:40.406506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os \nimport gc\nfrom sklearn.cluster import KMeans\nfrom sklearn.model_selection import StratifiedKFold\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\nfrom sklearn.metrics import roc_auc_score,log_loss\nimport warnings \nfrom sklearn.metrics import precision_score\nimport pickle\nfrom tqdm.auto import tqdm\nimport sys\n%matplotlib inline\nwarnings.filterwarnings(\"ignore\")\nsys.path.append('/kaggle/working/mysitepackages')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T07:49:44.370287Z","iopub.execute_input":"2023-05-07T07:49:44.370834Z","iopub.status.idle":"2023-05-07T07:49:47.449293Z","shell.execute_reply.started":"2023-05-07T07:49:44.370774Z","shell.execute_reply":"2023-05-07T07:49:47.447010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reduce Memory Usage\n# reference : https://www.kaggle.com/code/arjanso/reducing-dataframe-memory-size-by-65 @ARJANGROEN\n\ndef reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-07T07:49:48.820104Z","iopub.execute_input":"2023-05-07T07:49:48.820573Z","iopub.status.idle":"2023-05-07T07:49:48.837383Z","shell.execute_reply.started":"2023-05-07T07:49:48.820533Z","shell.execute_reply":"2023-05-07T07:49:48.835403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading metadata files\ntdcsfog_metadata = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\ndefog_metadata = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\n# daily_metadata=pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/daily_metadata.csv')\ntdcsfog_metadata['Module'] ='tdcsfog'\ndefog_metadata['Module'] = 'defog'\n# daily_metadata['Module']='daily'\nmetadata = pd.concat([tdcsfog_metadata,defog_metadata])\nmetadata.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T07:49:56.373503Z","iopub.execute_input":"2023-05-07T07:49:56.374190Z","iopub.status.idle":"2023-05-07T07:49:56.443489Z","shell.execute_reply.started":"2023-05-07T07:49:56.374134Z","shell.execute_reply":"2023-05-07T07:49:56.442101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# metadata","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:47:47.458917Z","iopub.execute_input":"2023-05-06T13:47:47.459528Z","iopub.status.idle":"2023-05-06T13:47:47.488744Z","shell.execute_reply.started":"2023-05-06T13:47:47.459474Z","shell.execute_reply":"2023-05-06T13:47:47.487166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/code/jazivxt/familiar-solvs\nsubjects = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv')\ntasks = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tasks.csv')\ntasks['Duration'] = tasks['End'] - tasks['Begin']\ntasks = pd.pivot_table(tasks, values=['Duration'], index=['Id'], columns=['Task'], aggfunc='sum', fill_value=0)\ntasks.columns = [c[-1] for c in tasks.columns]\ntasks = tasks.reset_index()\ntasks['t_kmeans'] = KMeans(n_clusters=10, random_state=3).fit_predict(tasks[tasks.columns[1:]])\n\nsubjects = subjects.fillna(0).groupby('Subject').median()\nsubjects = subjects.reset_index()\nsubjects['s_kmeans'] = KMeans(n_clusters=10, random_state=3).fit_predict(subjects[subjects.columns[1:]])\nsubjects=subjects.rename(columns={'Visit':'s_Visit','Age':'s_Age','YearsSinceDx':'s_YearsSinceDx','UPDRSIII_On':'s_UPDRSIII_On','UPDRSIII_Off':'s_UPDRSIII_Off','NFOGQ':'s_NFOGQ'})","metadata":{"execution":{"iopub.status.busy":"2023-05-07T07:50:01.601038Z","iopub.execute_input":"2023-05-07T07:50:01.601485Z","iopub.status.idle":"2023-05-07T07:50:02.383596Z","shell.execute_reply.started":"2023-05-07T07:50:01.601451Z","shell.execute_reply":"2023-05-07T07:50:02.381785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#REF:\n# https://www.kaggle.com/code/xzj19013742/groupkfold-cross-validation-tsflex\n# https://www.kaggle.com/code/jeroenvdd/time-series-tsflex\n\nfrom seglearn.feature_functions import base_features, emg_features\n\nfrom tsflex.features import FeatureCollection, MultipleFeatureDescriptors\nfrom tsflex.features.integrations import seglearn_feature_dict_wrapper\n\n\nbasic_feats = MultipleFeatureDescriptors(\n    functions=seglearn_feature_dict_wrapper(base_features()),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=[5_000],\n    strides=[5_000],\n)\n\nemg_feats = emg_features()\ndel emg_feats['simple square integral'] # is same as abs_energy (which is in base_features)\n\nemg_feats = MultipleFeatureDescriptors(\n    functions=seglearn_feature_dict_wrapper(emg_feats),\n    series_names=['AccV', 'AccML', 'AccAP'],\n    windows=[5_000],\n    strides=[5_000],\n)\n\nfc = FeatureCollection([basic_feats, emg_feats])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T07:50:06.573314Z","iopub.execute_input":"2023-05-07T07:50:06.573844Z","iopub.status.idle":"2023-05-07T07:50:06.650115Z","shell.execute_reply.started":"2023-05-07T07:50:06.573801Z","shell.execute_reply":"2023-05-07T07:50:06.648687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metadata_complex = metadata.merge(subjects,how='left',on='Subject').copy()\nmetadata_complex['Medication'] = metadata_complex['Medication'].factorize()[0]","metadata":{"execution":{"iopub.status.busy":"2023-05-07T07:50:10.852976Z","iopub.execute_input":"2023-05-07T07:50:10.853821Z","iopub.status.idle":"2023-05-07T07:50:10.871962Z","shell.execute_reply.started":"2023-05-07T07:50:10.853774Z","shell.execute_reply":"2023-05-07T07:50:10.870252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"start load defog\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# #reference: https://www.kaggle.com/code/ghrangel/read-data-and-merge\n# DATA_ROOT_DEFOG = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/'\n# defog = pd.DataFrame()\n# try:\n#     for root, dirs, files in os.walk(DATA_ROOT_DEFOG):\n#         for name in files:       \n#             f = os.path.join(root, name)\n#             df_list= pd.read_csv(f, usecols=['Time', 'AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn' , 'Walking','Valid','Task'])\n#             words = name.split('.')[0]\n#             df_list['Id']= words\n#             df_list = pd.merge(df_list, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n#             df_list = pd.merge(df_list, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_kmeans']], how='left', on='Id').fillna(-1)\n#             defog_feats = fc.calculate(df_list, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n#             df_list = df_list.merge(defog_feats, how=\"left\", left_index=True, right_index=True)\n#             df_list.fillna(method=\"ffill\", inplace=True)\n#             defog = pd.concat([defog, df_list], axis=0)\n# except Exception as e:\n#     raise e\n# defog.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:47:58.670050Z","iopub.execute_input":"2023-05-06T13:47:58.670436Z","iopub.status.idle":"2023-05-06T13:51:05.554908Z","shell.execute_reply.started":"2023-05-06T13:47:58.670403Z","shell.execute_reply":"2023-05-06T13:51:05.553050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defog = reduce_memory_usage(defog)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:52:42.789786Z","iopub.execute_input":"2023-05-06T13:52:42.790719Z","iopub.status.idle":"2023-05-06T13:53:27.618131Z","shell.execute_reply.started":"2023-05-06T13:52:42.790668Z","shell.execute_reply":"2023-05-06T13:53:27.616771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defog = defog[(defog['Task']==1)&(defog['Valid']==1)]","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:53:27.620783Z","iopub.execute_input":"2023-05-06T13:53:27.621386Z","iopub.status.idle":"2023-05-06T13:53:29.974679Z","shell.execute_reply.started":"2023-05-06T13:53:27.621336Z","shell.execute_reply":"2023-05-06T13:53:29.973363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(\"Shape of the defog dataset with all Valid entries \"+str(defog.shape))\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:53:29.976129Z","iopub.execute_input":"2023-05-06T13:53:29.976485Z","iopub.status.idle":"2023-05-06T13:53:29.983097Z","shell.execute_reply.started":"2023-05-06T13:53:29.976450Z","shell.execute_reply":"2023-05-06T13:53:29.981869Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(\"Finish load defog\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(\"start load tdcsfog\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# DATA_ROOT_TDCSFOG = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/'\n# tdcsfog = pd.DataFrame()\n# try:\n#     for root, dirs, files in os.walk(DATA_ROOT_TDCSFOG):\n#         for name in files:       \n#             f = os.path.join(root, name)\n#             df_list= pd.read_csv(f, usecols=['Time', 'AccV', 'AccML', 'AccAP', 'StartHesitation', 'Turn' , 'Walking'])\n#             words = name.split('.')[0]\n#             df_list['Id']= words\n#             df_list = pd.merge(df_list, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n#             df_list = pd.merge(df_list, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_kmeans']], how='left', on='Id').fillna(-1)\n#             tdcsfog_feats = fc.calculate(df_list, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n#             df_list = df_list.merge(tdcsfog_feats, how=\"left\", left_index=True, right_index=True)\n#             df_list.fillna(method=\"ffill\", inplace=True)\n#             tdcsfog = pd.concat([tdcsfog, df_list], axis=0)\n# except Exception as e:\n#     raise e\n# tdcsfog.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:53:29.985831Z","iopub.execute_input":"2023-05-06T13:53:29.986817Z","iopub.status.idle":"2023-05-06T14:06:22.652820Z","shell.execute_reply.started":"2023-05-06T13:53:29.986779Z","shell.execute_reply":"2023-05-06T14:06:22.651176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tdcsfog = reduce_memory_usage(tdcsfog)\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:06:22.654730Z","iopub.execute_input":"2023-05-06T14:06:22.655097Z","iopub.status.idle":"2023-05-06T14:06:46.096871Z","shell.execute_reply.started":"2023-05-06T14:06:22.655050Z","shell.execute_reply":"2023-05-06T14:06:46.095483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(\"finish load tdcsfog\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_set = pd.concat([defog, tdcsfog])","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:06:46.098639Z","iopub.execute_input":"2023-05-06T14:06:46.098981Z","iopub.status.idle":"2023-05-06T14:06:49.334688Z","shell.execute_reply.started":"2023-05-06T14:06:46.098948Z","shell.execute_reply":"2023-05-06T14:06:49.333572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# defog = None\n# tdcsfog = None","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:06:49.336276Z","iopub.execute_input":"2023-05-06T14:06:49.336728Z","iopub.status.idle":"2023-05-06T14:06:49.341721Z","shell.execute_reply.started":"2023-05-06T14:06:49.336694Z","shell.execute_reply":"2023-05-06T14:06:49.340663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_set.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:51:05.760265Z","iopub.status.idle":"2023-05-06T13:51:05.761052Z","shell.execute_reply.started":"2023-05-06T13:51:05.760845Z","shell.execute_reply":"2023-05-06T13:51:05.760868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:06:49.342877Z","iopub.execute_input":"2023-05-06T14:06:49.343783Z","iopub.status.idle":"2023-05-06T14:06:49.540303Z","shell.execute_reply.started":"2023-05-06T14:06:49.343731Z","shell.execute_reply":"2023-05-06T14:06:49.539206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_set","metadata":{"execution":{"iopub.status.busy":"2023-05-06T13:51:05.764189Z","iopub.status.idle":"2023-05-06T13:51:05.765078Z","shell.execute_reply.started":"2023-05-06T13:51:05.764828Z","shell.execute_reply":"2023-05-06T13:51:05.764854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_set.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T11:02:57.348078Z","iopub.execute_input":"2023-05-06T11:02:57.348660Z","iopub.status.idle":"2023-05-06T11:02:57.371299Z","shell.execute_reply.started":"2023-05-06T11:02:57.348602Z","shell.execute_reply":"2023-05-06T11:02:57.369966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_before_split_to_valid = train_set.drop(['Test', 'Valid', 'Id', 'Subject','Task', 'Time'], axis=1)\n# train_before_split_to_valid","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:07:24.857558Z","iopub.execute_input":"2023-05-06T14:07:24.858029Z","iopub.status.idle":"2023-05-06T14:07:32.776395Z","shell.execute_reply.started":"2023-05-06T14:07:24.857987Z","shell.execute_reply":"2023-05-06T14:07:32.775030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_before_split_to_valid.info()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T11:03:05.457823Z","iopub.execute_input":"2023-05-06T11:03:05.458367Z","iopub.status.idle":"2023-05-06T11:03:05.476616Z","shell.execute_reply.started":"2023-05-06T11:03:05.458320Z","shell.execute_reply":"2023-05-06T11:03:05.474922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold, StratifiedKFold, train_test_split, GridSearchCV\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T07:50:23.363701Z","iopub.execute_input":"2023-05-07T07:50:23.364208Z","iopub.status.idle":"2023-05-07T07:50:23.370339Z","shell.execute_reply.started":"2023-05-07T07:50:23.364165Z","shell.execute_reply":"2023-05-07T07:50:23.369040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# conditions = [\n#     (train_before_split_to_valid['StartHesitation'] == 1),\n#     (train_before_split_to_valid['Turn'] == 1),\n#     (train_before_split_to_valid['Walking'] == 1)]\n# choices = ['StartHesitation', 'Turn', 'Walking']\n# train_before_split_to_valid['event'] = np.select(conditions, choices, default='Normal')","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:07:40.871804Z","iopub.execute_input":"2023-05-06T14:07:40.872313Z","iopub.status.idle":"2023-05-06T14:07:44.452564Z","shell.execute_reply.started":"2023-05-06T14:07:40.872267Z","shell.execute_reply":"2023-05-06T14:07:44.451174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_before_split_to_valid['event'].value_counts().to_frame().style.background_gradient()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:07:44.454598Z","iopub.execute_input":"2023-05-06T14:07:44.455045Z","iopub.status.idle":"2023-05-06T14:07:45.509299Z","shell.execute_reply.started":"2023-05-06T14:07:44.455009Z","shell.execute_reply":"2023-05-06T14:07:45.507888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nle = LabelEncoder()\n\n# train_before_split_to_valid['target'] = le.fit_transform(train_before_split_to_valid['event'])","metadata":{"execution":{"iopub.status.busy":"2023-05-07T07:50:27.361497Z","iopub.execute_input":"2023-05-07T07:50:27.362007Z","iopub.status.idle":"2023-05-07T07:50:27.367887Z","shell.execute_reply.started":"2023-05-07T07:50:27.361966Z","shell.execute_reply":"2023-05-07T07:50:27.366671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_before_split_to_valid = train_before_split_to_valid.drop(['StartHesitation', 'Turn', 'Walking'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:07:51.016797Z","iopub.execute_input":"2023-05-06T14:07:51.017362Z","iopub.status.idle":"2023-05-06T14:07:53.661543Z","shell.execute_reply.started":"2023-05-06T14:07:51.017320Z","shell.execute_reply":"2023-05-06T14:07:53.660185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_before_split_to_valid.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T11:57:12.841964Z","iopub.execute_input":"2023-05-06T11:57:12.842454Z","iopub.status.idle":"2023-05-06T11:57:12.865025Z","shell.execute_reply.started":"2023-05-06T11:57:12.842408Z","shell.execute_reply":"2023-05-06T11:57:12.863572Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_set, val_set = train_test_split(train_before_split_to_valid, test_size=0.2, stratify=train_before_split_to_valid['target'], random_state=42)\n\n# # check the distribution of target values in the training set\n# train_dist = train_set['target'].value_counts(normalize=True)\n# print(\"Training set target distribution:\\n\", train_dist)\n\n# # check the distribution of target values in the validation set\n# val_dist = val_set['target'].value_counts(normalize=True)\n# print(\"Validation set target distribution:\\n\", val_dist)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:07:54.010793Z","iopub.execute_input":"2023-05-06T14:07:54.011259Z","iopub.status.idle":"2023-05-06T14:08:13.040908Z","shell.execute_reply.started":"2023-05-06T14:07:54.011199Z","shell.execute_reply":"2023-05-06T14:08:13.039313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_set.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-06T11:58:09.834317Z","iopub.execute_input":"2023-05-06T11:58:09.834863Z","iopub.status.idle":"2023-05-06T11:58:09.844550Z","shell.execute_reply.started":"2023-05-06T11:58:09.834805Z","shell.execute_reply":"2023-05-06T11:58:09.843115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# val_set.shape\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T11:58:12.139964Z","iopub.execute_input":"2023-05-06T11:58:12.140572Z","iopub.status.idle":"2023-05-06T11:58:12.154479Z","shell.execute_reply.started":"2023-05-06T11:58:12.140511Z","shell.execute_reply":"2023-05-06T11:58:12.153479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_train = train_set.drop(['event','target'], axis=1)\n# y_train = train_set['target']\n# X_val = val_set.drop(['event','target'], axis=1)\n# y_val = val_set['target']","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:08:13.042940Z","iopub.execute_input":"2023-05-06T14:08:13.043620Z","iopub.status.idle":"2023-05-06T14:08:15.418740Z","shell.execute_reply.started":"2023-05-06T14:08:13.043581Z","shell.execute_reply":"2023-05-06T14:08:15.417354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_val.shape","metadata":{"execution":{"iopub.status.busy":"2023-05-06T11:58:36.777625Z","iopub.execute_input":"2023-05-06T11:58:36.778075Z","iopub.status.idle":"2023-05-06T11:58:36.786997Z","shell.execute_reply.started":"2023-05-06T11:58:36.778038Z","shell.execute_reply":"2023-05-06T11:58:36.785535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(\"start train\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import warnings\n# import lightgbm as lgb\n# warnings.filterwarnings('ignore')\n\n\n\n# #setting up the parameters\n# params={}\n# params['learning_rate']=0.02\n# params['boosting_type']='gbdt' #GradientBoostingDecisionTree\n# params['objective']='multiclass' #Multi-class target feature\n# params['metric']='multi_logloss' #metric for multi-class\n# params['max_depth']=8\n# params['num_class']=4 #no.of unique values in the target class not inclusive of the end value\n# params['verbose']=-1\n\n# lgb_train = lgb.Dataset(X_train, label=y_train)\n# lgb_valid = lgb.Dataset(X_val, label=y_val)\n\n# #training the model\n# model_lgb = lgb.train(\n#     params = params,\n#     train_set = lgb_train,\n#     num_boost_round = 150,\n#     valid_sets = [lgb_valid, lgb_train],\n#     early_stopping_rounds = 25,\n#     verbose_eval = 25,\n#     )\n# print(\"Finish train\")","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:08:15.420676Z","iopub.execute_input":"2023-05-06T14:08:15.421910Z","iopub.status.idle":"2023-05-06T14:31:46.968501Z","shell.execute_reply.started":"2023-05-06T14:08:15.421861Z","shell.execute_reply":"2023-05-06T14:31:46.967181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pickle\n# # Save the model as a pickle file\n# output_file = \"/kaggle/working/lgb_extra_features.pkl\"\n# with open(output_file, 'wb') as f:\n#     pickle.dump(model_lgb, f)\n\n# print(\"Model saved as\", output_file)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:31:46.971305Z","iopub.execute_input":"2023-05-06T14:31:46.971650Z","iopub.status.idle":"2023-05-06T14:31:47.045429Z","shell.execute_reply.started":"2023-05-06T14:31:46.971616Z","shell.execute_reply":"2023-05-06T14:31:47.043932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# y_pred = model_lgb.predict(X_val)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:35:38.830556Z","iopub.execute_input":"2023-05-06T14:35:38.831141Z","iopub.status.idle":"2023-05-06T14:36:13.016588Z","shell.execute_reply.started":"2023-05-06T14:35:38.831072Z","shell.execute_reply":"2023-05-06T14:36:13.015379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# from sklearn.metrics import accuracy_score\n# y_pred_class = [np.argmax(line) for line in y_pred]\n# accuracy = accuracy_score(y_val, y_pred_class)\n# print(\"Validation accuracy:\", accuracy)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:36:13.018500Z","iopub.execute_input":"2023-05-06T14:36:13.019269Z","iopub.status.idle":"2023-05-06T14:36:17.626782Z","shell.execute_reply.started":"2023-05-06T14:36:13.019224Z","shell.execute_reply":"2023-05-06T14:36:17.625292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np\n# from sklearn.metrics import precision_recall_curve\n# from sklearn.metrics import average_precision_score\n\n# def calculate_ap_for_class(y_true, y_pred, class_label):\n#     \"\"\"\n#     Calculate the average precision score for a given class.\n\n#     Parameters:\n#     - y_true (array-like): True labels.\n#     - y_pred (array-like): Predicted labels.\n#     - class_label (int): Class label for which to calculate the AP score.\n\n#     Returns:\n#     - ap_score (float): Average precision score for the given class.\n#     \"\"\"\n\n#     # Convert y_true to a numpy array\n#     y_true = np.array(y_true)\n\n#     # Get the indices of samples that belong to the given class\n#     class_indices = np.where(y_true == class_label)[0]\n\n#     # Create a new array with true labels where 1 indicates samples that belong to the given class\n#     y_true_class = np.zeros_like(y_true)\n#     y_true_class[class_indices] = 1\n\n#     # Extract the predicted probabilities for the given class\n#     y_pred_class = y_pred[:, class_label]\n\n#     # Calculate precision and recall values at different thresholds\n#     ap_score = average_precision_score(y_true_class, y_pred_class)\n\n\n#     return ap_score\n","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:36:17.628759Z","iopub.execute_input":"2023-05-06T14:36:17.629583Z","iopub.status.idle":"2023-05-06T14:36:17.639626Z","shell.execute_reply.started":"2023-05-06T14:36:17.629531Z","shell.execute_reply":"2023-05-06T14:36:17.638171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ap_scores = []\n# ap_scores.append(calculate_ap_for_class(y_val, y_pred, 0))\n# ap_scores.append(calculate_ap_for_class(y_val, y_pred, 1))\n# ap_scores.append(calculate_ap_for_class(y_val, y_pred, 2))\n# ap_scores.append(calculate_ap_for_class(y_val, y_pred, 3))\n# print(\"MAP score:\", np.mean(ap_scores))","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:36:17.643299Z","iopub.execute_input":"2023-05-06T14:36:17.643838Z","iopub.status.idle":"2023-05-06T14:36:20.451816Z","shell.execute_reply.started":"2023-05-06T14:36:17.643782Z","shell.execute_reply":"2023-05-06T14:36:20.450461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print(ap_scores)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:36:20.452938Z","iopub.execute_input":"2023-05-06T14:36:20.453295Z","iopub.status.idle":"2023-05-06T14:36:20.459664Z","shell.execute_reply.started":"2023-05-06T14:36:20.453261Z","shell.execute_reply":"2023-05-06T14:36:20.458538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output_defog = pd.read_pickle('/kaggle/input/lgb-extra-features-model/lgb_extra_features.pkl')","metadata":{"execution":{"iopub.status.busy":"2023-05-07T07:50:36.261824Z","iopub.execute_input":"2023-05-07T07:50:36.262957Z","iopub.status.idle":"2023-05-07T07:50:36.381793Z","shell.execute_reply.started":"2023-05-07T07:50:36.262895Z","shell.execute_reply":"2023-05-07T07:50:36.379851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:38:37.424168Z","iopub.execute_input":"2023-05-06T14:38:37.425021Z","iopub.status.idle":"2023-05-06T14:38:37.649098Z","shell.execute_reply.started":"2023-05-06T14:38:37.424966Z","shell.execute_reply":"2023-05-06T14:38:37.647640Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# del train_set","metadata":{"execution":{"iopub.status.busy":"2023-05-06T14:39:10.134187Z","iopub.execute_input":"2023-05-06T14:39:10.135267Z","iopub.status.idle":"2023-05-06T14:39:10.590687Z","shell.execute_reply.started":"2023-05-06T14:39:10.135209Z","shell.execute_reply":"2023-05-06T14:39:10.589355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_ROOT_DEFOG = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/defog'\ndefog_test = pd.DataFrame()\ntry:\n    for root, dirs, files in os.walk(DATA_ROOT_DEFOG):\n        for name in files:       \n            f = os.path.join(root, name)\n            df_list= pd.read_csv(f)\n            words = name.split('.')[0]\n            df_list['Id']= words\n            df_list = pd.merge(df_list, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n            df_list = pd.merge(df_list, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_kmeans']], how='left', on='Id').fillna(-1)\n            tdcsfog_feats = fc.calculate(df_list, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n            df_list = df_list.merge(tdcsfog_feats, how=\"left\", left_index=True, right_index=True)\n            df_list.fillna(method=\"ffill\", inplace=True)\n            df_list['Id'] = df_list['Id'].astype(str) + '_' + df_list['Time'].astype(str)\n            df_list.set_index('Id',inplace=True)\n            df_list.drop(['Subject','Time', 'Test'],axis = 1,inplace = True)\n            defog_preds = output_defog.predict(df_list)\n            defog_test = pd.concat([defog_test, df_list], axis=0)\n            defog_test['event'] = np.argmax(defog_preds, axis=-1)\n            defog_test['target'] = le.fit_transform(defog_test['event'])\n            defog_test['event'] = le.inverse_transform(defog_test['event'])\n            defog_test['StartHesitation'] = np.where(defog_test['event']=='StartHesitation', 1, 0)\n            defog_test['Turn'] = np.where(defog_test['event']=='Turn', 1, 0)\n            defog_test['Walking'] = np.where(defog_test['event']=='Walking', 1, 0)\n            defog_test = defog_test[['StartHesitation','Turn','Walking']]\n            defog_test = defog_test.reset_index()\nexcept Exception as e:\n    raise e","metadata":{"execution":{"iopub.status.busy":"2023-05-07T07:50:45.662057Z","iopub.execute_input":"2023-05-07T07:50:45.662517Z","iopub.status.idle":"2023-05-07T07:50:51.347904Z","shell.execute_reply.started":"2023-05-07T07:50:45.662479Z","shell.execute_reply":"2023-05-07T07:50:51.345617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATA_ROOT_TDCSFOG = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/test/tdcsfog'\ntdcsfog_test = pd.DataFrame()\ntry:\n    for root, dirs, files in os.walk(DATA_ROOT_TDCSFOG):\n        for name in files:       \n            f = os.path.join(root, name)\n            df_list= pd.read_csv(f)\n            words = name.split('.')[0]\n            df_list['Id']= words\n            df_list = pd.merge(df_list, tasks[['Id','t_kmeans']], how='left', on='Id').fillna(-1)\n            df_list = pd.merge(df_list, metadata_complex[['Id','Subject']+['Visit','Test','Medication','s_kmeans']], how='left', on='Id').fillna(-1)\n            tdcsfog_feats = fc.calculate(df_list, return_df=True, include_final_window=True, approve_sparsity=True, window_idx=\"begin\").astype(np.float32)\n            df_list = df_list.merge(tdcsfog_feats, how=\"left\", left_index=True, right_index=True)\n            df_list.fillna(method=\"ffill\", inplace=True)\n            df_list['Id'] = df_list['Id'].astype(str) + '_' + df_list['Time'].astype(str)\n            df_list.set_index('Id',inplace=True)\n            df_list.drop(['Subject','Time', 'Test'],axis = 1,inplace = True)\n            tfog_preds = output_defog.predict(df_list)\n            tdcsfog_test = pd.concat([tdcsfog_test, df_list], axis=0)\n            tdcsfog_test['event'] = np.argmax(tfog_preds, axis=-1)\n            tdcsfog_test['target'] = le.fit_transform(tdcsfog_test['event'])\n            tdcsfog_test['event'] = le.inverse_transform(tdcsfog_test['event'])\n            tdcsfog_test['StartHesitation'] = np.where(tdcsfog_test['event']=='StartHesitation', 1, 0)\n            tdcsfog_test['Turn'] = np.where(tdcsfog_test['event']=='Turn', 1, 0)\n            tdcsfog_test['Walking'] = np.where(tdcsfog_test['event']=='Walking', 1, 0)\n            tdcsfog_test = tdcsfog_test[['StartHesitation','Turn','Walking']]\n            tdcsfog_test = tdcsfog_test.reset_index()\nexcept Exception as e:\n    raise e","metadata":{"execution":{"iopub.status.busy":"2023-05-07T07:51:17.806065Z","iopub.execute_input":"2023-05-07T07:51:17.806682Z","iopub.status.idle":"2023-05-07T07:51:18.363600Z","shell.execute_reply.started":"2023-05-07T07:51:17.806614Z","shell.execute_reply":"2023-05-07T07:51:18.361769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.concat([tdcsfog_test,defog_test])\nsubmission.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-07T07:51:24.675002Z","iopub.execute_input":"2023-05-07T07:51:24.675616Z","iopub.status.idle":"2023-05-07T07:51:24.709800Z","shell.execute_reply.started":"2023-05-07T07:51:24.675553Z","shell.execute_reply":"2023-05-07T07:51:24.708450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-06T12:29:45.857533Z","iopub.execute_input":"2023-05-06T12:29:45.858001Z","iopub.status.idle":"2023-05-06T12:29:47.329736Z","shell.execute_reply.started":"2023-05-06T12:29:45.857960Z","shell.execute_reply":"2023-05-06T12:29:47.328375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv',index = False)","metadata":{"execution":{"iopub.status.busy":"2023-05-06T12:30:16.952946Z","iopub.execute_input":"2023-05-06T12:30:16.953431Z","iopub.status.idle":"2023-05-06T12:30:17.392170Z","shell.execute_reply.started":"2023-05-06T12:30:16.953392Z","shell.execute_reply":"2023-05-06T12:30:17.390852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Dataset**","metadata":{}},{"cell_type":"code","source":"# class FOGDataset(Dataset):\n#     def __init__(self, fpaths, scale=9.806, split=\"train\"):\n#         super(FOGDataset, self).__init__()\n#         tm = time.time()\n#         self.split = split\n#         self.scale = scale\n        \n#         self.fpaths = fpaths\n#         self.dfs = [self.read(f[0], f[1]) for f in fpaths]\n#         self.f_ids = [os.path.basename(f[0])[:-4] for f in self.fpaths]\n        \n#         self.end_indices = []\n#         self.shapes = []\n#         _length = 0\n#         for df in self.dfs:\n#             self.shapes.append(df.shape[0])\n#             _length += df.shape[0]\n#             self.end_indices.append(_length)\n        \n#         self.dfs = np.concatenate(self.dfs, axis=0).astype(np.float16)\n#         self.length = self.dfs.shape[0]\n        \n#         shape1 = self.dfs.shape[1]\n        \n#         self.dfs = np.concatenate([np.zeros((cfg.wx*cfg.window_past, shape1)), self.dfs, np.zeros((cfg.wx*cfg.window_future, shape1))], axis=0)\n#         print(f\"Dataset initialized in {time.time() - tm} secs!\")\n#         gc.collect()\n        \n#     def read(self, f, _type):\n#         df = pd.read_csv(f)\n#         if self.split == \"test\":\n#             return np.array(df)\n        \n#         if _type ==\"tdcs\":\n#             df['Valid'] = 1\n#             df['Task'] = 1\n#             df['tdcs'] = 1\n#         else:\n#             df['tdcs'] = 0\n        \n#         return np.array(df)\n            \n#     def __getitem__(self, index):\n#         if self.split == \"train\":\n#             row_idx = random.randint(0, self.length-1) + cfg.wx*cfg.window_past\n#         elif self.split == \"test\":\n#             for i,e in enumerate(self.end_indices):\n#                 if index >= e:\n#                     continue\n#                 df_idx = i\n#                 break\n\n#             row_idx_true = self.shapes[df_idx] - (self.end_indices[df_idx] - index)\n#             _id = self.f_ids[df_idx] + \"_\" + str(row_idx_true)\n#             row_idx = index + cfg.wx*cfg.window_past\n#         else:\n#             row_idx = index + cfg.wx*cfg.window_past\n            \n#         #scale = 9.806 if self.dfs[row_idx, -1] == 1 else 1.0\n#         x = self.dfs[row_idx - cfg.wx*cfg.window_past : row_idx + cfg.wx*cfg.window_future, 1:4]\n#         x = x[::cfg.wx, :][::-1, :]\n#         x = torch.tensor(x.astype('float'))#/scale\n        \n#         t = self.dfs[row_idx, -3]*self.dfs[row_idx, -2]\n        \n#         if self.split == \"test\":\n#             return _id, x, t\n        \n#         y = self.dfs[row_idx, 4:7].astype('float')\n#         y = torch.tensor(y)\n        \n#         return x, y, t\n    \n#     def __len__(self):\n#         # return self.length\n#         if self.split == \"train\":\n#             return 5_000_000\n#         return self.length","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:33:15.554756Z","iopub.execute_input":"2023-05-01T16:33:15.556018Z","iopub.status.idle":"2023-05-01T16:33:15.577202Z","shell.execute_reply.started":"2023-05-01T16:33:15.555957Z","shell.execute_reply":"2023-05-01T16:33:15.575888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# gc.collect()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:33:31.394301Z","iopub.execute_input":"2023-05-01T16:33:31.394692Z","iopub.status.idle":"2023-05-01T16:33:31.609724Z","shell.execute_reply.started":"2023-05-01T16:33:31.394658Z","shell.execute_reply":"2023-05-01T16:33:31.608578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model**","metadata":{}},{"cell_type":"code","source":"# def _block(in_features, out_features, drop_rate):\n#     return nn.Sequential(\n#         nn.Linear(in_features, out_features),\n#         nn.BatchNorm1d(out_features),\n#         nn.ReLU(),\n#         nn.Dropout(drop_rate)\n#     )\n\n# class FOGModel(nn.Module):\n#     def __init__(self, p=cfg.model_dropout, dim=cfg.model_hidden, nblocks=cfg.model_nblocks):\n#         super(FOGModel, self).__init__()\n#         self.dropout = nn.Dropout(p)\n#         self.in_layer = nn.Linear(cfg.window_size*3, dim)\n#         self.blocks = nn.Sequential(*[_block(dim, dim, p) for _ in range(nblocks)])\n#         self.out_layer = nn.Linear(dim, 3)\n        \n#     def forward(self, x):\n#         x = x.view(-1, cfg.window_size*3)\n#         x = self.in_layer(x)\n#         for block in self.blocks:\n#             x = block(x)\n#         x = self.out_layer(x)\n#         return x","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:36:05.074314Z","iopub.execute_input":"2023-05-01T16:36:05.075616Z","iopub.status.idle":"2023-05-01T16:36:05.086571Z","shell.execute_reply.started":"2023-05-01T16:36:05.075553Z","shell.execute_reply":"2023-05-01T16:36:05.085061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def count_parameters(model):\n#     return sum(p.numel() for p in model.parameters() if p.requires_grad)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:36:14.821396Z","iopub.execute_input":"2023-05-01T16:36:14.821796Z","iopub.status.idle":"2023-05-01T16:36:14.827703Z","shell.execute_reply.started":"2023-05-01T16:36:14.821758Z","shell.execute_reply":"2023-05-01T16:36:14.826182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train**","metadata":{}},{"cell_type":"code","source":"# from torch.cuda.amp import GradScaler\n\n# def train_one_epoch(model, loader, optimizer, criterion):\n#     loss_sum = 0.\n#     scaler = GradScaler()\n    \n#     model.train()\n#     for x,y,t in tqdm(loader):\n#         x = x.to(cfg.device).float()\n#         y = y.to(cfg.device).float()\n#         t = t.to(cfg.device).float()\n        \n#         y_pred = model(x)\n#         loss = criterion(y_pred, y)\n#         loss = torch.mean(loss*t.unsqueeze(-1), dim=1)\n        \n#         t_sum = torch.sum(t)\n#         if t_sum > 0:\n#             loss = torch.sum(loss)/t_sum\n#         else:\n#             loss = torch.sum(loss)*0.\n        \n#         # loss.backward()\n#         scaler.scale(loss).backward()\n#         # optimizer.step()\n#         scaler.step(optimizer)\n#         scaler.update()\n        \n#         optimizer.zero_grad()\n        \n#         loss_sum += loss.item()\n    \n#     print(f\"Train Loss: {(loss_sum/len(loader)):.04f}\")\n    \n\n# def validation_one_epoch(model, loader, criterion):\n#     loss_sum = 0.\n#     y_true_epoch = []\n#     y_pred_epoch = []\n#     t_valid_epoch = []\n    \n#     model.eval()\n#     for x,y,t in tqdm(loader):\n#         x = x.to(cfg.device).float()\n#         y = y.to(cfg.device).float()\n#         t = t.to(cfg.device).float()\n        \n#         with torch.no_grad():\n#             y_pred = model(x)\n#             loss = criterion(y_pred, y)\n#             loss = torch.mean(loss*t.unsqueeze(-1), dim=1)\n            \n#             t_sum = torch.sum(t)\n#             if t_sum > 0:\n#                 loss = torch.sum(loss)/t_sum\n#             else:\n#                 loss = torch.sum(loss)*0.\n        \n#         loss_sum += loss.item()\n#         y_true_epoch.append(y.cpu().numpy())\n#         y_pred_epoch.append(y_pred.cpu().numpy())\n#         t_valid_epoch.append(t.cpu().numpy())\n        \n#     y_true_epoch = np.concatenate(y_true_epoch, axis=0)\n#     y_pred_epoch = np.concatenate(y_pred_epoch, axis=0)\n    \n#     t_valid_epoch = np.concatenate(t_valid_epoch, axis=0)\n#     y_true_epoch = y_true_epoch[t_valid_epoch > 0, :]\n#     y_pred_epoch = y_pred_epoch[t_valid_epoch > 0, :]\n    \n#     scores = [average_precision_score(y_true_epoch[:,i], y_pred_epoch[:,i]) for i in range(3)]\n#     mean_score = np.mean(scores)\n#     print(f\"Validation Loss: {(loss_sum/len(loader)):.04f}, Validation Score: {mean_score:.03f}, ClassWise: {scores[0]:.03f},{scores[1]:.03f},{scores[2]:.03f}\")\n    \n#     return mean_score","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:37:14.632437Z","iopub.execute_input":"2023-05-01T16:37:14.632866Z","iopub.status.idle":"2023-05-01T16:37:14.652891Z","shell.execute_reply.started":"2023-05-01T16:37:14.632824Z","shell.execute_reply":"2023-05-01T16:37:14.651462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model = FOGModel().to(cfg.device)\n# print(f\"Number of parameters in model - {count_parameters(model):,}\")\n\n# train_dataset = FOGDataset(train_fpaths, split=\"train\")\n# valid_dataset = FOGDataset(valid_fpaths, split=\"valid\")\n# print(f\"lengths of datasets: train - {len(train_dataset)}, valid - {len(valid_dataset)}\")\n\n# train_loader = DataLoader(train_dataset, batch_size=cfg.batch_size, num_workers=5, shuffle=True)\n# valid_loader = DataLoader(valid_dataset, batch_size=cfg.batch_size, num_workers=5)\n\n# optimizer = torch.optim.Adam(model.parameters(), lr=cfg.lr)\n# criterion = torch.nn.BCEWithLogitsLoss(reduction='none').to(cfg.device)\n# # sched = torch.optim.lr_scheduler.StepLR(optimizer, step_size=1, gamma=0.85)\n\n# max_score = 0.0\n\n# print(\"=\"*50)\n# for epoch in range(cfg.num_epochs):\n#     print(f\"Epoch: {epoch}\")\n#     train_one_epoch(model, train_loader, optimizer, criterion)\n#     score = validation_one_epoch(model, valid_loader, criterion)\n#     # sched.step()\n\n#     if score > max_score:\n#         max_score = score\n#         torch.save(model.state_dict(), \"best_model_state.h5\")\n#         print(\"Saving Model ...\")\n\n#     print(\"=\"*50)\n    \n# gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:38:02.523077Z","iopub.execute_input":"2023-05-01T16:38:02.523519Z","iopub.status.idle":"2023-05-01T17:46:31.750114Z","shell.execute_reply.started":"2023-05-01T16:38:02.523477Z","shell.execute_reply":"2023-05-01T17:46:31.748682Z"},"trusted":true},"execution_count":null,"outputs":[]}]}