{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-08-21T16:41:42.859323Z","iopub.execute_input":"2023-08-21T16:41:42.859822Z","iopub.status.idle":"2023-08-21T16:41:43.288373Z","shell.execute_reply.started":"2023-08-21T16:41:42.859785Z","shell.execute_reply":"2023-08-21T16:41:43.287333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport warnings\nimport optuna\nfrom joblib import load\nimport joblib\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report\nfrom sklearn.model_selection import train_test_split, cross_val_score\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as mpatches\nimport plotly.express as px","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:41:52.462459Z","iopub.execute_input":"2023-08-21T16:41:52.462872Z","iopub.status.idle":"2023-08-21T16:41:52.471861Z","shell.execute_reply.started":"2023-08-21T16:41:52.46284Z","shell.execute_reply":"2023-08-21T16:41:52.469979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class color:\n   PURPLE = '\\033[95m'\n   CYAN = '\\033[96m'\n   DARKCYAN = '\\033[36m'\n   BLUE = '\\033[94m'\n   GREEN = '\\033[92m'\n   YELLOW = '\\033[93m'\n   RED = '\\033[91m'\n   BOLD = '\\033[1m'\n   UNDERLINE = '\\033[4m'\n   END = '\\033[0m'","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:41:56.235325Z","iopub.execute_input":"2023-08-21T16:41:56.235757Z","iopub.status.idle":"2023-08-21T16:41:56.24213Z","shell.execute_reply.started":"2023-08-21T16:41:56.235725Z","shell.execute_reply":"2023-08-21T16:41:56.240848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"warnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:41:59.596473Z","iopub.execute_input":"2023-08-21T16:41:59.596884Z","iopub.status.idle":"2023-08-21T16:41:59.602848Z","shell.execute_reply.started":"2023-08-21T16:41:59.596853Z","shell.execute_reply":"2023-08-21T16:41:59.601434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Time Series\n\n* Train Folder","metadata":{}},{"cell_type":"code","source":"# let's check what directories are in the train folder\nos.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train\")","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:42:02.636355Z","iopub.execute_input":"2023-08-21T16:42:02.6368Z","iopub.status.idle":"2023-08-21T16:42:02.646356Z","shell.execute_reply.started":"2023-08-21T16:42:02.636765Z","shell.execute_reply":"2023-08-21T16:42:02.644849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How many files are in folder tdcsfog/.\n\ntemp = len(os.listdir(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog\"))\nprint(\n    f\"Number of files in folder tdcsfog/: {color.BLUE}{temp}{color.END}\",\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:42:05.560993Z","iopub.execute_input":"2023-08-21T16:42:05.561378Z","iopub.status.idle":"2023-08-21T16:42:05.56864Z","shell.execute_reply.started":"2023-08-21T16:42:05.561347Z","shell.execute_reply":"2023-08-21T16:42:05.567353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# train_tdcsfog dataset\n* We will check for time series","metadata":{}},{"cell_type":"code","source":"train_tdcsfog_example_df = pd.read_csv(\"/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/003f117e14.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:11:10.017403Z","iopub.execute_input":"2023-08-21T16:11:10.017791Z","iopub.status.idle":"2023-08-21T16:11:10.03331Z","shell.execute_reply.started":"2023-08-21T16:11:10.01776Z","shell.execute_reply":"2023-08-21T16:11:10.032393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = len(train_tdcsfog_example_df)\nprint(\n    f\"Length of dataframe: {color.BLUE}{temp}{color.END}\",\n)","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:11:12.207129Z","iopub.execute_input":"2023-08-21T16:11:12.207488Z","iopub.status.idle":"2023-08-21T16:11:12.213177Z","shell.execute_reply.started":"2023-08-21T16:11:12.20746Z","shell.execute_reply":"2023-08-21T16:11:12.212012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_tdcsfog_example_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_tdcsfog_example_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:11:16.197295Z","iopub.execute_input":"2023-08-21T16:11:16.197651Z","iopub.status.idle":"2023-08-21T16:11:16.234898Z","shell.execute_reply.started":"2023-08-21T16:11:16.197621Z","shell.execute_reply":"2023-08-21T16:11:16.233976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Is there any NaN values in the dataframe.\n\ntrain_tdcsfog_example_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:11:18.301153Z","iopub.execute_input":"2023-08-21T16:11:18.301835Z","iopub.status.idle":"2023-08-21T16:11:18.311511Z","shell.execute_reply.started":"2023-08-21T16:11:18.301793Z","shell.execute_reply":"2023-08-21T16:11:18.310302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for column in ['AccV','AccML','AccAP']:\n    fig = px.line(train_tdcsfog_example_df, x=\"Time\", y=column, color_discrete_sequence=['darkslateblue'])\n    fig.update_layout(\n        title={\n            'text': f\"{column} Time Series\",\n            'y':0.95,\n            'x':0.5,\n            'xanchor': 'center',\n            'yanchor': 'top'\n        }\n    )\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:11:20.008879Z","iopub.execute_input":"2023-08-21T16:11:20.009248Z","iopub.status.idle":"2023-08-21T16:11:21.598332Z","shell.execute_reply.started":"2023-08-21T16:11:20.009217Z","shell.execute_reply":"2023-08-21T16:11:21.597299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# WE WILL CHECK SIX DIFFERENT DATASETS FOR ML MODELS.","metadata":{}},{"cell_type":"code","source":"daily_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/daily_metadata.csv')\ndefog_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\ntdcsfog_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\nevents_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/events.csv')\nsubjects_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv')\ntasks_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tasks.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:11:27.9359Z","iopub.execute_input":"2023-08-21T16:11:27.936308Z","iopub.status.idle":"2023-08-21T16:11:27.994092Z","shell.execute_reply.started":"2023-08-21T16:11:27.936278Z","shell.execute_reply":"2023-08-21T16:11:27.993167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 1.DAILY DATASET\n\n* The Daily Living (daily) dataset, comprising one week of continuous 24/7 recordings from sixty-five subjects. \n\n* Forty-five subjects exhibit FOG symptoms and also have series in the defog dataset, while the other twenty subjects do not exhibit FOG symptoms and do not have series elsewhere in the data.\n\n* Id The data series the event occured in.\n\n* subjects.csv Metadata for each Subject in the study, including their Age and Sex.\n\n* Visit Lab visits consist of a baseline assessment, two post-treatment assessments for different treatment stages, and one follow-up assessment.","metadata":{}},{"cell_type":"code","source":"daily_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:11:30.90365Z","iopub.execute_input":"2023-08-21T16:11:30.904354Z","iopub.status.idle":"2023-08-21T16:11:30.914276Z","shell.execute_reply.started":"2023-08-21T16:11:30.904322Z","shell.execute_reply":"2023-08-21T16:11:30.913388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_df","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:11:33.638212Z","iopub.execute_input":"2023-08-21T16:11:33.638574Z","iopub.status.idle":"2023-08-21T16:11:33.654814Z","shell.execute_reply.started":"2023-08-21T16:11:33.638542Z","shell.execute_reply":"2023-08-21T16:11:33.653726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:11:36.038442Z","iopub.execute_input":"2023-08-21T16:11:36.038827Z","iopub.status.idle":"2023-08-21T16:11:36.044839Z","shell.execute_reply.started":"2023-08-21T16:11:36.038796Z","shell.execute_reply":"2023-08-21T16:11:36.04364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_df[daily_df['Visit']==1]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_df[daily_df['Visit']==2]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_df[daily_df['Visit']==1].shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_df[daily_df['Visit']==2].shape","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:11:47.269392Z","iopub.execute_input":"2023-08-21T16:11:47.269779Z","iopub.status.idle":"2023-08-21T16:11:47.278708Z","shell.execute_reply.started":"2023-08-21T16:11:47.269746Z","shell.execute_reply":"2023-08-21T16:11:47.277754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 2. The DeFOG (defog) dataset \n\n* Comprising data series collected in the subject's home, as subjects completed a FOG-provoking protocol\n\n* Medication Subjects may have been either off or on anti-parkinsonian medication during the recording.","metadata":{}},{"cell_type":"code","source":"defog_df.tail()","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:11:50.208273Z","iopub.execute_input":"2023-08-21T16:11:50.208629Z","iopub.status.idle":"2023-08-21T16:11:50.218617Z","shell.execute_reply.started":"2023-08-21T16:11:50.208598Z","shell.execute_reply":"2023-08-21T16:11:50.217693Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:11:53.357174Z","iopub.execute_input":"2023-08-21T16:11:53.357536Z","iopub.status.idle":"2023-08-21T16:11:53.369723Z","shell.execute_reply.started":"2023-08-21T16:11:53.357506Z","shell.execute_reply":"2023-08-21T16:11:53.368066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_df.info","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:11:55.976258Z","iopub.execute_input":"2023-08-21T16:11:55.976631Z","iopub.status.idle":"2023-08-21T16:11:55.98908Z","shell.execute_reply.started":"2023-08-21T16:11:55.9766Z","shell.execute_reply":"2023-08-21T16:11:55.987718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3. tdcsfog DATASET\n\n* Subjects lab recordings.","metadata":{}},{"cell_type":"code","source":"tdcsfog_df.head()\n# subjects lab recordings.","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:11:59.812577Z","iopub.execute_input":"2023-08-21T16:11:59.813027Z","iopub.status.idle":"2023-08-21T16:11:59.826991Z","shell.execute_reply.started":"2023-08-21T16:11:59.812992Z","shell.execute_reply":"2023-08-21T16:11:59.826166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:12:03.272498Z","iopub.execute_input":"2023-08-21T16:12:03.273202Z","iopub.status.idle":"2023-08-21T16:12:03.279385Z","shell.execute_reply.started":"2023-08-21T16:12:03.273168Z","shell.execute_reply":"2023-08-21T16:12:03.278486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_df['Visit'].value_counts()\n# 2 kere ziyarete gelme sayisi 245.","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:12:05.73236Z","iopub.execute_input":"2023-08-21T16:12:05.732739Z","iopub.status.idle":"2023-08-21T16:12:05.743302Z","shell.execute_reply.started":"2023-08-21T16:12:05.732706Z","shell.execute_reply":"2023-08-21T16:12:05.742163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_df['Visit'].value_counts().plot(kind='bar')\n\n# number of visits garphic\n# most visit number is \"2\"","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:12:09.388456Z","iopub.execute_input":"2023-08-21T16:12:09.38887Z","iopub.status.idle":"2023-08-21T16:12:09.660847Z","shell.execute_reply.started":"2023-08-21T16:12:09.388834Z","shell.execute_reply":"2023-08-21T16:12:09.659828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 4. SUBJECTS DATASET\n\n*  Visit Only available for subjects in the daily and defog datasets.\n\n*  YearsSinceDx Years since Parkinson's diagnosis.\n\n*  UPDRSIIIOn/UPDRSIIIOff Unified Parkinson's Disease Rating Scale score during on/off medication respectively.\n\n*  NFOGQ Self-report FoG questionnaire score.\n","metadata":{}},{"cell_type":"code","source":"subjects_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:12:13.408325Z","iopub.execute_input":"2023-08-21T16:12:13.408708Z","iopub.status.idle":"2023-08-21T16:12:13.425441Z","shell.execute_reply.started":"2023-08-21T16:12:13.40865Z","shell.execute_reply":"2023-08-21T16:12:13.424367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects_df['Sex'].value_counts().plot(kind='bar')","metadata":{"execution":{"iopub.status.busy":"2023-08-21T16:12:15.692161Z","iopub.execute_input":"2023-08-21T16:12:15.692558Z","iopub.status.idle":"2023-08-21T16:12:15.889502Z","shell.execute_reply.started":"2023-08-21T16:12:15.692528Z","shell.execute_reply":"2023-08-21T16:12:15.888588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 5. TASKS DATASET\n\n* Id The data series where the task was measured.\n\n* Begin Time (s) the task began.\n\n* End Time (s) the task ended.\n\n* Task One of seven tasks types in the DeFOG protocol.","metadata":{}},{"cell_type":"code","source":"tasks_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#What is the distibution of the difference between fields Begin and End.\n\n\ntasks_df[\"Duration\"] = tasks_df.End - tasks_df.Begin\nfig = px.histogram(x=tasks_df[\"Duration\"], nbins=30, color_discrete_sequence=['lightslategrey'])\nfig.update_layout(\n    xaxis_title=\"Task duration\",\n    title={\n        'text': \"Distribution of task durations\",\n        'y':0.95,\n        'x':0.5,\n        'xanchor': 'center',\n        'yanchor': 'top'\n    }\n)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How the bar chart looks like for categorical field Task.\n\ntasks_task_counts = tasks_df.Task.value_counts()\n\nfig = px.bar(x=tasks_task_counts.index, y=tasks_task_counts.values, color_discrete_sequence=['lightslategrey'])\nfig.update_layout(xaxis_title=\"Task\", yaxis_title=\"Count\")\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How the boxplot looks like for each task on duration.\n\nfig = px.box(\n    tasks_df, \n    x='Duration', y='Task',\n    orientation='h', height=1000, \n    color_discrete_sequence=['lightslategrey']\n)\n\nfig.update_layout(\n    title={\n        'text': \"Task Duration Boxplot\",\n        'y':0.98,\n        'x':0.5,\n        'xanchor': 'center',\n        'yanchor': 'top'\n    },\n    margin=dict(l=50, r=0, t=50, b=50)\n)\nfig.show(autosize=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 6. EVENTS DATASET\n\n* events.csv Metadata for each FoG event in all data series. The event times agree with the labels in the data series.\n\n* Visit The data series the event occured in.\n\n* Init Time (s) the event began.\n\n* Completion Time (s) the event ended.\n\n* Type Whether StartHesitation, Turn, or Walking.\n\n* Kinetic Whether the event was kinetic (1) and involved movement, or akinetic (0) and static.","metadata":{}},{"cell_type":"code","source":"events_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"events_df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = events_df\ndataset.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.isnull().sum()\n\n# Type and Kinetic columns have missing values.","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"events_df['Type'].value_counts().to_frame().style.background_gradient()\n\n# Distribution of FOG event.","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(15,7))\nax =sns.countplot(x='Type',data=events_df, palette=\"rocket\")\nax.set_title('Distribution of Type', fontsize=16, color='grey')\nfor container in ax.containers:\n    ax.bar_label(container)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig = plt.figure(figsize=(15,7))\nax =sns.countplot(x='Type',data=events_df, hue='Kinetic', palette=\"rocket\")\nax.set_title('Distribution of Type by Kinetic', fontsize=16, color='grey')\nfor container in ax.containers:\n    ax.bar_label(container)\nplt.show()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#events_df['Type'].value_counts().plot(kind='bar', color='pink')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#events_df['Kinetic'].value_counts().plot(kind='bar', color='pink')\n\nfig = plt.figure(figsize=(15,7))\nax =sns.countplot(x='Kinetic',data=events_df, palette=\"rocket\")\nax.set_title('Distribution of Kinetic', fontsize=16, color='grey')\nfor container in ax.containers:\n    ax.bar_label(container)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# What is the distibution of the field Init.\n\nfig = px.histogram(events_df, x=\"Init\", nbins=50)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distribution of event durations.\n\nevent_time = events_df.Completion - events_df.Init\nfig = px.histogram(x=event_time, nbins=30)\nfig.update_layout(\n    xaxis_title=\"Event duration\",\n    title={\n        'text': \"Distribution of event durations\",\n        'y':0.95,\n        'x':0.5,\n        'xanchor': 'center',\n        'yanchor': 'top'\n    }\n)\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# How the bar chart looks like for categorical field Type.\n\nevents_type_counts = events_df.Type.value_counts()\n\nfig = px.bar(x=events_type_counts.index, y=events_type_counts.values)\nfig.update_layout(xaxis_title=\"Type\", yaxis_title=\"Count\")\nfig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***FEATURE ENGINEERING:***\n\n**ADDING 'StartHesitation', 'Turn', 'Walking' COLUMNS**\n\n**ADDING 'DURATION' COLUMN**\n\n**ADDING 'TARGET' COLUMN**","metadata":{}},{"cell_type":"code","source":"def preprocessing_data(dataset):\n    dataset.drop_duplicates(inplace=True)\n    # Converting categorical features\n    dataset['Type'] = dataset['Type'].astype('category')\n    dataset['StartHesitation'] = np.where(dataset['Type'] == 'StartHesitation', 1, 0) # Where True, yield x=1, otherwise yield y=0.\n    dataset['Turn'] = np.where(dataset['Type'] == 'Turn', 1, 0)\n    dataset['Walking'] = np.where(dataset['Type'] == 'Walking', 1, 0)\n    # Creating a feature called Duration\n    dataset['Duration'] = dataset['Completion'] - dataset['Init']\n    # Defining the value of the target feature\n    dataset['Target'] = np.where(dataset['Type'] == 'StartHesitation', dataset['StartHesitation'], \n                                 np.where(dataset['Type'] == 'Turn', dataset['Turn'], dataset['Walking']))\n    return dataset","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scatter plot matrix\nimport pandas\nimport pandas as pd\nimport matplotlib\nimport matplotlib.pyplot as plt\nfrom pandas.plotting import scatter_matrix\nscatter_matrix(dataset)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_X_Y(dataset):\n    X = dataset.drop(['StartHesitation', 'Turn', 'Walking', 'Target'], axis=1)\n    Y = dataset['Target']\n    return X, Y","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model_evaluation(X, Y, model):\n    score = cross_val_score(model, X, Y, cv=5, scoring='accuracy')\n    return np.mean(score), np.std(score)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def submission_evaluation(y_true, y_pred):\n    acc_score= accuracy_score(y_true, y_pred)\n    con_mat = confusion_matrix(y_true, y_pred)\n    return acc_score, con_mat\ndf = preprocessing_data(dataset)\ndf","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby(['Id','Type'])['Duration'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(df.corr(),annot=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for column in ['Init','Completion']:\n    fig = px.line(events_df, x=\"Duration\", y=column, color_discrete_sequence=['darkslateblue'])\n    fig.update_layout(\n        title={\n            'text': f\"{column} Time Series\",\n            'y':0.95,\n            'x':0.5,\n            'xanchor': 'center',\n            'yanchor': 'top'\n        }\n    )\n    fig.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby(['Id','Type'])['Duration'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import cufflinks as cf\nimport plotly.offline\nfrom plotly.express import scatter_3d,pie,line\nfrom ipywidgets import interact\nimport plotly.express as px\ndef column_histogram(col):\n    fig = px.histogram(df,\n                       x=col,\n                       #nbins=80,\n                       )\n    fig.show()\ncols =  df.columns\n\ninteract(column_histogram,col=cols);","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datetime import datetime\n\n# start and end timestamps\nstart_ts = 11.38470\nend_ts = 41.1847\n\n# convert timestamps to datetime object\ndt1 = datetime.fromtimestamp(start_ts)\nprint('Datetime Start:', dt1)\ndt2 = datetime.fromtimestamp(end_ts)\nprint('Datetime End:', dt2)\n\n# Difference between two timestamps\n# in hours:minutes:seconds format\ndelta = dt2 - dt1\nprint('Difference is:', delta)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import datetime\n\ntimestamp = 11.38470\t41.1847\ndt_object = datetime.datetime.fromtimestamp(timestamp)\n\nprint(\"Timestamp:\", timestamp)\nprint(\"Datetime object:\", dt_object)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, Y_train, Y_test = train_test_split(df.drop('Target', axis=1), df['Target'], test_size=0.3, random_state=40)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a list of numerical and categorical features\nnumeric_features = ['Init', 'Completion', 'Duration', 'StartHesitation','Turn', 'Walking', 'Kinetic']\ncategorical_features = ['Id', 'Type']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating transformers for numerical and categorical features\n\nnumeric_transformer = Pipeline(steps=[\n('imputer', SimpleImputer(strategy='median')),\n('scaler', StandardScaler())\n])\n\ncategorical_transformer = Pipeline(steps=[\n('imputer', SimpleImputer\n(strategy='most_frequent')),\n('onehot', OneHotEncoder(handle_unknown='ignore'))\n])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processor = ColumnTransformer(\ntransformers=[\n('numeric', numeric_transformer, numeric_features),\n('categorical', categorical_transformer, categorical_features)\n])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sin_pipeline = Pipeline(steps=[('Processor', processor),\n('classifier', RandomForestClassifier())])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training the model\nsin_pipeline.fit(X_train, Y_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = sin_pipeline.predict(X_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = pd.DataFrame({'Id': X_test['Id'], 'Predictions': pred})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Accuracy score on test data: {accuracy_score(Y_test, pred)}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_one = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/sample_submission.csv')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_two = pd.DataFrame({'Id': ['003f117e14_' + str(i) for i in range(286370)],\n                               'StartHesitation': 0,\n                               'Turn': 0,\n                               'Walking': 0})\n\nfor i in range(len(pred)):\n    id_str = sub_one['Id'][i]\n    sub_two.loc[sub_two['Id'] == id_str, ['StartHesitation', 'Turn', 'Walking']] = pred[i]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_two.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_one.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.merge(sub_one[['Id']], sub_two, on='Id', how='left')\nsubmission.fillna(0, inplace=True)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('submission.csv')\nprint(submission)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Save the merged dataset to a CSV file\nsubmission.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Saving the model\njoblib.dump(sin_pipeline, 'models.joblib')\n#Loading the model\nload_pipeline = joblib.load('models.joblib')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}