{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":41880,"databundleVersionId":5677426,"sourceType":"competition"}],"dockerImageVersionId":30558,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport plotly.express as px\nimport cufflinks as cf\nfrom plotly.offline import download_plotlyjs,init_notebook_mode,iplot\n%matplotlib inline","metadata":{"execution":{"iopub.status.busy":"2024-01-25T18:18:39.394598Z","iopub.execute_input":"2024-01-25T18:18:39.395236Z","iopub.status.idle":"2024-01-25T18:18:43.415293Z","shell.execute_reply.started":"2024-01-25T18:18:39.395201Z","shell.execute_reply":"2024-01-25T18:18:43.414414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Table of Contents\n+ <a href=\"#1\">Understanding Dataset</a>\n+ <a href=\"#2\">Understanding Freezing of Gait</a>\n+ <a href=\"#3\">Exploratory Data Analysis</a>\n+ <a href=\"#4\">Data Preprocessing</a>\n+ <a href=\"#5\">Creating Model</a>\n+ <a href=\"#6\">Evaluating Model</a>","metadata":{}},{"cell_type":"markdown","source":"\n<a id=\"1\"><h1>Understanding Dataset</h1></a>","metadata":{}},{"cell_type":"markdown","source":"\nWe have two major folders here that we are working on in our train folder <b>tdcsfog</b> and <b>defog</b>. \n\nLets see how both files in these folder looks like..","metadata":{}},{"cell_type":"code","source":"tdcsfog_003f117e14 = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog/003f117e14.csv')\ntdcsfog_003f117e14.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"* **AccV: Vertical**\n* **AccML: Mediolateral**\n* **AccAP: Anteroposterior**","metadata":{}},{"cell_type":"code","source":"tdcsfog_003f117e14.tail()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_003f117e14.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_003f117e14.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_02ea782681 = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog/02ea782681.csv')\ndefog_02ea782681.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_02ea782681.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_02ea782681.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"2\"><h1>Understanding Freezing of Gait</h1></a>","metadata":{}},{"cell_type":"markdown","source":"https://en.wikipedia.org/wiki/Parkinsonian_gait ","metadata":{}},{"cell_type":"markdown","source":"<a id=\"3\"><h1>Exploratory Data Analysis</h1></a>","metadata":{}},{"cell_type":"code","source":"#combining all tdcsfog '.csv' train files\n\ntdcsfog_path= '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/tdcsfog'\ntdcsfog_list= []\n\nfor file_name in os.listdir(tdcsfog_path):\n    if file_name.endswith('.csv'):\n        file_path= os.path.join(tdcsfog_path,file_name)\n        df= pd.read_csv(file_path)\n        df['Time']= df['Time']/(len(df)-1) \n        tdcsfog_list.append(df)\n     \ntdcsfog= pd.concat(tdcsfog_list,axis= 0)\ntdcsfog","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#reducing memory usage of dataset\n\ndef reduce_memory_usage(df):\n    \n    init_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(init_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n#                     if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n#                         df[col] = df[col].astype(np.float16)\n                    if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog = reduce_memory_usage(tdcsfog)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(tdcsfog.corr(),annot= True,cmap='magma')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2= pd.DataFrame(np.concatenate([\n    ['total entries'] * len(tdcsfog),\n    ['Start Hesitation'] *  int(tdcsfog['StartHesitation'].mean() * len(tdcsfog)),\n    ['Turn'] * int(tdcsfog['Turn'].mean() * len(tdcsfog)),\n    ['Walking'] * int(tdcsfog['Walking'].mean() * len(tdcsfog))]),\n    columns= ['Number of 1s']              \n    )\n\nsns.countplot(data= df2, x='Number of 1s')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Majority of target variables are 0**","metadata":{}},{"cell_type":"code","source":"sns.pairplot(tdcsfog[['AccV','AccML','AccAP']])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_means = tdcsfog.groupby('Time').mean().reset_index()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_means['StartHesitation']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_means.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_means[tdcsfog_means['StartHesitation']==1].head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog[tdcsfog['Time']==0.009147]['StartHesitation']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_means['StartHesitation'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#plt.figure(figsize=(12,6))\n#plt.plot(tdcsfog_means['Time'], tdcsfog_means['StartHesitation'], label = 'StartHesitation',color='blue')\n#plt.plot(tdcsfog_means['Time'], tdcsfog_means['StartHesitation'], label = 'Turn',color='green')\n#plt.plot(tdcsfog_means['Time'], tdcsfog_means['StartHesitation'], label = 'Walking',color='red')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plt.figure(figsize=(12,6))\n# plt.plot(tdcsfog_means['Time'], tdcsfog_means['StartHesitation'], label = 'StartHesitation',color='blue')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots(figsize = (10, 6))\n\nax.plot(tdcsfog['Time'], tdcsfog['StartHesitation'], label = 'StartHesitation')\nax.plot(tdcsfog['Time'], tdcsfog['Turn'], label = 'Turn')\nax.plot(tdcsfog['Time'], tdcsfog['Walking'], label = 'Walking')\n\nax.set_xlabel('Time')\nax.set_ylabel('Binary Status(0 or 1)')\nax.set_title('Relationship between Time and Movement Status')\n\nax.legend(loc='upper left',bbox_to_anchor=(1,0.5))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**It can be seen that 'StartHesitation' has mostly occurred during specific times, mainly between 'Time' 0 to 0.1 whereas 'Walking' seems to have a positive correlation with 'Time' as it increases with time. 'Turn' seems to have the least correlation with time among the other 2 features with mostly concentrated below Time= 0.5**","metadata":{}},{"cell_type":"code","source":"defog_path = '/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/train/defog'\n\ndefog_list = []\n\nfor file_name in os.listdir(defog_path):\n    if file_name.endswith('.csv'):\n        file_path = os.path.join(defog_path, file_name)\n        file = pd.read_csv(file_path)\n        file.Time = file.Time / (len(file) - 1)\n        defog_list.append(file)\n\ndefog = pd.concat(defog_list, axis = 0)\n\ndefog","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog= reduce_memory_usage(defog)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"4\"><h1>Data Preprocessing</h1></a>","metadata":{}},{"cell_type":"code","source":"defog= defog[(defog['Valid']==1) & (defog['Task']==1)]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog.dropna()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog= defog.iloc[:,:7]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"merged= pd.concat([tdcsfog,defog],axis=0)\nmerged","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_merged = merged.iloc[:,0:4]  \nX = tdcsfog.iloc[:,0:4]  \ny1 = merged['StartHesitation']  # target variable for StartHesitation\ny2 = merged['Turn']  # target variable for Turn\ny3 = tdcsfog['Walking']  # target variable for Walking","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**As majority of target variables are 0, we create 3 balanced datasets with equal number of 0s and 1s to get better results.**","metadata":{}},{"cell_type":"code","source":"X_merged.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y1.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.where(y1==1)[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y1_ones= np.where(y1==1)[0]\n\nn1_ones= (y1==1).sum()\ny1_zeros= np.random.choice(np.where(y1==0)[0],size= n1_ones,replace= False)\n\ny1_balanced_idx= np.sort(np.concatenate([y1_zeros,y1_ones]))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y1_balanced_idx","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X1_balanced= X_merged.iloc[y1_balanced_idx,:]\ny1_balanced= y1.iloc[y1_balanced_idx]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y2_ones= np.where(y2==1)[0]\n\nn2_ones= (y2==1).sum()\ny2_zeros= np.random.choice(np.where(y2==0)[0],size= n2_ones,replace= False)\n\ny2_balanced_idx= np.sort(np.concatenate([y2_zeros,y2_ones]))\n\nX2_balanced= X_merged.iloc[y2_balanced_idx,:]\ny2_balanced= y2.iloc[y2_balanced_idx]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y3_ones= np.where(y3==1)[0]\n\nn3_ones= (y3==1).sum()\ny3_zeros= np.random.choice(np.where(y3==0)[0],size= n3_ones,replace= False)\n\ny3_balanced_idx= np.sort(np.concatenate([y3_zeros,y3_ones]))\n\nX3_balanced= X.iloc[y3_balanced_idx,:]\ny3_balanced= y3.iloc[y3_balanced_idx]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX1_train, X1_test, y1_train, y1_test = train_test_split(X1_balanced, y1_balanced, test_size = 0.2, random_state = 42)\nX2_train, X2_test, y2_train, y2_test = train_test_split(X2_balanced, y2_balanced, test_size = 0.2, random_state = 42)\nX3_train, X3_test, y3_train, y3_test = train_test_split(X3_balanced, y3_balanced, test_size = 0.2, random_state = 42)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scaler1 = StandardScaler()\nX1_train = scaler1.fit_transform(X1_train)\nX1_test = scaler1.transform(X1_test)\n\nscaler2 = StandardScaler()\nX2_train = scaler2.fit_transform(X2_train)\nX2_test = scaler2.transform(X2_test)\n\nscaler3 = StandardScaler()\nX3_train = scaler3.fit_transform(X3_train)\nX3_test = scaler3.transform(X3_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"5\"><h1>Creating Model</h1></a>","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1 = LogisticRegression()\nmodel2 = LogisticRegression()\nmodel3 = LogisticRegression()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model1.fit(X1_train, y1_train)\nmodel2.fit(X2_train, y2_train)\nmodel3.fit(X3_train, y3_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"6\"><h1>Evaluating Model</h1></a>","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import classification_report,confusion_matrix","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y1_pred = model1.predict(X1_test)\ny2_pred = model2.predict(X2_test)\ny3_pred = model3.predict(X3_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('StartHesitation: \\n',classification_report(y1_test,y1_pred))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Turn: \\n',classification_report(y2_test,y2_pred))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Walking: \\n',classification_report(y3_test,y3_pred))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('StartHesitation: \\n',confusion_matrix(y1_test,y1_pred))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Turn: \\n',confusion_matrix(y2_test,y2_pred))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Walking: \\n',confusion_matrix(y3_test,y3_pred))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import joblib","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# saving models to disk\n# joblib.dump(model1, 'model1.joblib')\n# joblib.dump(model2, 'model2.joblib')\n# joblib.dump(model3, 'model3.joblib')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Thank You**","metadata":{}}]}