{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Plotting Functions\nimport matplotlib.pyplot as plt\n\n\n!pip install seaborn --upgrade #Update Seaborn for Plotting\nimport seaborn as sns\nsns.set_style('darkgrid')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Loading Datasets**","metadata":{}},{"cell_type":"code","source":"#loading daily_metadatasets \ndaily_data = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/daily_metadata.csv')\ndaily_data.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Importing dependencies**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import svm\nfrom sklearn.metrics import accuracy_score","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_data.tail()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# number of rows and Columns in this dataset\ndaily_data.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# getting the statistical measures of the data\ndaily_data.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_data.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_data['Visit'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_data.groupby('Visit').mean()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# separating the data and labels\nX = daily_data.drop(columns = 'Visit', axis=1)\nY = daily_data['Visit']","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(Y)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train Test Split Datasets**","metadata":{}},{"cell_type":"code","source":"X_train, X_test, Y_train, Y_test = train_test_split(X,Y, test_size = 0.2, stratify=Y, random_state=2)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(X.shape, X_train.shape, X_test.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Training Model**","metadata":{}},{"cell_type":"code","source":"classifier = svm.SVC(kernel='linear')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check the data types\ndaily_data.dtypes","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# generate df with 1 col and 4 rows\ndata = {\n    \"dataset\": [\"Id\", \"Subject\", \"Beginning of recording\"]\n}\n\n# show head\ndaily_data = pd.DataFrame(data)\ndaily_data.head()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# apply get_dummies function\ndf_encoded = pd.get_dummies(daily_data[\"dataset\"])\ndf_encoded .head()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Converting categorical to numerical\nimport pandas as pd\n\n# generate df with 1 col and 4 rows\ndata = {\n    \"dataset\": [\"Id\", \"Subject\", \"Beginning of recording\"]\n}\n\n# one-hot-encode using sklearn\nfrom sklearn.preprocessing import OneHotEncoder\nencoder = OneHotEncoder()\nencoded_results = encoder.fit_transform(daily_data).toarray()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoded_results","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_data = encoder.fit_transform(X)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x_train.shape)\nx_train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train & Test Datasets**","metadata":{}},{"cell_type":"code","source":"x_train,x_test,Y_train,Y_test=train_test_split(df,Y,test_size=0.25)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Training Model**","metadata":{}},{"cell_type":"code","source":"classifier.fit(x_train,Y_train,)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model Evaluation and Accuracy**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# accuracy on training data\nX_train_prediction = classifier.predict(x_train)\ntraining_data_accuracy = accuracy_score(X_train_prediction, Y_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Accuracy on Training data : ', training_data_accuracy)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# accuracy score on the test data\nX_test_prediction = classifier.predict(x_test)\ntest_data_accuracy = accuracy_score(X_test_prediction, Y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Accuracy on Testing data: ', test_data_accuracy)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Detective System**","metadata":{}},{"cell_type":"code","source":"#input_data you will have to put inputs in the open and closed blackets\ninput_data = ()\n# changing the input_data to numpy array\ninput_data_as_numpy_array = np.asarray(input_data)\n\n# reshape the array as we are predicting for one instance\ninput_data_reshaped = input_data_as_numpy_array.reshape(1,-1)\n\nprediction = classifier.predict(input_data_reshaped)\nprint(prediction)\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**SAVING MODELS**","metadata":{}},{"cell_type":"code","source":"import pickle","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_name = 'daily_metadatasets _model.sav'\npickle.dump(classifier, open(model_name, 'wb'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loading the saved model\nloaded_model = pickle.load(open('daily_metadatasets _model.sav', 'rb'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#loading datasets for defog\ndf = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Loading datasets for events\ndf = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/events.csv')\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Loading datasets for subjects\ndf = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv')\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Loading datasets for tdcsfog_metadata\ndf = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Importing libralies\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.offline as py\nimport plotly.graph_objs as go\nimport plotly.offline as py\nimport plotly.express as px","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Overview of Datasets Summary for tDCS FOG (tdcsfog) dataset**","metadata":{}},{"cell_type":"code","source":"df.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Visualizations**","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pandas.core.apply import SpecificationError\nfrom os import supports_effective_ids\ndef figure(num=None,  # autoincrement if None, else integer from 1-N\n           figsize=None,  # defaults to rc figure.figsize\n           dpi=None,  # defaults to rc figure.dpi\n           facecolor=supports_effective_ids,  # defaults to rc figure.facecolor\n           edgecolor=SpecificationError,  # defaults to rc figure.edgecolor\n           frameon=True,\n          \n           clear=False,\n           **kwargs\n           ):\n  plt.figure(figsize=(10,10), dpi = 300)\n           \n \nsns.pairplot(df)\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(10,10), dpi = 300)\nsns.heatmap(\ndata = df.corr(),\n annot=True,fmt= '0.1g',\n vmin=-1,\n cmap='winter')\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = df.drop(columns=[\"Id\",  \"Subject\" , \"Visit\",\"Test\",\"Medication\"])\ny = df[\"Medication\"]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Scaling tdcsfog**","metadata":{}},{"cell_type":"code","source":"x=df.iloc[:,2:].values\ny=df.iloc[:,1].values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Id'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#mapping categorical values to numerical values\ndf = df.replace('[]')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x=df.iloc[:,2:].values\ny=df.iloc[:,1].values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nsc = StandardScaler()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nmodel = LogisticRegression()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes.sample(5)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"one_hot_encoded_training_predictors = pd.get_dummies(df)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"one_hot_encoded_training_predictors = pd.get_dummies(df)\none_hot_encoded_test_predictors = pd.get_dummies(df)\nfinal_train, final_test = one_hot_encoded_training_predictors.align(one_hot_encoded_test_predictors,\n                                                                    join='left', \n                                                                    axis=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Medication'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Converting Categorical values to numerical values**","metadata":{}},{"cell_type":"code","source":"#mapping categorical values to numerical values\ndf['Medication']=df['Medication'].map({'on':1,'off':0})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**1---->On\n0-----> Off**","metadata":{}},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x=df.iloc[:,0:].values\ny=df.iloc[:,1].values","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#mapping categorical values to numerical values\ndf['Id']=df['Id'].map({'003f117e14':0})","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['Id'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.replace(r'^\\s*$', np.nan, regex=True)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.reset_index(drop = True).head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" df = df.set_index(\"Id\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check the data types\ndf.dtypes","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\n# generate df with 1 col and 4 rows\ndata = {\n    \"dataset\": [\"Id\", \"Subject\", \"Visit\", \"Test\",\"Medication\"]\n}\n\n# show head\ndf = pd.DataFrame(data)\ndf.head()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**One-hot encoding**","metadata":{}},{"cell_type":"code","source":"# apply get_dummies function\ndf_encoded = pd.get_dummies(df[\"dataset\"])\ndf_encoded .head()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**pandas categorical to numeric**","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\n# generate df with 1 col and 4 rows\ndata = {\n    \"dataset\": [\"Id\", \"Subject\", \"Visit\", \"Test\",\"Medication\"]\n}\n\n# one-hot-encode using sklearn\nfrom sklearn.preprocessing import OneHotEncoder\nencoder = OneHotEncoder()\nencoded_results = encoder.fit_transform(df).toarray()\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encoded_results","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = encoder.fit_transform(X)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Selected Model**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Train Test Split Datasets**","metadata":{}},{"cell_type":"code","source":"x_train,x_test,y_train,y_test=train_test_split(df,y,test_size=0.25)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(x_train.shape)\nx_train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Training Model**","metadata":{}},{"cell_type":"code","source":"model.fit(x_train,y_train,)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model Evaluation & Accuracy Score**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# accuracy on training data\nX_train_prediction = model.predict(x_train)\ntraining_data_accuracy = accuracy_score(X_train_prediction, y_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Accuracy on Training data : ', training_data_accuracy)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Logistic Regression Model was selected as high accuracy of 1.0**","metadata":{}},{"cell_type":"markdown","source":"**Let's Save This Model \" Logistic Regression **","metadata":{}},{"cell_type":"code","source":"import pickle","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_logistic_regression = 'tdcsfog_metadata _model.sav'\npickle.dump(classifier, open(model_logistic_regression, 'wb'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loading the saved model\nloaded_model = pickle.load(open('tdcsfog_metadata _model.sav', 'rb'))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Building Predictive or Detecting System**","metadata":{}},{"cell_type":"code","source":"#input_data you will have to put inputs in the open and closed blackets\ninput_data = ()\n\n# changing the input_data to numpy array\ninput_data_as_numpy_array = np.asarray(input_data)\n\n# reshape the array as we are predicting for one instance\ninput_data_reshaped = input_data_as_numpy_array.reshape(1,-1)\n\nprediction = model.predict(input_data_reshaped)\nprint(prediction)\n\nif (prediction[0] == 0):\n  print('stop of each freezing episode ')\nelse:\n  print('start each freezing episode ')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Other Models Evaluation & Accuracy Score against Logistic Regression**","metadata":{}},{"cell_type":"code","source":"X_test_prediction = model.predict(x_test)\ntest_data_accuracy = accuracy_score(X_test_prediction, y_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Accuracy on Test data : ', test_data_accuracy)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#K-Nearest Neighbours(KNN)\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.metrics import classification_report","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"knn_model = KNeighborsClassifier(n_neighbors=3)\nknn_model.fit(x_train, y_train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = knn_model.predict(x_test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(y_test, y_pred))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#support vector classifier\nfrom sklearn.svm import SVC\nsvm_model = SVC()\nlg_model = SVC()\nsvm_model = lg_model.fit(x_train, y_train)\ny_pred = svm_model.predict(x_test)\nprint(classification_report(y_test, y_pred))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"x_train shape :\",x_train.shape)\nprint(\"y_train shape :\",y_train.shape)\nprint(\"x_test shape :\",x_test.shape)\nprint(\"y_test shape :\",y_test.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df= pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Loading datasets for tasks\ndf = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tasks.csv')\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Loading datasets for sample_submission\ndf = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/sample_submission.csv')\ndf.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}