{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-23T09:31:42.229112Z","iopub.execute_input":"2023-05-23T09:31:42.229459Z","iopub.status.idle":"2023-05-23T09:31:42.375096Z","shell.execute_reply.started":"2023-05-23T09:31:42.229428Z","shell.execute_reply":"2023-05-23T09:31:42.374186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport os\nimport warnings\nimport optuna\nfrom joblib import load\nimport joblib\n\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, OneHotEncoder\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report\nfrom sklearn.model_selection import train_test_split, cross_val_score\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as mpatches","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:42.376437Z","iopub.execute_input":"2023-05-23T09:31:42.377584Z","iopub.status.idle":"2023-05-23T09:31:44.191402Z","shell.execute_reply.started":"2023-05-23T09:31:42.377549Z","shell.execute_reply":"2023-05-23T09:31:44.190318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"warnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.192580Z","iopub.execute_input":"2023-05-23T09:31:44.192914Z","iopub.status.idle":"2023-05-23T09:31:44.200783Z","shell.execute_reply.started":"2023-05-23T09:31:44.192882Z","shell.execute_reply":"2023-05-23T09:31:44.199899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/daily_metadata.csv')\ndefog_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/defog_metadata.csv')\ntdcsfog_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tdcsfog_metadata.csv')\nevents_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/events.csv')\nsubjects_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/subjects.csv')\ntasks_df = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/tasks.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.203149Z","iopub.execute_input":"2023-05-23T09:31:44.203725Z","iopub.status.idle":"2023-05-23T09:31:44.253409Z","shell.execute_reply.started":"2023-05-23T09:31:44.203695Z","shell.execute_reply":"2023-05-23T09:31:44.252627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"daily_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.254532Z","iopub.execute_input":"2023-05-23T09:31:44.254851Z","iopub.status.idle":"2023-05-23T09:31:44.273657Z","shell.execute_reply.started":"2023-05-23T09:31:44.254819Z","shell.execute_reply":"2023-05-23T09:31:44.272862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"defog_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.277207Z","iopub.execute_input":"2023-05-23T09:31:44.277464Z","iopub.status.idle":"2023-05-23T09:31:44.287848Z","shell.execute_reply.started":"2023-05-23T09:31:44.277442Z","shell.execute_reply":"2023-05-23T09:31:44.286946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tdcsfog_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.289534Z","iopub.execute_input":"2023-05-23T09:31:44.291046Z","iopub.status.idle":"2023-05-23T09:31:44.304207Z","shell.execute_reply.started":"2023-05-23T09:31:44.291007Z","shell.execute_reply":"2023-05-23T09:31:44.303227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"events_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.305572Z","iopub.execute_input":"2023-05-23T09:31:44.305993Z","iopub.status.idle":"2023-05-23T09:31:44.320338Z","shell.execute_reply.started":"2023-05-23T09:31:44.305923Z","shell.execute_reply":"2023-05-23T09:31:44.319383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"subjects_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.325612Z","iopub.execute_input":"2023-05-23T09:31:44.325899Z","iopub.status.idle":"2023-05-23T09:31:44.339137Z","shell.execute_reply.started":"2023-05-23T09:31:44.325876Z","shell.execute_reply":"2023-05-23T09:31:44.338205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tasks_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.344109Z","iopub.execute_input":"2023-05-23T09:31:44.344496Z","iopub.status.idle":"2023-05-23T09:31:44.357943Z","shell.execute_reply.started":"2023-05-23T09:31:44.344463Z","shell.execute_reply":"2023-05-23T09:31:44.357125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = events_df\ndataset.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.359198Z","iopub.execute_input":"2023-05-23T09:31:44.359726Z","iopub.status.idle":"2023-05-23T09:31:44.383665Z","shell.execute_reply.started":"2023-05-23T09:31:44.359695Z","shell.execute_reply":"2023-05-23T09:31:44.382777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocessing_data(dataset):\n    dataset.drop_duplicates(inplace=True)\n    # Converting categorical features\n    dataset['Type'] = dataset['Type'].astype('category')\n    dataset['StartHesitation'] = np.where(dataset['Type'] == 'StartHesitation', 1, 0)\n    dataset['Turn'] = np.where(dataset['Type'] == 'Turn', 1, 0)\n    dataset['Walking'] = np.where(dataset['Type'] == 'Walking', 1, 0)\n    # Creating a feature called Duration\n    dataset['Duration'] = dataset['Completion'] - dataset['Init']\n    # Defining the value of the target feature\n    dataset['Target'] = np.where(dataset['Type'] == 'StartHesitation', dataset['StartHesitation'], \n                                 np.where(dataset['Type'] == 'Turn', dataset['Turn'], dataset['Walking']))\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.384922Z","iopub.execute_input":"2023-05-23T09:31:44.385492Z","iopub.status.idle":"2023-05-23T09:31:44.393639Z","shell.execute_reply.started":"2023-05-23T09:31:44.385460Z","shell.execute_reply":"2023-05-23T09:31:44.392613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_X_Y(dataset):\n    X = dataset.drop(['StartHesitation', 'Turn', 'Walking', 'Target'], axis=1)\n    Y = dataset['Target']\n    return X, Y","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.395639Z","iopub.execute_input":"2023-05-23T09:31:44.395923Z","iopub.status.idle":"2023-05-23T09:31:44.403875Z","shell.execute_reply.started":"2023-05-23T09:31:44.395899Z","shell.execute_reply":"2023-05-23T09:31:44.402766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model_evaluation(X, Y, model):\n    score = cross_val_score(model, X, Y, cv=5, scoring='accuracy')\n    return np.mean(score), np.std(score)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.405359Z","iopub.execute_input":"2023-05-23T09:31:44.406009Z","iopub.status.idle":"2023-05-23T09:31:44.414060Z","shell.execute_reply.started":"2023-05-23T09:31:44.405971Z","shell.execute_reply":"2023-05-23T09:31:44.413328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def submission_evaluation(y_true, y_pred):\n    acc_score= accuracy_score(y_true, y_pred)\n    con_mat = confusion_matrix(y_true, y_pred)\n    return acc_score, con_mat\ndf = preprocessing_data(dataset)\ndf","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.415369Z","iopub.execute_input":"2023-05-23T09:31:44.415928Z","iopub.status.idle":"2023-05-23T09:31:44.448593Z","shell.execute_reply.started":"2023-05-23T09:31:44.415898Z","shell.execute_reply":"2023-05-23T09:31:44.447651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, Y_train, Y_test = train_test_split(df.drop('Target', axis=1), df['Target'], test_size=0.3, random_state=40)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.450051Z","iopub.execute_input":"2023-05-23T09:31:44.450815Z","iopub.status.idle":"2023-05-23T09:31:44.460002Z","shell.execute_reply.started":"2023-05-23T09:31:44.450775Z","shell.execute_reply":"2023-05-23T09:31:44.458954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating a list of numerical and categorical features\nnumeric_features = ['Init', 'Completion', 'Duration', 'StartHesitation','Turn', 'Walking', 'Kinetic']\ncategorical_features = ['Id', 'Type']","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.461304Z","iopub.execute_input":"2023-05-23T09:31:44.462054Z","iopub.status.idle":"2023-05-23T09:31:44.469512Z","shell.execute_reply.started":"2023-05-23T09:31:44.462022Z","shell.execute_reply":"2023-05-23T09:31:44.468701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Creating transformers for numerical and categorical features\n\nnumeric_transformer = Pipeline(steps=[\n('imputer', SimpleImputer(strategy='median')),\n('scaler', StandardScaler())\n])\n\ncategorical_transformer = Pipeline(steps=[\n('imputer', SimpleImputer\n(strategy='most_frequent')),\n('onehot', OneHotEncoder(handle_unknown='ignore'))\n])","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.471007Z","iopub.execute_input":"2023-05-23T09:31:44.471741Z","iopub.status.idle":"2023-05-23T09:31:44.480314Z","shell.execute_reply.started":"2023-05-23T09:31:44.471710Z","shell.execute_reply":"2023-05-23T09:31:44.479548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"processor = ColumnTransformer(\ntransformers=[\n('numeric', numeric_transformer, numeric_features),\n('categorical', categorical_transformer, categorical_features)\n])","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.481774Z","iopub.execute_input":"2023-05-23T09:31:44.482408Z","iopub.status.idle":"2023-05-23T09:31:44.489670Z","shell.execute_reply.started":"2023-05-23T09:31:44.482377Z","shell.execute_reply":"2023-05-23T09:31:44.489003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sin_pipeline = Pipeline(steps=[('Processor', processor),\n('classifier', RandomForestClassifier())])","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.491409Z","iopub.execute_input":"2023-05-23T09:31:44.492112Z","iopub.status.idle":"2023-05-23T09:31:44.499443Z","shell.execute_reply.started":"2023-05-23T09:31:44.492081Z","shell.execute_reply":"2023-05-23T09:31:44.498758Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Training the model\nsin_pipeline.fit(X_train, Y_train)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.500880Z","iopub.execute_input":"2023-05-23T09:31:44.501621Z","iopub.status.idle":"2023-05-23T09:31:44.951703Z","shell.execute_reply.started":"2023-05-23T09:31:44.501588Z","shell.execute_reply":"2023-05-23T09:31:44.950646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = sin_pipeline.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:44.953180Z","iopub.execute_input":"2023-05-23T09:31:44.953776Z","iopub.status.idle":"2023-05-23T09:31:44.998088Z","shell.execute_reply.started":"2023-05-23T09:31:44.953728Z","shell.execute_reply":"2023-05-23T09:31:44.997315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = pd.DataFrame({'Id': X_test['Id'], 'Predictions': pred})","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:45.004119Z","iopub.execute_input":"2023-05-23T09:31:45.004386Z","iopub.status.idle":"2023-05-23T09:31:45.009257Z","shell.execute_reply.started":"2023-05-23T09:31:45.004362Z","shell.execute_reply":"2023-05-23T09:31:45.008227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Accuracy score on test data: {accuracy_score(Y_test, pred)}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:45.010871Z","iopub.execute_input":"2023-05-23T09:31:45.011333Z","iopub.status.idle":"2023-05-23T09:31:45.021578Z","shell.execute_reply.started":"2023-05-23T09:31:45.011297Z","shell.execute_reply":"2023-05-23T09:31:45.020667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_one = pd.read_csv('/kaggle/input/tlvmc-parkinsons-freezing-gait-prediction/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:45.023085Z","iopub.execute_input":"2023-05-23T09:31:45.024123Z","iopub.status.idle":"2023-05-23T09:31:45.271609Z","shell.execute_reply.started":"2023-05-23T09:31:45.024055Z","shell.execute_reply":"2023-05-23T09:31:45.270517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_two = pd.DataFrame({'Id': ['003f117e14_' + str(i) for i in range(286370)],\n                               'StartHesitation': 0,\n                               'Turn': 0,\n                               'Walking': 0})\n\nfor i in range(len(pred)):\n    id_str = sub_one['Id'][i]\n    sub_two.loc[sub_two['Id'] == id_str, ['StartHesitation', 'Turn', 'Walking']] = pred[i]","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:31:45.273610Z","iopub.execute_input":"2023-05-23T09:31:45.274359Z","iopub.status.idle":"2023-05-23T09:32:25.328513Z","shell.execute_reply.started":"2023-05-23T09:31:45.274324Z","shell.execute_reply":"2023-05-23T09:32:25.327586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_two.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:32:25.330041Z","iopub.execute_input":"2023-05-23T09:32:25.330434Z","iopub.status.idle":"2023-05-23T09:32:25.419172Z","shell.execute_reply.started":"2023-05-23T09:32:25.330398Z","shell.execute_reply":"2023-05-23T09:32:25.418208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_one.info()","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:32:25.420624Z","iopub.execute_input":"2023-05-23T09:32:25.421224Z","iopub.status.idle":"2023-05-23T09:32:25.514869Z","shell.execute_reply.started":"2023-05-23T09:32:25.421187Z","shell.execute_reply":"2023-05-23T09:32:25.512812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.merge(sub_one[['Id']], sub_two, on='Id', how='left')\nsubmission.fillna(0, inplace=True)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:32:25.516419Z","iopub.execute_input":"2023-05-23T09:32:25.516739Z","iopub.status.idle":"2023-05-23T09:32:27.028113Z","shell.execute_reply.started":"2023-05-23T09:32:25.516707Z","shell.execute_reply":"2023-05-23T09:32:27.027159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('submission.csv')\nprint(submission)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:33:19.037319Z","iopub.execute_input":"2023-05-23T09:33:19.038210Z","iopub.status.idle":"2023-05-23T09:33:19.247848Z","shell.execute_reply.started":"2023-05-23T09:33:19.038162Z","shell.execute_reply":"2023-05-23T09:33:19.246685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Save the merged dataset to a CSV file\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:33:23.255591Z","iopub.execute_input":"2023-05-23T09:33:23.255949Z","iopub.status.idle":"2023-05-23T09:33:24.303332Z","shell.execute_reply.started":"2023-05-23T09:33:23.255919Z","shell.execute_reply":"2023-05-23T09:33:24.302137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Saving the model\njoblib.dump(sin_pipeline, 'models.joblib')\n#Loading the model\nload_pipeline = joblib.load('models.joblib')","metadata":{"execution":{"iopub.status.busy":"2023-05-23T09:33:27.485798Z","iopub.execute_input":"2023-05-23T09:33:27.486765Z","iopub.status.idle":"2023-05-23T09:33:27.604293Z","shell.execute_reply.started":"2023-05-23T09:33:27.486719Z","shell.execute_reply":"2023-05-23T09:33:27.603412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}