{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session ","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":1.040742,"end_time":"2022-07-25T10:32:17.398538","exception":false,"start_time":"2022-07-25T10:32:16.357796","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:40.302931Z","iopub.execute_input":"2022-07-29T04:58:40.303469Z","iopub.status.idle":"2022-07-29T04:58:40.939106Z","shell.execute_reply.started":"2022-07-29T04:58:40.303358Z","shell.execute_reply":"2022-07-29T04:58:40.937785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load titanic data set from csv\ntitanic_df = pd.read_csv('../input/titanic/train.csv')\ntitanic_df.head(3)","metadata":{"papermill":{"duration":0.049605,"end_time":"2022-07-25T10:32:17.461080","exception":false,"start_time":"2022-07-25T10:32:17.411475","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:40.941324Z","iopub.execute_input":"2022-07-29T04:58:40.942004Z","iopub.status.idle":"2022-07-29T04:58:40.981331Z","shell.execute_reply.started":"2022-07-29T04:58:40.941956Z","shell.execute_reply":"2022-07-29T04:58:40.980464Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display train datasets summary info.\nprint(titanic_df.info())","metadata":{"papermill":{"duration":0.03844,"end_time":"2022-07-25T10:32:17.511894","exception":false,"start_time":"2022-07-25T10:32:17.473454","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:40.984652Z","iopub.execute_input":"2022-07-29T04:58:40.984993Z","iopub.status.idle":"2022-07-29T04:58:41.009493Z","shell.execute_reply.started":"2022-07-29T04:58:40.984962Z","shell.execute_reply":"2022-07-29T04:58:41.008442Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Remove all na value from test set \ntitanic_df['Age'].fillna(titanic_df['Age'].mean(), inplace = True)\ntitanic_df['Cabin'].fillna('N', inplace = True)\ntitanic_df['Embarked'].fillna('N', inplace = True)\n\nprint(\"=== Null counts ===\", titanic_df.isnull().sum().sum())","metadata":{"papermill":{"duration":0.025287,"end_time":"2022-07-25T10:32:17.550484","exception":false,"start_time":"2022-07-25T10:32:17.525197","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:41.012272Z","iopub.execute_input":"2022-07-29T04:58:41.012628Z","iopub.status.idle":"2022-07-29T04:58:41.023620Z","shell.execute_reply.started":"2022-07-29T04:58:41.012597Z","shell.execute_reply":"2022-07-29T04:58:41.022375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Distribution for Sex : \\n\", titanic_df['Sex'].value_counts())\nprint(\"Distribution for Cabin : \\n\", titanic_df['Cabin'].value_counts())\nprint(\"Distribution for Embarked : \\n\", titanic_df['Embarked'].value_counts())\n","metadata":{"papermill":{"duration":0.024889,"end_time":"2022-07-25T10:32:17.587746","exception":false,"start_time":"2022-07-25T10:32:17.562857","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:41.024988Z","iopub.execute_input":"2022-07-29T04:58:41.025848Z","iopub.status.idle":"2022-07-29T04:58:41.038761Z","shell.execute_reply.started":"2022-07-29T04:58:41.025790Z","shell.execute_reply":"2022-07-29T04:58:41.037935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Titanic Cabins layout \n![Titanic cabins layout image](https://storage.googleapis.com/kagglesdsdata/datasets/2356904/3971424/titanic_cabin_layout.jpg?X-Goog-Algorithm=GOOG4-RSA-SHA256&X-Goog-Credential=databundle-worker-v2%40kaggle-161607.iam.gserviceaccount.com%2F20220801%2Fauto%2Fstorage%2Fgoog4_request&X-Goog-Date=20220801T094615Z&X-Goog-Expires=345599&X-Goog-SignedHeaders=host&X-Goog-Signature=252a10ebf6eb68f73561b578e503229e4bcdb76f0a7a607c2ab4b42f2a15ce061cc2e640275d4150241afaf542603556b73f7bc8fe5c157722a5727cb6278c1faa33aef105973fc00284855aacaa4eb9652228e9a5ddef79b8fa373a53f3a620d48c883b64876e3a09009fd0184e4b09cee352b20e761f3227ea942220d6d37fecb0471eecb9d6bc009b4a5f2e785952e14ba7f649fca58c296b95a7521e53d81c956802cb79bb91f31851954388e433df5c75568843a08be0176aaf787756f80ac550408390668f5e5fb177d284117b5548ff1f50dba4673dacc277c783964e11b985468b7b7f14f60821af8834ad10588ae339fdb3dc95999ca124a74c6792)\n\nIn the event of a disaster, you can expect more high-rise cabin users to survive.\n\nYou can classify customers by floor using the capital letters(first letters) of cabin tiket.\nSo, now extract the first letter of cabin ","metadata":{"papermill":{"duration":0.011622,"end_time":"2022-07-25T10:32:17.611681","exception":false,"start_time":"2022-07-25T10:32:17.600059","status":"completed"},"tags":[]}},{"cell_type":"code","source":"titanic_df['Cabin'] = titanic_df['Cabin'].str[:1]\nprint(titanic_df['Cabin'].value_counts())\n","metadata":{"papermill":{"duration":0.023102,"end_time":"2022-07-25T10:32:17.646561","exception":false,"start_time":"2022-07-25T10:32:17.623459","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:41.040044Z","iopub.execute_input":"2022-07-29T04:58:41.040902Z","iopub.status.idle":"2022-07-29T04:58:41.053361Z","shell.execute_reply.started":"2022-07-29T04:58:41.040872Z","shell.execute_reply":"2022-07-29T04:58:41.052076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# group by sex, survived\n\ntitanic_df.groupby(['Sex','Survived'])['Survived'].count()","metadata":{"papermill":{"duration":0.024423,"end_time":"2022-07-25T10:32:17.682775","exception":false,"start_time":"2022-07-25T10:32:17.658352","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:41.055217Z","iopub.execute_input":"2022-07-29T04:58:41.056089Z","iopub.status.idle":"2022-07-29T04:58:41.069461Z","shell.execute_reply.started":"2022-07-29T04:58:41.056045Z","shell.execute_reply":"2022-07-29T04:58:41.068127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.barplot(x='Sex', y = 'Survived' , data=titanic_df)","metadata":{"papermill":{"duration":0.292587,"end_time":"2022-07-25T10:32:17.987478","exception":false,"start_time":"2022-07-25T10:32:17.694891","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:41.071198Z","iopub.execute_input":"2022-07-29T04:58:41.071666Z","iopub.status.idle":"2022-07-29T04:58:41.284950Z","shell.execute_reply.started":"2022-07-29T04:58:41.071613Z","shell.execute_reply":"2022-07-29T04:58:41.283734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This graph shows the female are more survived than the male. <p>\nFor each room, we add the room class to the graph to show the survival proabilities by gender. ","metadata":{"papermill":{"duration":0.012482,"end_time":"2022-07-25T10:32:18.013177","exception":false,"start_time":"2022-07-25T10:32:18.000695","status":"completed"},"tags":[]}},{"cell_type":"code","source":"sns.barplot(x='Pclass', y='Survived', hue='Sex', data=titanic_df)","metadata":{"papermill":{"duration":0.388476,"end_time":"2022-07-25T10:32:18.414323","exception":false,"start_time":"2022-07-25T10:32:18.025847","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:41.286863Z","iopub.execute_input":"2022-07-29T04:58:41.287291Z","iopub.status.idle":"2022-07-29T04:58:41.565027Z","shell.execute_reply.started":"2022-07-29T04:58:41.287251Z","shell.execute_reply":"2022-07-29T04:58:41.563909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next we classify the survival proabilities by age","metadata":{"papermill":{"duration":0.012873,"end_time":"2022-07-25T10:32:18.440353","exception":false,"start_time":"2022-07-25T10:32:18.427480","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def get_category(age) :\n    cat = ''\n    if age <= -1 : cat = 'Unknown'\n    elif age <= 5 : cat = 'Baby'\n    elif age <= 12 : cat = 'Child'\n    elif age <= 18 : cat = 'Teenager'\n    elif age <= 25 : cat = 'Student'\n    elif age <= 35 : cat = 'Young Adult'\n    elif age <= 60 : cat = 'Adult'\n    else : cat = 'Elderly'\n        \n    return cat\n\nplt.figure(figsize=(10,6))\n\ngroup_names = ['Unknown','Baby','Child', 'Teenager','Student','Young Adult','Adult','Elderly']\n\n# Use lambda for converting age\ntitanic_df['Age_cat'] = titanic_df['Age'].apply(lambda x : get_category(x))\nsns.barplot( x = 'Age_cat', y = 'Survived', hue = 'Sex', data=titanic_df, order=group_names)\ntitanic_df.drop('Age_cat', axis=1, inplace = True)","metadata":{"papermill":{"duration":0.735376,"end_time":"2022-07-25T10:32:19.188823","exception":false,"start_time":"2022-07-25T10:32:18.453447","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:41.569568Z","iopub.execute_input":"2022-07-29T04:58:41.571900Z","iopub.status.idle":"2022-07-29T04:58:42.187417Z","shell.execute_reply.started":"2022-07-29T04:58:41.571860Z","shell.execute_reply":"2022-07-29T04:58:42.186226Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Convert string category data type to integer category type using sklearn package.","metadata":{"papermill":{"duration":0.013091,"end_time":"2022-07-25T10:32:19.215322","exception":false,"start_time":"2022-07-25T10:32:19.202231","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from sklearn import preprocessing\n\ndef encode_features(dataDF) :\n    features = ['Cabin','Sex','Embarked']\n    \n    for feature in features :\n        le = preprocessing.LabelEncoder()\n        le = le.fit(dataDF[feature])\n        dataDF[feature] = le.transform(dataDF[feature])\n        \n    return dataDF\n\ntitanic_df = encode_features(titanic_df)\ntitanic_df.head()\n","metadata":{"papermill":{"duration":0.167083,"end_time":"2022-07-25T10:32:19.395336","exception":false,"start_time":"2022-07-25T10:32:19.228253","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:42.188593Z","iopub.execute_input":"2022-07-29T04:58:42.188886Z","iopub.status.idle":"2022-07-29T04:58:42.271082Z","shell.execute_reply.started":"2022-07-29T04:58:42.188860Z","shell.execute_reply":"2022-07-29T04:58:42.269886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop the unnecessary field (PassengerId, Name, Ticket)\ntitanic_df.drop(['PassengerId', 'Name','Ticket'], axis = 1, inplace = True)","metadata":{"papermill":{"duration":0.021983,"end_time":"2022-07-25T10:32:19.431409","exception":false,"start_time":"2022-07-25T10:32:19.409426","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:42.272513Z","iopub.execute_input":"2022-07-29T04:58:42.273567Z","iopub.status.idle":"2022-07-29T04:58:42.280251Z","shell.execute_reply.started":"2022-07-29T04:58:42.273534Z","shell.execute_reply":"2022-07-29T04:58:42.279045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_df.head()","metadata":{"papermill":{"duration":0.029987,"end_time":"2022-07-25T10:32:19.475030","exception":false,"start_time":"2022-07-25T10:32:19.445043","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:42.281698Z","iopub.execute_input":"2022-07-29T04:58:42.282025Z","iopub.status.idle":"2022-07-29T04:58:42.302842Z","shell.execute_reply.started":"2022-07-29T04:58:42.281998Z","shell.execute_reply":"2022-07-29T04:58:42.301498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_titanic_df = titanic_df['Survived']\nX_titanic_df = titanic_df.drop('Survived', axis = 1)","metadata":{"papermill":{"duration":0.022892,"end_time":"2022-07-25T10:32:19.511959","exception":false,"start_time":"2022-07-25T10:32:19.489067","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:42.304596Z","iopub.execute_input":"2022-07-29T04:58:42.305289Z","iopub.status.idle":"2022-07-29T04:58:42.312512Z","shell.execute_reply.started":"2022-07-29T04:58:42.305248Z","shell.execute_reply":"2022-07-29T04:58:42.311387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Describe data set \ny_titanic_df\nX_titanic_df.info()","metadata":{"papermill":{"duration":0.030761,"end_time":"2022-07-25T10:32:19.556845","exception":false,"start_time":"2022-07-25T10:32:19.526084","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:42.313918Z","iopub.execute_input":"2022-07-29T04:58:42.314385Z","iopub.status.idle":"2022-07-29T04:58:42.333078Z","shell.execute_reply.started":"2022-07-29T04:58:42.314357Z","shell.execute_reply":"2022-07-29T04:58:42.331750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX_train, X_test, y_train, y_test = train_test_split(X_titanic_df, y_titanic_df, test_size=0.2, random_state = 11)","metadata":{"papermill":{"duration":0.074927,"end_time":"2022-07-25T10:32:19.646461","exception":false,"start_time":"2022-07-25T10:32:19.571534","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:42.334331Z","iopub.execute_input":"2022-07-29T04:58:42.334831Z","iopub.status.idle":"2022-07-29T04:58:42.403110Z","shell.execute_reply.started":"2022-07-29T04:58:42.334801Z","shell.execute_reply":"2022-07-29T04:58:42.401772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#1. DecisionTree\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.metrics import accuracy_score\n\n\ndt_clf = DecisionTreeClassifier(random_state = 11)\n\ndt_clf.fit(X_train, y_train)\ndt_pred = dt_clf.predict(X_test)\nprint(\"== DecisionTree Accuracy : {0:.4f}\".format(accuracy_score(y_test, dt_pred)))","metadata":{"papermill":{"duration":0.166269,"end_time":"2022-07-25T10:32:19.826809","exception":false,"start_time":"2022-07-25T10:32:19.660540","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:42.404471Z","iopub.execute_input":"2022-07-29T04:58:42.405203Z","iopub.status.idle":"2022-07-29T04:58:42.583623Z","shell.execute_reply.started":"2022-07-29T04:58:42.405162Z","shell.execute_reply":"2022-07-29T04:58:42.582300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#2. RandomForest \nfrom sklearn.ensemble import RandomForestClassifier\n\nrf_clf = RandomForestClassifier(random_state = 11)\nrf_clf.fit(X_train, y_train)\nrf_pred = rf_clf.predict(X_test)\nprint(\"== RandomForest Accuracy : {0:.4f}\".format(accuracy_score(y_test, rf_pred)))\n\n\n","metadata":{"papermill":{"duration":0.299524,"end_time":"2022-07-25T10:32:20.140729","exception":false,"start_time":"2022-07-25T10:32:19.841205","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:42.585283Z","iopub.execute_input":"2022-07-29T04:58:42.585756Z","iopub.status.idle":"2022-07-29T04:58:42.836201Z","shell.execute_reply.started":"2022-07-29T04:58:42.585710Z","shell.execute_reply":"2022-07-29T04:58:42.834993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#3. LogisticRegression\nfrom sklearn.linear_model import LogisticRegression\n\nlr_clf = LogisticRegression()\n\nlr_clf.fit(X_train, y_train)\nlr_pred = lr_clf.predict(X_test)\nprint(\"== LogisticRegression Accuracy : {0:.4f}\".format(accuracy_score(y_test, lr_pred)))","metadata":{"papermill":{"duration":0.06152,"end_time":"2022-07-25T10:32:20.216951","exception":false,"start_time":"2022-07-25T10:32:20.155431","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:42.837927Z","iopub.execute_input":"2022-07-29T04:58:42.838599Z","iopub.status.idle":"2022-07-29T04:58:42.877540Z","shell.execute_reply.started":"2022-07-29T04:58:42.838557Z","shell.execute_reply":"2022-07-29T04:58:42.876240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Kfold\n","metadata":{"papermill":{"duration":0.013925,"end_time":"2022-07-25T10:32:20.245001","exception":false,"start_time":"2022-07-25T10:32:20.231076","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from sklearn.model_selection import KFold\n\ndef exec_Kfold(clf, folds = 5) :\n    kfold = KFold(n_splits = folds)\n    scores = []\n    \n    for iter_count, (train_index, test_index) in enumerate(kfold.split(X_titanic_df)) :\n        X_train, X_test = X_titanic_df.values[train_index] , X_titanic_df.values[test_index]\n        y_train, y_test = y_titanic_df.values[train_index] , y_titanic_df.values[test_index]\n        \n        clf.fit(X_train, y_train)\n        \n        predictions = clf.predict(X_test)\n        \n        accuracy = accuracy_score(y_test, predictions)\n        scores.append(accuracy)\n        print(\"=== Cross validaty {0} , Accuracy : {1:.4f}\".format(iter_count, accuracy))\n        \n    mean_score = np.mean(scores)\n    print(\"== Mean accuracy : {0:.4f} \".format(mean_score))\n    \n#exec_kfolds\nexec_Kfold(dt_clf, folds = 5)\n","metadata":{"papermill":{"duration":0.040029,"end_time":"2022-07-25T10:32:20.298829","exception":false,"start_time":"2022-07-25T10:32:20.258800","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:42.879230Z","iopub.execute_input":"2022-07-29T04:58:42.879587Z","iopub.status.idle":"2022-07-29T04:58:42.902263Z","shell.execute_reply.started":"2022-07-29T04:58:42.879557Z","shell.execute_reply":"2022-07-29T04:58:42.901392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cross_val_score API\n- Cross_val_score API uses StratifiedKFold ","metadata":{"papermill":{"duration":0.013787,"end_time":"2022-07-25T10:32:20.326373","exception":false,"start_time":"2022-07-25T10:32:20.312586","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\n\nscores = cross_val_score(dt_clf, X_titanic_df, y_titanic_df, cv= 5)\n\nfor iter_count, accuracy in enumerate(scores) :\n     print(\"=== Cross validaty {0} , Accuracy : {1:.4f}\".format(iter_count, accuracy))\n        \nprint(\"== Mean accuracy : {0:.4f} \".format(np.mean(scores)))","metadata":{"papermill":{"duration":0.055457,"end_time":"2022-07-25T10:32:20.396117","exception":false,"start_time":"2022-07-25T10:32:20.340660","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:42.903314Z","iopub.execute_input":"2022-07-29T04:58:42.904091Z","iopub.status.idle":"2022-07-29T04:58:42.945535Z","shell.execute_reply.started":"2022-07-29T04:58:42.904054Z","shell.execute_reply":"2022-07-29T04:58:42.944186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# GridSearchCV for hyper-parameter tunning\n","metadata":{"papermill":{"duration":0.014198,"end_time":"2022-07-25T10:32:20.424772","exception":false,"start_time":"2022-07-25T10:32:20.410574","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\n\nparameters = {'max_depth' : [2,3,5,10]\n             , 'min_samples_split':[2,3,5]\n             , 'min_samples_leaf':[1,5,8] }\n\ngrid_dclf = GridSearchCV(dt_clf, param_grid = parameters, scoring = 'accuracy', cv = 5)\ngrid_dclf.fit(X_train, y_train)\n\nprint(\"GridSearchCV best hyper parameter : \", grid_dclf.best_params_)\nprint(\"GridSearchCV best accuracy : \" , grid_dclf.best_score_)\nbest_dclf = grid_dclf.best_estimator_\n\ndpredictions = best_dclf.predict(X_test)\n\naccuracy = accuracy_score(y_test, dpredictions) \n\nprint(\"Test Set DecisionTreeClassifier accuracy : {0:.4f}\".format(accuracy))\n","metadata":{"papermill":{"duration":0.881545,"end_time":"2022-07-25T10:32:21.320516","exception":false,"start_time":"2022-07-25T10:32:20.438971","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:42.946886Z","iopub.execute_input":"2022-07-29T04:58:42.947333Z","iopub.status.idle":"2022-07-29T04:58:43.836309Z","shell.execute_reply.started":"2022-07-29T04:58:42.947272Z","shell.execute_reply":"2022-07-29T04:58:43.835283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load titanic test data set from csv\ntitanic_test_df = pd.read_csv('../input/titanic/test.csv')\ntitanic_test_df.head(3)","metadata":{"papermill":{"duration":0.040278,"end_time":"2022-07-25T10:32:21.375527","exception":false,"start_time":"2022-07-25T10:32:21.335249","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:43.837673Z","iopub.execute_input":"2022-07-29T04:58:43.838066Z","iopub.status.idle":"2022-07-29T04:58:43.862292Z","shell.execute_reply.started":"2022-07-29T04:58:43.838034Z","shell.execute_reply":"2022-07-29T04:58:43.861229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop table for test set\n# Drop the unnecessary field (PassengerId /*remained for making answer sheet*/ , Name, Ticket)\ntitanic_test_df.drop([ 'Name','Ticket'], axis = 1, inplace = True)","metadata":{"papermill":{"duration":0.027763,"end_time":"2022-07-25T10:32:21.417997","exception":false,"start_time":"2022-07-25T10:32:21.390234","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:43.863812Z","iopub.execute_input":"2022-07-29T04:58:43.864170Z","iopub.status.idle":"2022-07-29T04:58:43.871227Z","shell.execute_reply.started":"2022-07-29T04:58:43.864126Z","shell.execute_reply":"2022-07-29T04:58:43.870019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_test_df.info()","metadata":{"papermill":{"duration":0.028959,"end_time":"2022-07-25T10:32:21.461612","exception":false,"start_time":"2022-07-25T10:32:21.432653","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:43.872447Z","iopub.execute_input":"2022-07-29T04:58:43.872772Z","iopub.status.idle":"2022-07-29T04:58:43.891507Z","shell.execute_reply.started":"2022-07-29T04:58:43.872744Z","shell.execute_reply":"2022-07-29T04:58:43.890301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Replace all na value from test set \ntitanic_test_df['Age'].fillna(titanic_df['Age'].mean(), inplace = True)\ntitanic_test_df['Cabin'].fillna('N', inplace = True)\ntitanic_test_df['Embarked'].fillna('N', inplace = True)\n\nprint(\"=== Null counts ===\", titanic_test_df.isnull().sum().sum())","metadata":{"papermill":{"duration":0.026851,"end_time":"2022-07-25T10:32:21.503403","exception":false,"start_time":"2022-07-25T10:32:21.476552","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:43.893538Z","iopub.execute_input":"2022-07-29T04:58:43.894305Z","iopub.status.idle":"2022-07-29T04:58:43.904764Z","shell.execute_reply.started":"2022-07-29T04:58:43.894261Z","shell.execute_reply":"2022-07-29T04:58:43.903478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_test_df.isna().sum()","metadata":{"papermill":{"duration":0.026635,"end_time":"2022-07-25T10:32:21.544913","exception":false,"start_time":"2022-07-25T10:32:21.518278","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:43.906081Z","iopub.execute_input":"2022-07-29T04:58:43.907275Z","iopub.status.idle":"2022-07-29T04:58:43.916502Z","shell.execute_reply.started":"2022-07-29T04:58:43.907244Z","shell.execute_reply":"2022-07-29T04:58:43.915423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_test_df[titanic_test_df['Fare'].isna() == True]","metadata":{"papermill":{"duration":0.033743,"end_time":"2022-07-25T10:32:21.593799","exception":false,"start_time":"2022-07-25T10:32:21.560056","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:43.922032Z","iopub.execute_input":"2022-07-29T04:58:43.922383Z","iopub.status.idle":"2022-07-29T04:58:43.938457Z","shell.execute_reply.started":"2022-07-29T04:58:43.922353Z","shell.execute_reply":"2022-07-29T04:58:43.937303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fare missing data is Pclass 3 and age is over 60, so i allocate fare 0 (or min)\n\ntitanic_test_df['Fare'].min()\ntitanic_test_df['Fare'].fillna( 0, inplace = True)","metadata":{"papermill":{"duration":0.023328,"end_time":"2022-07-25T10:32:21.632571","exception":false,"start_time":"2022-07-25T10:32:21.609243","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:43.940140Z","iopub.execute_input":"2022-07-29T04:58:43.941217Z","iopub.status.idle":"2022-07-29T04:58:43.947702Z","shell.execute_reply.started":"2022-07-29T04:58:43.941173Z","shell.execute_reply":"2022-07-29T04:58:43.946719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Discover test data set again\nprint(\"=== Null counts ===\", titanic_test_df.isnull().sum().sum())","metadata":{"papermill":{"duration":0.024585,"end_time":"2022-07-25T10:32:21.672130","exception":false,"start_time":"2022-07-25T10:32:21.647545","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:43.949276Z","iopub.execute_input":"2022-07-29T04:58:43.950365Z","iopub.status.idle":"2022-07-29T04:58:43.961232Z","shell.execute_reply.started":"2022-07-29T04:58:43.950319Z","shell.execute_reply":"2022-07-29T04:58:43.960145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert Cabin grade to first letter\ntitanic_test_df['Cabin'] = titanic_test_df['Cabin'].str[:1]\nprint(titanic_test_df['Cabin'].value_counts())","metadata":{"papermill":{"duration":0.025024,"end_time":"2022-07-25T10:32:21.711988","exception":false,"start_time":"2022-07-25T10:32:21.686964","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:43.964776Z","iopub.execute_input":"2022-07-29T04:58:43.965293Z","iopub.status.idle":"2022-07-29T04:58:43.974605Z","shell.execute_reply.started":"2022-07-29T04:58:43.965261Z","shell.execute_reply":"2022-07-29T04:58:43.973295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_df.info()","metadata":{"papermill":{"duration":0.030235,"end_time":"2022-07-25T10:32:21.757430","exception":false,"start_time":"2022-07-25T10:32:21.727195","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:43.975940Z","iopub.execute_input":"2022-07-29T04:58:43.976259Z","iopub.status.idle":"2022-07-29T04:58:43.994352Z","shell.execute_reply.started":"2022-07-29T04:58:43.976230Z","shell.execute_reply":"2022-07-29T04:58:43.993170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_test_df.info()","metadata":{"papermill":{"duration":0.030687,"end_time":"2022-07-25T10:32:21.804801","exception":false,"start_time":"2022-07-25T10:32:21.774114","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:43.996006Z","iopub.execute_input":"2022-07-29T04:58:43.996465Z","iopub.status.idle":"2022-07-29T04:58:44.014374Z","shell.execute_reply.started":"2022-07-29T04:58:43.996420Z","shell.execute_reply":"2022-07-29T04:58:44.012996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 1. drop first row , and store the another array\n\ntitanic_passenger_id = titanic_test_df['PassengerId']\ntitanic_test_df = titanic_test_df.drop('PassengerId',axis = 1)","metadata":{"papermill":{"duration":0.024698,"end_time":"2022-07-25T10:32:21.844782","exception":false,"start_time":"2022-07-25T10:32:21.820084","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:44.017477Z","iopub.execute_input":"2022-07-29T04:58:44.019753Z","iopub.status.idle":"2022-07-29T04:58:44.026881Z","shell.execute_reply.started":"2022-07-29T04:58:44.019699Z","shell.execute_reply":"2022-07-29T04:58:44.025413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_test_df = encode_features(titanic_test_df)\ntitanic_test_df","metadata":{"papermill":{"duration":0.036049,"end_time":"2022-07-25T10:32:21.896268","exception":false,"start_time":"2022-07-25T10:32:21.860219","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:44.028331Z","iopub.execute_input":"2022-07-29T04:58:44.029009Z","iopub.status.idle":"2022-07-29T04:58:44.052370Z","shell.execute_reply.started":"2022-07-29T04:58:44.028973Z","shell.execute_reply":"2022-07-29T04:58:44.050942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict using best_estimator\n\ntitanic_test_df['Survived'] = best_dclf.predict(titanic_test_df)","metadata":{"papermill":{"duration":0.026746,"end_time":"2022-07-25T10:32:21.938858","exception":false,"start_time":"2022-07-25T10:32:21.912112","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:44.053981Z","iopub.execute_input":"2022-07-29T04:58:44.054790Z","iopub.status.idle":"2022-07-29T04:58:44.064145Z","shell.execute_reply.started":"2022-07-29T04:58:44.054754Z","shell.execute_reply":"2022-07-29T04:58:44.063175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_test_df","metadata":{"papermill":{"duration":0.037021,"end_time":"2022-07-25T10:32:21.991689","exception":false,"start_time":"2022-07-25T10:32:21.954668","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:44.065570Z","iopub.execute_input":"2022-07-29T04:58:44.066149Z","iopub.status.idle":"2022-07-29T04:58:44.085853Z","shell.execute_reply.started":"2022-07-29T04:58:44.066114Z","shell.execute_reply":"2022-07-29T04:58:44.084695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_test_df.insert(loc=0, column = 'PassengerId', value=titanic_passenger_id)","metadata":{"papermill":{"duration":0.024541,"end_time":"2022-07-25T10:32:22.032849","exception":false,"start_time":"2022-07-25T10:32:22.008308","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:44.087852Z","iopub.execute_input":"2022-07-29T04:58:44.088761Z","iopub.status.idle":"2022-07-29T04:58:44.096771Z","shell.execute_reply.started":"2022-07-29T04:58:44.088714Z","shell.execute_reply":"2022-07-29T04:58:44.095494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"titanic_test_df","metadata":{"papermill":{"duration":0.038645,"end_time":"2022-07-25T10:32:22.088011","exception":false,"start_time":"2022-07-25T10:32:22.049366","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:44.098222Z","iopub.execute_input":"2022-07-29T04:58:44.099363Z","iopub.status.idle":"2022-07-29T04:58:44.120632Z","shell.execute_reply.started":"2022-07-29T04:58:44.099315Z","shell.execute_reply":"2022-07-29T04:58:44.119207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# csv output\nsubmission_data_set = titanic_test_df[['PassengerId','Survived']]","metadata":{"papermill":{"duration":0.025801,"end_time":"2022-07-25T10:32:22.130756","exception":false,"start_time":"2022-07-25T10:32:22.104955","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:44.122169Z","iopub.execute_input":"2022-07-29T04:58:44.122695Z","iopub.status.idle":"2022-07-29T04:58:44.128370Z","shell.execute_reply.started":"2022-07-29T04:58:44.122662Z","shell.execute_reply":"2022-07-29T04:58:44.127384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_data_set","metadata":{"papermill":{"duration":0.029584,"end_time":"2022-07-25T10:32:22.176732","exception":false,"start_time":"2022-07-25T10:32:22.147148","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:44.129951Z","iopub.execute_input":"2022-07-29T04:58:44.130675Z","iopub.status.idle":"2022-07-29T04:58:44.147270Z","shell.execute_reply.started":"2022-07-29T04:58:44.130630Z","shell.execute_reply":"2022-07-29T04:58:44.146440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nsubmission_data_set.to_csv('submission.csv'\n                          , sep = ','\n                          , na_rep = 'NaN'\n                          , index = False)","metadata":{"papermill":{"duration":0.028119,"end_time":"2022-07-25T10:32:22.221288","exception":false,"start_time":"2022-07-25T10:32:22.193169","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-29T04:58:44.148308Z","iopub.execute_input":"2022-07-29T04:58:44.149058Z","iopub.status.idle":"2022-07-29T04:58:44.157652Z","shell.execute_reply.started":"2022-07-29T04:58:44.149025Z","shell.execute_reply":"2022-07-29T04:58:44.156699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Precision, Recall, F1 score","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import precision_recall_curve\n\n# predict when label value is 1\npred_proba_class1 = lr_clf.predict_proba(X_test)[:, 1]\n\n# insert precision_recall_curve probabilites when both actual data and label data are 1\nprecisions, recalls, thresholds = precision_recall_curve(y_test, pred_proba_class1 )\nprint('Shape of returned classified threshold array : ', thresholds.shape)\n\nthr_index = np.arange(0, thresholds.shape[0], 15)\nprint('Top 10 index of thresholds array :', thr_index)\nprint('Sample 10 thresholds : ', np.round(thresholds[thr_index], 3))\n\nprint('Sample precision value :', np.round(precisions[thr_index],3))\nprint('Sample recall value : ', np.round(recalls[thr_index],3))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:58:44.158688Z","iopub.execute_input":"2022-07-29T04:58:44.159341Z","iopub.status.idle":"2022-07-29T04:58:44.170691Z","shell.execute_reply.started":"2022-07-29T04:58:44.159312Z","shell.execute_reply":"2022-07-29T04:58:44.169509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.ticker as ticker\n\ndef precision_recall_curve_plot(y_test, pred_proba_c1) :\n    precisions, recalls, thresholds = precision_recall_curve(y_test, pred_proba_c1) \n    \n    plt.figure(figsize=(8,6))\n    threshold_boundary = thresholds.shape[0]\n    plt.plot(thresholds, precisions[0:threshold_boundary], linestyle='--', label='precision')\n    plt.plot(thresholds, recalls[0:threshold_boundary], label='recall')\n    \n    start, end = plt.xlim()\n    plt.xticks(np.round(np.arange(start, end, 0.1), 2))\n    \n    plt.xlabel('Threshold value')\n    plt.ylabel('Precision and Recall value')\n    plt.legend()\n    plt.grid()\n    plt.show()\n    \nprecision_recall_curve_plot(y_test, lr_clf.predict_proba(X_test)[:,1])","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:58:44.171751Z","iopub.execute_input":"2022-07-29T04:58:44.172496Z","iopub.status.idle":"2022-07-29T04:58:44.396788Z","shell.execute_reply.started":"2022-07-29T04:58:44.172458Z","shell.execute_reply":"2022-07-29T04:58:44.395761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score\n\nf1 = f1_score(y_test, lr_pred)\nprint('F1 score is : {0:.4f}'.format(f1))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T04:58:44.398081Z","iopub.execute_input":"2022-07-29T04:58:44.398438Z","iopub.status.idle":"2022-07-29T04:58:44.407569Z","shell.execute_reply.started":"2022-07-29T04:58:44.398386Z","shell.execute_reply":"2022-07-29T04:58:44.406436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get a threshold value using roc_curve api in Titanic regression training Case","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import roc_curve\n\n# Proability when label is 1\npred_proba_class1 = lr_clf.predict_proba(X_test)[:,1]\n\nfprs, tprs, thresholds = roc_curve(y_test, pred_proba_class1)\n\n# Set a setp 5, except threshhold[0] because set a value randomly max(prob)+1 -> np.arange to start offset 1\nthr_index = np.arange(1, thresholds.shape[0], 5)\n\nprint('======= thr index 10 arrary : ', thr_index)\nprint('sample 10 indexs thresholds : ', np.round(thresholds[thr_index], 2))\n\nprint('sample thresholds fpr:', np.round(fprs[thr_index], 3))\nprint('sample thresholds tpr:', np.round(tprs[thr_index], 3))","metadata":{"execution":{"iopub.status.busy":"2022-07-29T05:24:09.026391Z","iopub.execute_input":"2022-07-29T05:24:09.026792Z","iopub.status.idle":"2022-07-29T05:24:09.038871Z","shell.execute_reply.started":"2022-07-29T05:24:09.026761Z","shell.execute_reply":"2022-07-29T05:24:09.037827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def roc_curve_plot(y_test, pred_proba_c1) :\n    fprs , tprs, thresholds = roc_curve(y_test, pred_proba_c1) \n    \n    # Draw roc curve by plot\n    plt.plot(fprs, tprs, label='ROC')\n    \n    # Draw straight line \n    plt.plot([0,1], [0,1], 'k--', label='Random')\n    \n    # fpr X axis scale to 0.1\n    start, end = plt.xlim()\n    plt.xticks(np.round(np.arange(start,end,0.1),2))\n    plt.xlim(0,1); plt.ylim(0,1)\n    plt.xlabel('FPR(1-Sensitivity)'); plt.ylabel('TPR(Recall)')\n    plt.legend()\n    \nroc_curve_plot(y_test, pred_proba_class1)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-29T05:25:13.548655Z","iopub.execute_input":"2022-07-29T05:25:13.549643Z","iopub.status.idle":"2022-07-29T05:25:13.713208Z","shell.execute_reply.started":"2022-07-29T05:25:13.549600Z","shell.execute_reply":"2022-07-29T05:25:13.712051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}