{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Importing required libraries","metadata":{"id":"l2Qtu6ecLioi"}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.model_selection import train_test_split\n","metadata":{"id":"8JHOSVROD2im","execution":{"iopub.status.busy":"2022-08-13T08:26:34.111325Z","iopub.execute_input":"2022-08-13T08:26:34.112052Z","iopub.status.idle":"2022-08-13T08:26:34.641549Z","shell.execute_reply.started":"2022-08-13T08:26:34.111973Z","shell.execute_reply":"2022-08-13T08:26:34.640413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Download the files and unzip three files - gender submission, train and test csv files. For our problem, we will just work on train.csv file.","metadata":{"id":"YwqysTf3LrA3"}},{"cell_type":"markdown","source":"Reading all the files and display the first five records.","metadata":{"id":"Nm3TM4b6MO78"}},{"cell_type":"code","source":"gender_submission = pd.read_csv('../input/titanic/gender_submission.csv')\ntrain_data = pd.read_csv('../input/newtrain-file/train.csv')\ntest_data = pd.read_csv('../input/titanic/test.csv')\n\ngender_submission.head()","metadata":{"id":"I6TK6I5ND8_Q","outputId":"ef1928c5-9b09-4e79-81da-5625d5d3238d","execution":{"iopub.status.busy":"2022-08-13T08:26:34.642836Z","iopub.execute_input":"2022-08-13T08:26:34.643189Z","iopub.status.idle":"2022-08-13T08:26:34.675167Z","shell.execute_reply.started":"2022-08-13T08:26:34.643158Z","shell.execute_reply":"2022-08-13T08:26:34.674076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"id":"zpWoN1o6E93-","outputId":"b8caae84-06ec-457c-cf97-06b6f2c68a1b","execution":{"iopub.status.busy":"2022-08-13T08:26:34.677978Z","iopub.execute_input":"2022-08-13T08:26:34.678633Z","iopub.status.idle":"2022-08-13T08:26:34.695301Z","shell.execute_reply.started":"2022-08-13T08:26:34.678591Z","shell.execute_reply":"2022-08-13T08:26:34.693806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"id":"z20N6x-FFBer","outputId":"75557b6a-d647-48bd-9aec-55042984d9df","execution":{"iopub.status.busy":"2022-08-13T08:26:34.698456Z","iopub.execute_input":"2022-08-13T08:26:34.699396Z","iopub.status.idle":"2022-08-13T08:26:34.716822Z","shell.execute_reply.started":"2022-08-13T08:26:34.699364Z","shell.execute_reply":"2022-08-13T08:26:34.715800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Get the info of all columns of the train dataset.","metadata":{"id":"q0No5Lu5MdS7"}},{"cell_type":"code","source":"train_data.info()","metadata":{"id":"TG51KHL-FkJ-","outputId":"1f269fc7-2a08-4d55-8416-87cca271bc2b","execution":{"iopub.status.busy":"2022-08-13T08:26:34.720143Z","iopub.execute_input":"2022-08-13T08:26:34.720529Z","iopub.status.idle":"2022-08-13T08:26:34.736240Z","shell.execute_reply.started":"2022-08-13T08:26:34.720499Z","shell.execute_reply":"2022-08-13T08:26:34.735057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Describe the train dataset.","metadata":{"id":"cXcl2SUCMuK3"}},{"cell_type":"code","source":"train_data.describe()","metadata":{"id":"X8WRLPOoF0y_","outputId":"72ce53b8-9f03-4bf1-b016-3ae5bead53af","execution":{"iopub.status.busy":"2022-08-13T08:26:34.737323Z","iopub.execute_input":"2022-08-13T08:26:34.737703Z","iopub.status.idle":"2022-08-13T08:26:34.767914Z","shell.execute_reply.started":"2022-08-13T08:26:34.737674Z","shell.execute_reply":"2022-08-13T08:26:34.766912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Consider only all columns except for 'Survived' , 'Name' , 'Ticket' , 'Fare' ,\t'Cabin' and 'Embarked'.","metadata":{"id":"kQTOiXu0MyA_"}},{"cell_type":"code","source":"Y = train_data['Survived']\nX = train_data.drop(['Survived','Name','Ticket','Fare',\t'Cabin','Embarked'], axis=1)\nY = pd.DataFrame(Y)\nY = Y.rename(columns = {0:'Survived'})","metadata":{"id":"uQk5_H2xHhSy","execution":{"iopub.status.busy":"2022-08-13T08:26:34.769270Z","iopub.execute_input":"2022-08-13T08:26:34.769748Z","iopub.status.idle":"2022-08-13T08:26:34.777108Z","shell.execute_reply.started":"2022-08-13T08:26:34.769707Z","shell.execute_reply":"2022-08-13T08:26:34.775920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Convert the string value of gender into numeric","metadata":{"id":"UsicUJV2NB0x"}},{"cell_type":"code","source":"def gender(x):\n  return 1 if x=='male' else 0 \n\nX['Sex'] = X['Sex'].apply(gender)\n\nX","metadata":{"id":"4ufL1-H3k6L6","outputId":"b0e63b9b-e160-41cf-8139-84694b276c0f","execution":{"iopub.status.busy":"2022-08-13T08:26:34.778486Z","iopub.execute_input":"2022-08-13T08:26:34.779506Z","iopub.status.idle":"2022-08-13T08:26:34.799247Z","shell.execute_reply.started":"2022-08-13T08:26:34.779465Z","shell.execute_reply":"2022-08-13T08:26:34.798074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Replace all null or Nan values of Age column with its median.","metadata":{"id":"7tiGphcdNKcS"}},{"cell_type":"code","source":"X['Age'] = X['Age'].fillna(X['Age'].median())","metadata":{"id":"ACAY-TSUl50L","execution":{"iopub.status.busy":"2022-08-13T08:26:34.802068Z","iopub.execute_input":"2022-08-13T08:26:34.803056Z","iopub.status.idle":"2022-08-13T08:26:34.808788Z","shell.execute_reply.started":"2022-08-13T08:26:34.803014Z","shell.execute_reply":"2022-08-13T08:26:34.808052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.info()","metadata":{"id":"gBOGODLDqZEr","outputId":"b2e68026-9e21-4d42-e15a-c6bf3df113cb","execution":{"iopub.status.busy":"2022-08-13T08:26:34.810399Z","iopub.execute_input":"2022-08-13T08:26:34.811161Z","iopub.status.idle":"2022-08-13T08:26:34.826736Z","shell.execute_reply.started":"2022-08-13T08:26:34.811119Z","shell.execute_reply":"2022-08-13T08:26:34.825894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Visualization.","metadata":{"id":"hAQhLlIINWMU"}},{"cell_type":"markdown","source":"Perform data visualization all the columns of data.","metadata":{"id":"Dk5A2fmQNa3r"}},{"cell_type":"code","source":"g = sns.PairGrid(X)\n# g.map_diag(sns.histplot)\n# g.map_offdiag(sns.scatterplot)\n\ng.map_upper(sns.scatterplot)\ng.map_lower(sns.kdeplot)\ng.map_diag(sns.kdeplot, lw=3, legend=False);","metadata":{"id":"W8oQet8Ts-S-","outputId":"6dac7412-7fea-4337-d2df-819953c97917","execution":{"iopub.status.busy":"2022-08-13T08:26:34.827735Z","iopub.execute_input":"2022-08-13T08:26:34.828460Z","iopub.status.idle":"2022-08-13T08:26:47.619614Z","shell.execute_reply.started":"2022-08-13T08:26:34.828422Z","shell.execute_reply":"2022-08-13T08:26:47.618509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Perform pairplot between all the columns.","metadata":{"id":"ZYCH8LKDNvay"}},{"cell_type":"code","source":"sns.pairplot(X, height=2.5,hue=\"Sex\",diag_kind=\"kde\")","metadata":{"id":"uvnuROTr4Cne","outputId":"f7233d3f-39e7-4b93-bd34-bddda7008656","execution":{"iopub.status.busy":"2022-08-13T08:26:47.620925Z","iopub.execute_input":"2022-08-13T08:26:47.621830Z","iopub.status.idle":"2022-08-13T08:26:54.269195Z","shell.execute_reply.started":"2022-08-13T08:26:47.621796Z","shell.execute_reply":"2022-08-13T08:26:54.268265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plot Age against Sex of the passengers.","metadata":{"id":"6fMd0mFRN3VJ"}},{"cell_type":"code","source":"sns.displot(data=X, x=\"Age\", hue=\"Sex\",  col=\"Sex\")","metadata":{"id":"nvsay9eK_phk","outputId":"609c0927-fc08-44bb-ba02-a27e09276b5a","execution":{"iopub.status.busy":"2022-08-13T08:26:54.273348Z","iopub.execute_input":"2022-08-13T08:26:54.273766Z","iopub.status.idle":"2022-08-13T08:26:54.992191Z","shell.execute_reply.started":"2022-08-13T08:26:54.273736Z","shell.execute_reply":"2022-08-13T08:26:54.991276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the above plot, it is clear the there were more males passengers as comapred to female. Also one can notice that majority of male - female were in late twenties.","metadata":{"id":"OUv8RebNOQV1"}},{"cell_type":"code","source":"f, axs = plt.subplots(1, 2, figsize=(8, 4), gridspec_kw=dict(width_ratios=[4, 3]))\nsns.scatterplot(data=X, x=\"PassengerId\", y=\"Age\", hue=\"Sex\", ax=axs[0])\nsns.histplot(data=X, x=\"Age\", hue=\"Age\", shrink=.8, alpha=.8, legend=False, ax=axs[1])\nf.tight_layout()","metadata":{"id":"dcR0vsrGAFUb","outputId":"f9ec2eb2-bdf1-4a38-c0b8-292b9b25784f","execution":{"iopub.status.busy":"2022-08-13T08:26:54.993269Z","iopub.execute_input":"2022-08-13T08:26:54.994045Z","iopub.status.idle":"2022-08-13T08:27:00.876869Z","shell.execute_reply.started":"2022-08-13T08:26:54.994010Z","shell.execute_reply":"2022-08-13T08:27:00.875842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Standized the input datasets.**","metadata":{}},{"cell_type":"code","source":"sc = StandardScaler()\nX = sc.fit_transform(X)","metadata":{"id":"awgICTBUIY7D","execution":{"iopub.status.busy":"2022-08-13T08:27:00.878267Z","iopub.execute_input":"2022-08-13T08:27:00.878578Z","iopub.status.idle":"2022-08-13T08:27:00.886994Z","shell.execute_reply.started":"2022-08-13T08:27:00.878549Z","shell.execute_reply":"2022-08-13T08:27:00.886253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train,X_test, Y_train,Y_test = train_test_split(X, Y ,test_size = 0.2, random_state=13)","metadata":{"id":"3MyB5OnxIuC1","execution":{"iopub.status.busy":"2022-08-13T08:27:00.888256Z","iopub.execute_input":"2022-08-13T08:27:00.888835Z","iopub.status.idle":"2022-08-13T08:27:00.900313Z","shell.execute_reply.started":"2022-08-13T08:27:00.888805Z","shell.execute_reply":"2022-08-13T08:27:00.899539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Logistic regression","metadata":{"id":"wlI5QQpnI969"}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nlog_clf = LogisticRegression().fit(X_train, Y_train)\npredictions = log_clf.predict(X_test)","metadata":{"id":"qGjX8yHYI_Ax","outputId":"ad8ced78-4b89-4b04-d68e-26649c41229c","execution":{"iopub.status.busy":"2022-08-13T08:27:00.901375Z","iopub.execute_input":"2022-08-13T08:27:00.901908Z","iopub.status.idle":"2022-08-13T08:27:00.979656Z","shell.execute_reply.started":"2022-08-13T08:27:00.901870Z","shell.execute_reply":"2022-08-13T08:27:00.978824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\nprint(classification_report(Y_test, predictions))","metadata":{"id":"a8vBq9CzJOEU","outputId":"3a35fb6c-3a98-4952-d357-d5e045d43f92","execution":{"iopub.status.busy":"2022-08-13T08:27:00.981009Z","iopub.execute_input":"2022-08-13T08:27:00.981576Z","iopub.status.idle":"2022-08-13T08:27:00.990513Z","shell.execute_reply.started":"2022-08-13T08:27:00.981544Z","shell.execute_reply":"2022-08-13T08:27:00.989720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The logistic regression shows the accuracy of 82%.","metadata":{}},{"cell_type":"markdown","source":"## K-Nearest Neighbour (KNN)\n\nLet’s now train K-Nearest Neighbour on the same data:","metadata":{"id":"qY7CJBxkJVBW"}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nneigh = KNeighborsClassifier()\nneigh.fit(X_train, Y_train)\npredictions = neigh.predict(X_test)","metadata":{"id":"jcVoqzDAJawr","outputId":"709f0695-82d0-49ec-8767-c94d59da33e7","execution":{"iopub.status.busy":"2022-08-13T08:27:00.992153Z","iopub.execute_input":"2022-08-13T08:27:00.992933Z","iopub.status.idle":"2022-08-13T08:27:01.051602Z","shell.execute_reply.started":"2022-08-13T08:27:00.992891Z","shell.execute_reply":"2022-08-13T08:27:01.050480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(Y_test, predictions))","metadata":{"id":"V3BJ_GRWJf-Z","outputId":"7a62cf46-b3d0-4e28-bd17-d16139422296","execution":{"iopub.status.busy":"2022-08-13T08:27:01.053034Z","iopub.execute_input":"2022-08-13T08:27:01.053470Z","iopub.status.idle":"2022-08-13T08:27:01.067083Z","shell.execute_reply.started":"2022-08-13T08:27:01.053427Z","shell.execute_reply":"2022-08-13T08:27:01.065923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"KNN with default values seems to work slightly worse than the logistic regression. The accuracy went down from 0.82 to 0.8 and average recall, precision, and f-score seem to be lower as well.","metadata":{"id":"WZBiVP_-Jl5s"}},{"cell_type":"markdown","source":"## Decision tree\n\nLet’s have a look at another classification algorithm","metadata":{"id":"TdKS0teXJsf8"}},{"cell_type":"code","source":"from sklearn.tree import DecisionTreeClassifier\ndtc_clf = DecisionTreeClassifier(random_state=0)\ndtc_clf.fit(X_train, Y_train)\npredictions = dtc_clf.predict(X_test)","metadata":{"id":"mUTcw3AaJuQE","execution":{"iopub.status.busy":"2022-08-13T08:27:01.068661Z","iopub.execute_input":"2022-08-13T08:27:01.069349Z","iopub.status.idle":"2022-08-13T08:27:01.102794Z","shell.execute_reply.started":"2022-08-13T08:27:01.069305Z","shell.execute_reply":"2022-08-13T08:27:01.101978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(Y_test, predictions))","metadata":{"id":"7O4qrNFxJ3qy","outputId":"b8ac8055-6dde-4812-c231-eb540184b8e2","execution":{"iopub.status.busy":"2022-08-13T08:27:01.104244Z","iopub.execute_input":"2022-08-13T08:27:01.104933Z","iopub.status.idle":"2022-08-13T08:27:01.114725Z","shell.execute_reply.started":"2022-08-13T08:27:01.104891Z","shell.execute_reply":"2022-08-13T08:27:01.113733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The accuracy went down from 82% to 77%.","metadata":{}},{"cell_type":"markdown","source":"## Random Forrest\n\nLet’s try to call the random forest classifier with its default parameters.\n\n","metadata":{"id":"uDpE5NMAKKdx"}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nrfc_clf = RandomForestClassifier(random_state=0)\nrfc_clf.fit(X_train, Y_train)\npredictions = rfc_clf.predict(X_test)","metadata":{"id":"OFqdb5OOKLhN","outputId":"5dd27827-e337-47d4-9844-2269b8b08266","execution":{"iopub.status.busy":"2022-08-13T08:27:01.115970Z","iopub.execute_input":"2022-08-13T08:27:01.116632Z","iopub.status.idle":"2022-08-13T08:27:01.389787Z","shell.execute_reply.started":"2022-08-13T08:27:01.116591Z","shell.execute_reply":"2022-08-13T08:27:01.388161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(Y_test, predictions))","metadata":{"id":"_fRwcXL5KUT1","outputId":"b6912745-cd21-4c20-b709-bfe9d3c88fc0","execution":{"iopub.status.busy":"2022-08-13T08:27:01.391096Z","iopub.execute_input":"2022-08-13T08:27:01.392167Z","iopub.status.idle":"2022-08-13T08:27:01.404294Z","shell.execute_reply.started":"2022-08-13T08:27:01.392120Z","shell.execute_reply":"2022-08-13T08:27:01.403401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have an accuracy of 83% here. ","metadata":{"id":"3LB04VG_KZvY"}},{"cell_type":"markdown","source":"## Gradient boosting\n\nIt is another tree style algorithm and it has been very effective for many machine learning problems.","metadata":{"id":"fuA6ZtlrKfNF"}},{"cell_type":"code","source":"from sklearn.ensemble import GradientBoostingClassifier\ngb_clf = GradientBoostingClassifier(random_state=0)\ngb_clf.fit(X_train, Y_train)\npredictions = gb_clf.predict(X_test)","metadata":{"id":"JoqK-NwGKel5","outputId":"2bc29393-5614-494e-b403-8f0b0a8eed35","execution":{"iopub.status.busy":"2022-08-13T08:27:01.405523Z","iopub.execute_input":"2022-08-13T08:27:01.406181Z","iopub.status.idle":"2022-08-13T08:27:01.526657Z","shell.execute_reply.started":"2022-08-13T08:27:01.406150Z","shell.execute_reply":"2022-08-13T08:27:01.525597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(classification_report(Y_test, predictions))","metadata":{"id":"qdOdDzgUKp18","outputId":"878e5d7e-0ce5-4f5b-99d7-ceb5ded930c4","execution":{"iopub.status.busy":"2022-08-13T08:27:01.528003Z","iopub.execute_input":"2022-08-13T08:27:01.528338Z","iopub.status.idle":"2022-08-13T08:27:01.539155Z","shell.execute_reply.started":"2022-08-13T08:27:01.528308Z","shell.execute_reply":"2022-08-13T08:27:01.538137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the classification report, we observe that Random Forest and Gradient Boosting ","metadata":{"id":"lqyvmQ_gP-5n"}},{"cell_type":"markdown","source":"In this step, we will dump the best of 5 models and load it to predict later.","metadata":{"id":"50gaGbh7OyFc"}},{"cell_type":"code","source":"import pickle\n\n# save the knn_model to disk\nfilename = 'best_clf.sav'\npickle.dump(rfc_clf, open(filename, 'wb'))","metadata":{"id":"wV7-SrxwOwtz","execution":{"iopub.status.busy":"2022-08-13T08:27:01.540303Z","iopub.execute_input":"2022-08-13T08:27:01.541298Z","iopub.status.idle":"2022-08-13T08:27:01.556636Z","shell.execute_reply.started":"2022-08-13T08:27:01.541258Z","shell.execute_reply":"2022-08-13T08:27:01.555779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the model from disk\nfilename = 'best_clf.sav'\nbest_clf_reloaded = pickle.load(open(filename, 'rb'))\n\npred = best_clf_reloaded.predict(X_test) \n\nprint(pred)","metadata":{"id":"F9gz0FibQrUn","outputId":"bcfe181d-4a46-4eb5-82ac-1f1f45798d84","execution":{"iopub.status.busy":"2022-08-13T08:27:01.557891Z","iopub.execute_input":"2022-08-13T08:27:01.558215Z","iopub.status.idle":"2022-08-13T08:27:01.589242Z","shell.execute_reply.started":"2022-08-13T08:27:01.558187Z","shell.execute_reply":"2022-08-13T08:27:01.588490Z"},"trusted":true},"execution_count":null,"outputs":[]}]}