{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## Titanic Machine Learning using XG Boost","metadata":{"_uuid":"4acbb0bc-0eec-4a2a-90e7-99ed30c45b8f","_cell_guid":"f62ac15c-52c8-4479-99d0-bc7e758e75f0","trusted":true}},{"cell_type":"markdown","source":"If you happen to find this notebook useful then please do upvote.\nAnd any critcisim or feedback is very apprecitaed","metadata":{"_uuid":"768ca29d-323d-4504-92c6-165baa1f09cf","_cell_guid":"a19acfe1-11d6-41c2-8ad1-8da3903d097a","trusted":true}},{"cell_type":"markdown","source":"### Let us first import all the  dependencies","metadata":{"_uuid":"3ff7b04d-0006-4485-b2aa-8f201de00b71","_cell_guid":"e68af505-e176-4bee-8c95-b11891499c81","trusted":true}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"dda48ea7-bb59-49b2-85fa-dfe0213e0f8c","_cell_guid":"a00738f2-05f6-467b-82d1-d1e7ea1cf819","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:10:57.916825Z","iopub.execute_input":"2022-08-10T11:10:57.917206Z","iopub.status.idle":"2022-08-10T11:10:57.924970Z","shell.execute_reply.started":"2022-08-10T11:10:57.917176Z","shell.execute_reply":"2022-08-10T11:10:57.924044Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, mean_squared_error\n\nfrom xgboost import XGBClassifier\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport matplotlib.pyplot as plt\nfrom matplotlib import rcParams\n# figure size in inches\nrcParams['figure.figsize'] = 11.7,8.27\nimport seaborn as sns\nsns.set_theme(style=\"darkgrid\")\nsns.set_palette('viridis')\nprint(\"Setup Complete\")","metadata":{"_uuid":"c5bed7e2-624a-44e8-a7b2-a0a07d8d026a","_cell_guid":"6e9791f1-391a-40b6-9e57-734ba7d34f40","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:10:57.932068Z","iopub.execute_input":"2022-08-10T11:10:57.932519Z","iopub.status.idle":"2022-08-10T11:10:58.891924Z","shell.execute_reply.started":"2022-08-10T11:10:57.932484Z","shell.execute_reply":"2022-08-10T11:10:58.890670Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/titanic/train.csv')\ndf.head(7)","metadata":{"_uuid":"4e24c79f-d639-4a97-9d70-279d01cf2cd8","_cell_guid":"2419a7b5-fddc-4e51-a002-0b10b82a5882","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:10:58.894253Z","iopub.execute_input":"2022-08-10T11:10:58.895366Z","iopub.status.idle":"2022-08-10T11:10:58.918528Z","shell.execute_reply.started":"2022-08-10T11:10:58.895303Z","shell.execute_reply":"2022-08-10T11:10:58.917046Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"_uuid":"94cc3e9e-62cf-4570-854f-601e1c67d3d3","_cell_guid":"a012ee3f-33d5-4112-98b2-e19245acd2ce","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:10:58.920151Z","iopub.execute_input":"2022-08-10T11:10:58.921112Z","iopub.status.idle":"2022-08-10T11:10:58.965362Z","shell.execute_reply.started":"2022-08-10T11:10:58.921072Z","shell.execute_reply":"2022-08-10T11:10:58.964423Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"###    Let us handle the missing values first so we can propely build a model on it","metadata":{"_uuid":"7433d424-c093-4a88-84b9-7dceed3344d8","_cell_guid":"021d6e88-62eb-43f1-adae-e24571483079","trusted":true}},{"cell_type":"code","source":"# Let's check how many null values\nprint('Sum of NaN values in each column\\n')\nprint(df.isnull().sum())\nprint('_____________________________________')\n\nprint('%age of NaN values in each column\\n')\nprint(df.isnull().sum() / len(df['Age']) * 100)","metadata":{"_uuid":"204f6c45-bfd1-4cd1-b634-5ff49e03c2d9","_cell_guid":"d42f2e74-1cfb-4000-a494-332f8f466ce0","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:10:58.968102Z","iopub.execute_input":"2022-08-10T11:10:58.968485Z","iopub.status.idle":"2022-08-10T11:10:58.980529Z","shell.execute_reply.started":"2022-08-10T11:10:58.968452Z","shell.execute_reply":"2022-08-10T11:10:58.979407Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# We can see that more than 77% of the Cabin column if filled with Nan Values\n\n# df[df['Cabin'] == np.nan].head(7) -- To view the NaN values, you won't se anything cause its not there XD\n\n\n# Let's visualize the NaN values for the normies out there\n\nsns.heatmap(df.isnull().transpose(),cmap=\"viridis\", cbar_kws={\"label\": 'missing Data'})","metadata":{"_uuid":"84fc02c1-5095-46ca-83eb-c790ea6fad13","_cell_guid":"824929c9-79b6-46cc-83fd-4255b4984a88","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:10:58.981862Z","iopub.execute_input":"2022-08-10T11:10:58.982431Z","iopub.status.idle":"2022-08-10T11:10:59.883755Z","shell.execute_reply.started":"2022-08-10T11:10:58.982376Z","shell.execute_reply":"2022-08-10T11:10:59.882513Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let us view the co-relation b/w the fields\nsns.heatmap(df.corr(), cmap=\"viridis\")","metadata":{"_uuid":"bde71227-ee1d-45e4-af3a-41a5195d476c","_cell_guid":"a3df430d-9b2c-4c00-b691-c302f62f03b6","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:10:59.885380Z","iopub.execute_input":"2022-08-10T11:10:59.886403Z","iopub.status.idle":"2022-08-10T11:11:00.193639Z","shell.execute_reply.started":"2022-08-10T11:10:59.886366Z","shell.execute_reply":"2022-08-10T11:11:00.192400Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's drop the Cabin column as it won't be of much help for our model\n\nnewDf = df.drop('Cabin', axis=1)\nnewDf.head(7)","metadata":{"_uuid":"63f5b3ec-4780-44bf-bfda-481556bde46e","_cell_guid":"3a236127-bca2-4232-9d01-976c16bdc38a","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:00.195198Z","iopub.execute_input":"2022-08-10T11:11:00.195919Z","iopub.status.idle":"2022-08-10T11:11:00.215859Z","shell.execute_reply.started":"2022-08-10T11:11:00.195874Z","shell.execute_reply":"2022-08-10T11:11:00.214412Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Old Data frame size : {df.shape} New DataFame Size {newDf.shape}')","metadata":{"_uuid":"a44ed85a-c5e0-489f-ac23-2047ccc8fd39","_cell_guid":"7a4f16dc-1abe-4987-82e7-d25ecae948d6","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:00.217485Z","iopub.execute_input":"2022-08-10T11:11:00.218371Z","iopub.status.idle":"2022-08-10T11:11:00.228931Z","shell.execute_reply.started":"2022-08-10T11:11:00.218315Z","shell.execute_reply":"2022-08-10T11:11:00.227613Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now lets' impute the missing values of the Age column\n# Drop the Rows with respect to the missing values of Cabin column\n\n\nnewDf['Age'] = newDf['Age'].fillna(newDf['Age'].mean())\nnewDf = newDf.dropna(axis=0, how='any')\nprint('No. of Missing values in the newDf DataFrame \\n')\nprint(newDf.isnull().sum())","metadata":{"_uuid":"a1d9b1ee-e29b-43f3-8d95-f41925da9f39","_cell_guid":"9a483289-19c7-4388-a31a-bcc5b1cbb1ee","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:00.231413Z","iopub.execute_input":"2022-08-10T11:11:00.232604Z","iopub.status.idle":"2022-08-10T11:11:00.247905Z","shell.execute_reply.started":"2022-08-10T11:11:00.232561Z","shell.execute_reply":"2022-08-10T11:11:00.246685Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"newDf.info()","metadata":{"_uuid":"67a4e487-00fc-4ab0-aafc-c968301e45bb","_cell_guid":"da1d5e76-d348-424c-8c02-e48a7abef795","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:00.251777Z","iopub.execute_input":"2022-08-10T11:11:00.252328Z","iopub.status.idle":"2022-08-10T11:11:00.269799Z","shell.execute_reply.started":"2022-08-10T11:11:00.252296Z","shell.execute_reply":"2022-08-10T11:11:00.268426Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's us compare the stats b/w the old Data set with missing values,\n# and new DataSet with imputed values\n\nprint(f'Old DataSet Stats \\n {df.describe()}')\nprint('______________________________________')\nprint(f'Old DataSet Stats \\n {newDf.describe()}')","metadata":{"_uuid":"3e6ec707-4c50-4d6d-93f5-ad5cf2facc3e","_cell_guid":"22534fda-1292-4726-9a25-458871cefb7e","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:00.271441Z","iopub.execute_input":"2022-08-10T11:11:00.272207Z","iopub.status.idle":"2022-08-10T11:11:00.325265Z","shell.execute_reply.started":"2022-08-10T11:11:00.272161Z","shell.execute_reply":"2022-08-10T11:11:00.324060Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let us do some visualization to better understand the co-relation of each data column","metadata":{"_uuid":"83b3d27a-c6e2-4712-a999-b1896a6afea7","_cell_guid":"ed094820-9dc5-486c-a326-d2f480315e8a","trusted":true}},{"cell_type":"code","source":"\nsns.pairplot(newDf)","metadata":{"_uuid":"655c7986-a39b-4c2c-8c02-0f3b31dbb1a5","_cell_guid":"60619989-8b1a-4922-92eb-222dbc78f40d","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:00.327033Z","iopub.execute_input":"2022-08-10T11:11:00.327645Z","iopub.status.idle":"2022-08-10T11:11:09.401219Z","shell.execute_reply.started":"2022-08-10T11:11:00.327610Z","shell.execute_reply":"2022-08-10T11:11:09.400316Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now let's see which gender had a higher likelyhood / chance of survival\n\n# def percey(value, length)\n\n# print(f'Total Men who Survived the accident : {newDf['Sex']')\n\n# plt.pie(newDf[newDf['Sex'] and newDf['Survived'] == 1].value_counts(), labels= (newDf['Sex'].unique()), shadow=2.6)\n# newDf['Sex'].value_counts(normalize=True)\nsurvied = newDf[newDf['Survived'] == 1]['Sex'].value_counts()\nvictims = newDf[newDf['Survived'] == 0]['Sex'].value_counts()\n\nplt.pie(survied,labels= ['female', 'male'],\n        shadow=0.6,\n        explode=[0.2, 0],\n        radius=1.2,)\nplt.legend()\nplt.title('Titanic Surviors by Gender')","metadata":{"_uuid":"43f4b5d1-9d9e-4823-b317-7a1e3e5d1c8e","_cell_guid":"17568b7b-dc71-494f-a015-98dc9802c070","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:09.402774Z","iopub.execute_input":"2022-08-10T11:11:09.403433Z","iopub.status.idle":"2022-08-10T11:11:09.624349Z","shell.execute_reply.started":"2022-08-10T11:11:09.403394Z","shell.execute_reply":"2022-08-10T11:11:09.622994Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=newDf, x = 'Survived', hue='Sex',).set(xticklabels = ['Did not Survive', 'Survied'], title = 'Titanic Survival Data')","metadata":{"_uuid":"bd0a4d0d-c049-4e19-88f9-54f95e1cf811","_cell_guid":"b2844490-a6b2-407a-b2ac-abe1ab4940e2","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:09.626045Z","iopub.execute_input":"2022-08-10T11:11:09.627178Z","iopub.status.idle":"2022-08-10T11:11:09.810782Z","shell.execute_reply.started":"2022-08-10T11:11:09.627122Z","shell.execute_reply":"2022-08-10T11:11:09.809462Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(data=newDf, x = 'Sex', y ='Survived').set(title='Co - realation of Survival w.r.t Gender')","metadata":{"_uuid":"72adb9ee-a618-47dc-9dc8-7bdc54389f79","_cell_guid":"9fcfb46c-b832-486c-a247-4af1496d4957","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:09.814528Z","iopub.execute_input":"2022-08-10T11:11:09.815279Z","iopub.status.idle":"2022-08-10T11:11:10.059826Z","shell.execute_reply.started":"2022-08-10T11:11:09.815231Z","shell.execute_reply":"2022-08-10T11:11:10.058413Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(data = newDf, x = 'Age', y = 'Survived').set(title = 'Co - realation of Survival w.r.t Age')\n# We can observe that elderly had a higher chance of survival","metadata":{"_uuid":"28b21083-cc6c-4fe2-a909-549419f7f8ac","_cell_guid":"5f5d0a01-e923-4a4f-9613-649294286341","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:10.061689Z","iopub.execute_input":"2022-08-10T11:11:10.062113Z","iopub.status.idle":"2022-08-10T11:11:11.882016Z","shell.execute_reply.started":"2022-08-10T11:11:10.062074Z","shell.execute_reply":"2022-08-10T11:11:11.880750Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(data=newDf, x = 'Pclass', y = 'Survived')\nplt.figure()\n# Notice here how People in the First class also had a higher chance of survival\nsns.countplot(data=newDf, x ='Pclass', hue='Survived')","metadata":{"_uuid":"f03d61a2-2f68-47c9-a201-dc4e84f429db","_cell_guid":"e4794f6c-b5ca-45c0-a383-1746ad75b472","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:11.883395Z","iopub.execute_input":"2022-08-10T11:11:11.883734Z","iopub.status.idle":"2022-08-10T11:11:12.435254Z","shell.execute_reply.started":"2022-08-10T11:11:11.883703Z","shell.execute_reply":"2022-08-10T11:11:12.433855Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"newDf.info()","metadata":{"_uuid":"b793738d-cdd1-45d4-a7b5-3388e4ba471f","_cell_guid":"c09f068d-2098-4739-979e-dfddb7cf1919","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:12.437108Z","iopub.execute_input":"2022-08-10T11:11:12.438114Z","iopub.status.idle":"2022-08-10T11:11:12.453351Z","shell.execute_reply.started":"2022-08-10T11:11:12.438075Z","shell.execute_reply":"2022-08-10T11:11:12.452373Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(data=newDf, x = 'Fare', y= 'Survived').set(title = 'Co-relatibility of Fare w.r.t Survival')\n\n# The graph is rather confusing so let's leave the Fare aside for now","metadata":{"_uuid":"980c613d-0210-42f2-890a-711766da0fc6","_cell_guid":"ea27692c-28c6-494a-9f15-0760f0c15f1f","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:12.455042Z","iopub.execute_input":"2022-08-10T11:11:12.455798Z","iopub.status.idle":"2022-08-10T11:11:15.934392Z","shell.execute_reply.started":"2022-08-10T11:11:12.455762Z","shell.execute_reply":"2022-08-10T11:11:15.933540Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(data=newDf, x = 'Parch', y= 'Survived')\nplt.figure()\nsns.lineplot(data=newDf, x = 'SibSp', y= 'Survived')","metadata":{"_uuid":"90c1a738-e201-4160-ab58-0de7f6c7b5f8","_cell_guid":"24bed66a-81d6-4dc0-9798-e9ccbad08d46","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:15.935845Z","iopub.execute_input":"2022-08-10T11:11:15.936441Z","iopub.status.idle":"2022-08-10T11:11:16.811811Z","shell.execute_reply.started":"2022-08-10T11:11:15.936408Z","shell.execute_reply":"2022-08-10T11:11:16.810718Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's check for some outlier Data that could hinder the model's predictive performance","metadata":{"_uuid":"550ff9f1-6343-4a3e-b01c-8e7b0011bed8","_cell_guid":"de286739-a07a-4791-bdb8-ed095437e166","trusted":true}},{"cell_type":"code","source":"sns.boxplot(data=newDf)","metadata":{"_uuid":"65414d22-a2cf-48a2-9b62-3ca5219c2dd6","_cell_guid":"8a8fb25c-ed68-439e-9eb6-291bd3df5300","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:16.813410Z","iopub.execute_input":"2022-08-10T11:11:16.814011Z","iopub.status.idle":"2022-08-10T11:11:17.098158Z","shell.execute_reply.started":"2022-08-10T11:11:16.813977Z","shell.execute_reply":"2022-08-10T11:11:17.097302Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check if the Age has any outlier data\nnewDf = newDf[newDf['Age'] < 55]\nsns.boxplot(data=newDf, y = 'Age', x = 'Survived')","metadata":{"_uuid":"bcadd02e-5e26-47af-9f96-f934e3dbe1cf","_cell_guid":"98f42b47-30d8-468f-8f67-e42f2b8a7d3a","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:17.099538Z","iopub.execute_input":"2022-08-10T11:11:17.100091Z","iopub.status.idle":"2022-08-10T11:11:17.300307Z","shell.execute_reply.started":"2022-08-10T11:11:17.100060Z","shell.execute_reply":"2022-08-10T11:11:17.299110Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"newDf = newDf[newDf['Fare'] <55]\nsns.boxplot(data=newDf, x = 'Survived', y = 'Fare')","metadata":{"_uuid":"7fb7ab62-2049-42b4-b551-7e76a20f6307","_cell_guid":"611c42e0-ae65-4a9c-8e6d-e565e00470ad","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:17.301808Z","iopub.execute_input":"2022-08-10T11:11:17.302952Z","iopub.status.idle":"2022-08-10T11:11:17.517155Z","shell.execute_reply.started":"2022-08-10T11:11:17.302913Z","shell.execute_reply":"2022-08-10T11:11:17.515985Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = ['Age', 'Sex', \"Pclass\", \"SibSp\"]\nx = newDf[features]\nx['Sex'] = x['Sex'].map({'female': 0, 'male':1})\ny = newDf['Survived']\n\n# Let's split our dataset\n\nxTrain, xTest, yTrain, yTest = train_test_split(x,y, random_state=200,test_size=0.2)","metadata":{"_uuid":"5c120b06-a4e0-437b-9cd3-13d34c62145f","_cell_guid":"eac8a9a8-9127-4b70-899f-c5040db77844","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:17.518637Z","iopub.execute_input":"2022-08-10T11:11:17.519073Z","iopub.status.idle":"2022-08-10T11:11:17.530200Z","shell.execute_reply.started":"2022-08-10T11:11:17.519042Z","shell.execute_reply":"2022-08-10T11:11:17.528820Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tune the model's parameters for more accurate prediction\n\nmodel = XGBClassifier(booster = 'gbtree', n_estimators=350, learning_rate=0.7, colsample_bytree=0.4)\nmodel.fit(xTrain, yTrain)","metadata":{"_uuid":"1191fc4f-a004-4b4b-978a-1ae0e945e701","_cell_guid":"08082c8e-1e9e-4224-878a-ab262e3b4cfe","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:17.533019Z","iopub.execute_input":"2022-08-10T11:11:17.533561Z","iopub.status.idle":"2022-08-10T11:11:18.383656Z","shell.execute_reply.started":"2022-08-10T11:11:17.533512Z","shell.execute_reply":"2022-08-10T11:11:18.382519Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(xTest)","metadata":{"_uuid":"55104bb7-c2ed-4bb9-be91-8a40a9f4cf9e","_cell_guid":"8e620c45-7ebe-472b-84e8-7e098cd71090","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:18.386317Z","iopub.execute_input":"2022-08-10T11:11:18.386944Z","iopub.status.idle":"2022-08-10T11:11:18.399676Z","shell.execute_reply.started":"2022-08-10T11:11:18.386909Z","shell.execute_reply":"2022-08-10T11:11:18.396143Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Accuracy score on Train Data-Set : {accuracy_score(model.predict(xTrain), yTrain)}')\nprint(f'Accuracy score on Train Data-Set : {accuracy_score(model.predict(xTest), yTest)}')\nprint(f'MSE : {mean_squared_error(preds, yTest)}')","metadata":{"_uuid":"a2c9dd56-e6d2-4ebc-b417-96b6f80e9f08","_cell_guid":"ec387832-4539-4f80-8fd5-d747646661da","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:18.401300Z","iopub.execute_input":"2022-08-10T11:11:18.401920Z","iopub.status.idle":"2022-08-10T11:11:18.423014Z","shell.execute_reply.started":"2022-08-10T11:11:18.401882Z","shell.execute_reply":"2022-08-10T11:11:18.421801Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testDf = pd.read_csv('../input/titanic/test.csv')\ntestDf.head(7)","metadata":{"_uuid":"23f1e840-cfd5-4d9e-94e4-987e48c8e5e5","_cell_guid":"21c42fbc-4e6c-4777-a238-9722c3e3f1d4","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:18.428486Z","iopub.execute_input":"2022-08-10T11:11:18.428850Z","iopub.status.idle":"2022-08-10T11:11:18.454011Z","shell.execute_reply.started":"2022-08-10T11:11:18.428820Z","shell.execute_reply":"2022-08-10T11:11:18.452870Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets filter out to have the important features we have determined earlier\n\nnewTestDf = testDf[features]","metadata":{"_uuid":"b84cae70-02da-4025-aa0a-c3960cafea0a","_cell_guid":"2c77d232-b975-4bd4-bce6-477aba74ce3a","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:18.455310Z","iopub.execute_input":"2022-08-10T11:11:18.456040Z","iopub.status.idle":"2022-08-10T11:11:18.462557Z","shell.execute_reply.started":"2022-08-10T11:11:18.456005Z","shell.execute_reply":"2022-08-10T11:11:18.461149Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for NAN values in the test set\nnewTestDf.isnull().sum()","metadata":{"_uuid":"46dddbea-0e43-4ddf-89c6-e5ab9d20a35d","_cell_guid":"d49b97cb-9a20-42be-954c-7aea19b71e79","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:18.463784Z","iopub.execute_input":"2022-08-10T11:11:18.464130Z","iopub.status.idle":"2022-08-10T11:11:18.480665Z","shell.execute_reply.started":"2022-08-10T11:11:18.464099Z","shell.execute_reply":"2022-08-10T11:11:18.479505Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let us impute the missing values and proceed\nnewTestDf['Age'] = newTestDf['Age'].fillna(newTestDf['Age'].mean())\nnewTestDf['Sex'] = newTestDf['Sex'].map({'female': 0, 'male':1})\nnewTestDf.isnull().sum()","metadata":{"_uuid":"3c490d78-2281-4f77-aeb9-f1a9e5f1af44","_cell_guid":"0d262539-b215-44ae-96d8-2c962b6e2239","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:18.481853Z","iopub.execute_input":"2022-08-10T11:11:18.483049Z","iopub.status.idle":"2022-08-10T11:11:18.497969Z","shell.execute_reply.started":"2022-08-10T11:11:18.483011Z","shell.execute_reply":"2022-08-10T11:11:18.496748Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"newTestDf.head(7)","metadata":{"_uuid":"db72a9ba-576f-4c5a-872a-dd56b62d94da","_cell_guid":"288725b5-97fc-4b76-bbcb-ba696993069f","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:18.499284Z","iopub.execute_input":"2022-08-10T11:11:18.499721Z","iopub.status.idle":"2022-08-10T11:11:18.517828Z","shell.execute_reply.started":"2022-08-10T11:11:18.499687Z","shell.execute_reply":"2022-08-10T11:11:18.516500Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testPreds = model.predict(newTestDf)","metadata":{"_uuid":"427b1923-b691-4582-852c-829251d55744","_cell_guid":"da2440a4-2656-4d99-a3c4-53833bed7a06","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:18.519534Z","iopub.execute_input":"2022-08-10T11:11:18.520231Z","iopub.status.idle":"2022-08-10T11:11:18.535274Z","shell.execute_reply.started":"2022-08-10T11:11:18.520188Z","shell.execute_reply":"2022-08-10T11:11:18.534306Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"testPreds","metadata":{"_uuid":"3e092d70-3b9f-4f1a-84f9-d7330977e268","_cell_guid":"100ed75c-c350-4804-9075-9a057939ba45","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:18.536830Z","iopub.execute_input":"2022-08-10T11:11:18.538075Z","iopub.status.idle":"2022-08-10T11:11:18.546244Z","shell.execute_reply.started":"2022-08-10T11:11:18.538035Z","shell.execute_reply":"2022-08-10T11:11:18.545216Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'PassengerId': testDf['PassengerId'],\n                       'Survived': testPreds\n                      })\n\noutput.head(7)","metadata":{"_uuid":"b783eab7-b763-4bc3-9ec9-4cf1ffeaeb43","_cell_guid":"14e26ec3-fbb2-445a-b028-4c6dddc10cb2","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:18.547650Z","iopub.execute_input":"2022-08-10T11:11:18.548410Z","iopub.status.idle":"2022-08-10T11:11:18.563591Z","shell.execute_reply.started":"2022-08-10T11:11:18.548365Z","shell.execute_reply":"2022-08-10T11:11:18.562036Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.to_csv('submission.csv', index=False)","metadata":{"_uuid":"9c0af041-c66f-4219-8396-a727c94e9179","_cell_guid":"a1b5043c-e721-48db-ac67-eff0d1c905b2","collapsed":false,"execution":{"iopub.status.busy":"2022-08-10T11:11:18.565098Z","iopub.execute_input":"2022-08-10T11:11:18.565553Z","iopub.status.idle":"2022-08-10T11:11:18.574647Z","shell.execute_reply.started":"2022-08-10T11:11:18.565519Z","shell.execute_reply":"2022-08-10T11:11:18.573447Z"},"jupyter":{"outputs_hidden":false},"trusted":true},"execution_count":null,"outputs":[]}]}