{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"<!DOCTYPE html>\n<html>\n<head>\n<title>\n</title>\n<meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n</head>\n<body>\n<h1 style=\"color:DodgerBlue;\">\n1 IMPORTING LIBRARIES</h1>\n</body>\n</html>","metadata":{}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import classification_report,confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:17.858212Z","iopub.execute_input":"2022-07-05T13:29:17.858690Z","iopub.status.idle":"2022-07-05T13:29:18.093165Z","shell.execute_reply.started":"2022-07-05T13:29:17.858594Z","shell.execute_reply":"2022-07-05T13:29:18.091868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!DOCTYPE html>\n<html>\n<head>\n<title>\n</title>\n<meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n</head>\n<body>\n<h1 style=\"color:DodgerBlue;\">\n2 Data Import & Preprocessing</h1>\n</body>\n</html>","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntest_data = pd.read_csv(\"/kaggle/input/titanic/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.095602Z","iopub.execute_input":"2022-07-05T13:29:18.095955Z","iopub.status.idle":"2022-07-05T13:29:18.125340Z","shell.execute_reply.started":"2022-07-05T13:29:18.095923Z","shell.execute_reply":"2022-07-05T13:29:18.124085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train data info\ntrain_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.126838Z","iopub.execute_input":"2022-07-05T13:29:18.127174Z","iopub.status.idle":"2022-07-05T13:29:18.159365Z","shell.execute_reply.started":"2022-07-05T13:29:18.127146Z","shell.execute_reply":"2022-07-05T13:29:18.157994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test data info\ntest_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.162284Z","iopub.execute_input":"2022-07-05T13:29:18.163061Z","iopub.status.idle":"2022-07-05T13:29:18.177289Z","shell.execute_reply.started":"2022-07-05T13:29:18.163009Z","shell.execute_reply":"2022-07-05T13:29:18.176289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let us check missing value for train data\ntrain_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.178528Z","iopub.execute_input":"2022-07-05T13:29:18.179406Z","iopub.status.idle":"2022-07-05T13:29:18.199720Z","shell.execute_reply.started":"2022-07-05T13:29:18.179357Z","shell.execute_reply":"2022-07-05T13:29:18.198536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let us check missing value for test data\ntest_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.202785Z","iopub.execute_input":"2022-07-05T13:29:18.203216Z","iopub.status.idle":"2022-07-05T13:29:18.214496Z","shell.execute_reply.started":"2022-07-05T13:29:18.203182Z","shell.execute_reply":"2022-07-05T13:29:18.212977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!DOCTYPE html PUBLIC \"-//W3C//DTD HTML 4.01//EN\" \"http://www.w3.org/TR/html4/strict.dtd\">\n<html style = \"font-family:Calibri, Arial, Helvetica, sans-serif; font-size:11pt; background-color:white \">\n  <head>\n      <meta http-equiv=\"Content-Type\" content=\"text/html; charset=utf-8\">\n      <meta name=\"generator\" content=\"PhpSpreadsheet, https://github.com/PHPOffice/PhpSpreadsheet\">\n      <meta name=\"author\" content=\"Raj Dalsaniya\" />\n    <style type=\"text/css\">\n      a.comment-indicator:hover + div.comment { background:#ffd; position:absolute; display:block; border:1px solid black; padding:0.5em }\n      a.comment-indicator { background:red; display:inline-block; border:1px solid black; width:0.5em; height:0.5em }\n      div.comment { display:none }\n      table { border-collapse:collapse; page-break-after:always }\n      .gridlines td { border:1px dotted black }\n      .gridlines th { border:1px dotted black }\n      .b { text-align:center }\n      .e { text-align:center }\n      .f { text-align:right }\n      .inlineStr { text-align:left }\n      .n { text-align:right }\n      .s { text-align:left }\n      td.style0 { vertical-align:bottom; border-bottom:none #000000; border-top:none #000000; border-left:none #000000; border-right:none #000000; color:#000000; font-family:'Calibri'; font-size:11pt; background-color:white }\n      th.style0 { vertical-align:bottom; border-bottom:none #000000; border-top:none #000000; border-left:none #000000; border-right:none #000000; color:#000000; font-family:'Calibri'; font-size:11pt; background-color:white }\n      td.style1 { vertical-align:middle; text-align:center; border-bottom:none #000000; border-top:none #000000; border-left:none #000000; border-right:none #000000; color:#000000; font-family:'Calibri'; font-size:11pt; background-color:white }\n      th.style1 { vertical-align:middle; text-align:center; border-bottom:none #000000; border-top:none #000000; border-left:none #000000; border-right:none #000000; color:#000000; font-family:'Calibri'; font-size:11pt; background-color:white }\n      td.style2 { vertical-align:bottom; text-align:center; border-bottom:1px solid #000000 !important; border-top:1px solid #000000 !important; border-left:1px solid #000000 !important; border-right:1px solid #000000 !important; color:#000000; font-family:'Calibri'; font-size:11pt; background-color:white }\n      th.style2 { vertical-align:bottom; text-align:center; border-bottom:1px solid #000000 !important; border-top:1px solid #000000 !important; border-left:1px solid #000000 !important; border-right:1px solid #000000 !important; color:#000000; font-family:'Calibri'; font-size:11pt; background-color:white }\n      td.style3 { vertical-align:middle; text-align:center; border-bottom:1px solid #000000 !important; border-top:1px solid #000000 !important; border-left:1px solid #000000 !important; border-right:1px solid #000000 !important; color:#000000; font-family:'Var(--jp-code-font-family)'; font-size:10pt; background-color:white }\n      th.style3 { vertical-align:middle; text-align:center; border-bottom:1px solid #000000 !important; border-top:1px solid #000000 !important; border-left:1px solid #000000 !important; border-right:1px solid #000000 !important; color:#000000; font-family:'Var(--jp-code-font-family)'; font-size:10pt; background-color:white }\n      td.style4 { vertical-align:middle; text-align:center; border-bottom:1px solid #000000 !important; border-top:1px solid #000000 !important; border-left:1px solid #000000 !important; border-right:1px solid #000000 !important; color:#000000; font-family:'Calibri'; font-size:11pt; background-color:white }\n      th.style4 { vertical-align:middle; text-align:center; border-bottom:1px solid #000000 !important; border-top:1px solid #000000 !important; border-left:1px solid #000000 !important; border-right:1px solid #000000 !important; color:#000000; font-family:'Calibri'; font-size:11pt; background-color:white }\n      td.style5 { vertical-align:middle; text-align:center; border-bottom:1px solid #000000 !important; border-top:1px solid #000000 !important; border-left:1px solid #000000 !important; border-right:1px solid #000000 !important; color:#000000; font-family:'Calibri'; font-size:11pt; background-color:#FFFF00 }\n      th.style5 { vertical-align:middle; text-align:center; border-bottom:1px solid #000000 !important; border-top:1px solid #000000 !important; border-left:1px solid #000000 !important; border-right:1px solid #000000 !important; color:#000000; font-family:'Calibri'; font-size:11pt; background-color:#FFFF00 }\n      td.style6 { vertical-align:middle; text-align:center; border-bottom:1px solid #000000 !important; border-top:1px solid #000000 !important; border-left:1px solid #000000 !important; border-right:1px solid #000000 !important; color:#000000; font-family:'Calibri'; font-size:11pt; background-color:#4472C4 }\n      th.style6 { vertical-align:middle; text-align:center; border-bottom:1px solid #000000 !important; border-top:1px solid #000000 !important; border-left:1px solid #000000 !important; border-right:1px solid #000000 !important; color:#000000; font-family:'Calibri'; font-size:11pt; background-color:#4472C4 }\n      td.style7 { vertical-align:middle; text-align:center; border-bottom:none #000000; border-top:none #000000; border-left:none #000000; border-right:none #000000; color:#000000; font-family:'Calibri'; font-size:11pt; background-color:#FFFFFF }\n      th.style7 { vertical-align:middle; text-align:center; border-bottom:none #000000; border-top:none #000000; border-left:none #000000; border-right:none #000000; color:#000000; font-family:'Calibri'; font-size:11pt; background-color:#FFFFFF }\n      td.style8 { vertical-align:middle; text-align:center; border-bottom:none #000000; border-top:none #000000; border-left:none #000000; border-right:none #000000; color:#000000; font-family:'Var(--jp-code-font-family)'; font-size:10pt; background-color:#FFFFFF }\n      th.style8 { vertical-align:middle; text-align:center; border-bottom:none #000000; border-top:none #000000; border-left:none #000000; border-right:none #000000; color:#000000; font-family:'Var(--jp-code-font-family)'; font-size:10pt; background-color:#FFFFFF }\n      table.sheet0 col.col0 { width:91.49999895pt }\n      table.sheet0 col.col1 { width:91.49999895pt }\n      table.sheet0 col.col2 { width:91.49999895pt }\n      table.sheet0 col.col3 { width:26.43333303pt }\n      table.sheet0 col.col4 { width:42pt }\n      table.sheet0 col.col5 { width:42pt }\n      table.sheet0 col.col6 { width:42pt }\n      table.sheet0 tr { height:15pt }\n    </style>\n  </head>\n\n  <body>\n<style>\n@page { margin-left: 0.7in; margin-right: 0.7in; margin-top: 0.75in; margin-bottom: 0.75in; }\nbody { margin-left: 0.7in; margin-right: 0.7in; margin-top: 0.75in; margin-bottom: 0.75in; }\n</style>\n    <table border=\"0\" cellpadding=\"0\" cellspacing=\"0\" id=\"sheet0\" class=\"sheet0 gridlines\">\n        <col class=\"col0\">\n        <col class=\"col1\">\n        <col class=\"col2\">\n        <col class=\"col3\">\n        <col class=\"col4\">\n        <col class=\"col5\">\n        <col class=\"col6\">\n        <tbody>\n          <tr class=\"row0\">\n            <td class=\"column0 style5 s style5\" colspan=\"3\">TRAIN DATA</td>\n            <td class=\"column3 style7 null\"></td>\n            <td class=\"column4 style5 s style5\" colspan=\"3\">TEST DATA</td>\n          </tr>\n          <tr class=\"row1\">\n            <td class=\"column0 style6 s\">Column</td>\n            <td class=\"column1 style6 s\">Missing Row Count</td>\n            <td class=\"column2 style6 s\">Dtype</td>\n            <td class=\"column3 style7 null\"></td>\n            <td class=\"column4 style6 s\">Column</td>\n            <td class=\"column5 style6 s\">Missing Row Count</td>\n            <td class=\"column6 style6 s\">Dtype</td>\n          </tr>\n          <tr class=\"row2\">\n            <td class=\"column0 style3 s\">PassengerId      </td>\n            <td class=\"column1 style4 n\">0</td>\n            <td class=\"column2 style3 s\">int64  </td>\n            <td class=\"column3 style8 null\"></td>\n            <td class=\"column4 style3 s\">PassengerId      </td>\n            <td class=\"column5 style2 n\">0</td>\n            <td class=\"column6 style3 s\">int64  </td>\n          </tr>\n          <tr class=\"row3\">\n            <td class=\"column0 style3 s\">Survived         </td>\n            <td class=\"column1 style4 n\">0</td>\n            <td class=\"column2 style3 s\">int64  </td>\n            <td class=\"column3 style8 null\"></td>\n            <td class=\"column4 style3 s\">Survived         </td>\n            <td class=\"column5 style2 n\">0</td>\n            <td class=\"column6 style3 s\">int64  </td>\n          </tr>\n          <tr class=\"row4\">\n            <td class=\"column0 style3 s\">Pclass           </td>\n            <td class=\"column1 style4 n\">0</td>\n            <td class=\"column2 style3 s\">int64  </td>\n            <td class=\"column3 style8 null\"></td>\n            <td class=\"column4 style3 s\">Pclass           </td>\n            <td class=\"column5 style2 n\">0</td>\n            <td class=\"column6 style3 s\">int64  </td>\n          </tr>\n          <tr class=\"row5\">\n            <td class=\"column0 style3 s\">Name             </td>\n            <td class=\"column1 style4 n\">0</td>\n            <td class=\"column2 style4 s\">object</td>\n            <td class=\"column3 style7 null\"></td>\n            <td class=\"column4 style3 s\">Name             </td>\n            <td class=\"column5 style2 n\">0</td>\n            <td class=\"column6 style4 s\">object</td>\n          </tr>\n          <tr class=\"row6\">\n            <td class=\"column0 style3 s\">Sex              </td>\n            <td class=\"column1 style4 n\">0</td>\n            <td class=\"column2 style4 s\">object</td>\n            <td class=\"column3 style7 null\"></td>\n            <td class=\"column4 style3 s\">Sex              </td>\n            <td class=\"column5 style2 n\">0</td>\n            <td class=\"column6 style4 s\">object</td>\n          </tr>\n          <tr class=\"row7\">\n            <td class=\"column0 style3 s\">Age            </td>\n            <td class=\"column1 style4 n\">177</td>\n            <td class=\"column2 style3 s\">float64</td>\n            <td class=\"column3 style8 null\"></td>\n            <td class=\"column4 style3 s\">Age            </td>\n            <td class=\"column5 style2 n\">86</td>\n            <td class=\"column6 style3 s\">float64</td>\n          </tr>\n          <tr class=\"row8\">\n            <td class=\"column0 style3 s\">SibSp            </td>\n            <td class=\"column1 style4 n\">0</td>\n            <td class=\"column2 style3 s\">int64  </td>\n            <td class=\"column3 style8 null\"></td>\n            <td class=\"column4 style3 s\">SibSp            </td>\n            <td class=\"column5 style2 n\">0</td>\n            <td class=\"column6 style3 s\">int64  </td>\n          </tr>\n          <tr class=\"row9\">\n            <td class=\"column0 style3 s\">Parch            </td>\n            <td class=\"column1 style4 n\">0</td>\n            <td class=\"column2 style3 s\">int64  </td>\n            <td class=\"column3 style8 null\"></td>\n            <td class=\"column4 style3 s\">Parch            </td>\n            <td class=\"column5 style2 n\">0</td>\n            <td class=\"column6 style3 s\">int64  </td>\n          </tr>\n          <tr class=\"row10\">\n            <td class=\"column0 style3 s\">Ticket           </td>\n            <td class=\"column1 style4 n\">0</td>\n            <td class=\"column2 style4 s\">object</td>\n            <td class=\"column3 style7 null\"></td>\n            <td class=\"column4 style3 s\">Ticket           </td>\n            <td class=\"column5 style2 n\">0</td>\n            <td class=\"column6 style4 s\">object</td>\n          </tr>\n          <tr class=\"row11\">\n            <td class=\"column0 style3 s\">Fare             </td>\n            <td class=\"column1 style4 n\">0</td>\n            <td class=\"column2 style3 s\">float64</td>\n            <td class=\"column3 style8 null\"></td>\n            <td class=\"column4 style3 s\">Fare             </td>\n            <td class=\"column5 style2 n\">1</td>\n            <td class=\"column6 style3 s\">float64</td>\n          </tr>\n          <tr class=\"row12\">\n            <td class=\"column0 style3 s\">Cabin          </td>\n            <td class=\"column1 style4 n\">687</td>\n            <td class=\"column2 style4 s\">object</td>\n            <td class=\"column3 style7 null\"></td>\n            <td class=\"column4 style3 s\">Cabin          </td>\n            <td class=\"column5 style2 n\">327</td>\n            <td class=\"column6 style4 s\">object</td>\n          </tr>\n          <tr class=\"row13\">\n            <td class=\"column0 style3 s\">Embarked         </td>\n            <td class=\"column1 style4 n\">2</td>\n            <td class=\"column2 style4 s\">object</td>\n            <td class=\"column3 style7 null\"></td>\n            <td class=\"column4 style3 s\">Embarked         </td>\n            <td class=\"column5 style2 n\">0</td>\n            <td class=\"column6 style4 s\">object</td>\n          </tr>\n        </tbody>\n    </table>\n  </body>\n</html>","metadata":{}},{"cell_type":"code","source":"# let see train_data correlation for Age column with other columns\ntrain_data_corr = train_data.corr().abs().unstack().sort_values(kind = \"quicksort\", ascending = False).reset_index()\n\ntrain_data_corr.rename(columns = {\"level_0\":\"Feature 1\",\"level_1\":\"Feature 2\",0:\"Correlation coefficeint\"},inplace = True)\ntrain_data_corr[train_data_corr['Feature 1'] == 'Age']","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.217614Z","iopub.execute_input":"2022-07-05T13:29:18.218765Z","iopub.status.idle":"2022-07-05T13:29:18.249311Z","shell.execute_reply.started":"2022-07-05T13:29:18.218724Z","shell.execute_reply":"2022-07-05T13:29:18.247981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!DOCTYPE html>\n<html>\n<head>\n<title>\n</title>\n<meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n</head>\n<body>\n<h1 style=\"color:Red;\">\nAge Variable :\n</h1>\n<h2 style=\"color:DodgerBlue;\">\nMissing value in Age Group, we will fill using median. using whole dataset median value of Age is not good idea. since Age and Pclass is highly correlated with survival ( output variable) . so it is more good to grup age by Pclass.\n</h2>\n</body>\n</html>","metadata":{}},{"cell_type":"code","source":"age_by_pclass_sex = train_data.groupby([\"Pclass\",\"Sex\"]).median()[\"Age\"]\nfor pclass in range(1,4):\n    for sex in [\"female\",\"male\"] :\n        print(\"median value of pclass \"+str(pclass)+\" \"+ sex+\" is:\"+str(age_by_pclass_sex[pclass][sex]))\nprint('Median age of all passengers: {}'.format(train_data['Age'].median()))\n\n# let us impute missing age value using group of Pclass and sex\ntrain_data[\"Age\"] = train_data.groupby([\"Pclass\",\"Sex\"])[\"Age\"].apply(lambda x:x.fillna(x.median()))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.251507Z","iopub.execute_input":"2022-07-05T13:29:18.252378Z","iopub.status.idle":"2022-07-05T13:29:18.283016Z","shell.execute_reply.started":"2022-07-05T13:29:18.252340Z","shell.execute_reply":"2022-07-05T13:29:18.282022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age_by_pclass_sex_test = test_data.groupby([\"Pclass\",\"Sex\"]).median()[\"Age\"]\nfor pclass in range(1,4):\n    for sex in [\"female\",\"male\"]:\n        print(\"median value of pclass \"+str(pclass)+\" \"+ sex+\" is:\"+str(age_by_pclass_sex_test[pclass][sex]))\nprint('Median age of all passengers: {}'.format(test_data['Age'].median()))\n\n# let us impute missing age value using group of Pclass and sex for test data\ntest_data[\"Age\"] = test_data.groupby([\"Pclass\",\"Sex\"])[\"Age\"].apply(lambda x:x.fillna(x.median()))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.284684Z","iopub.execute_input":"2022-07-05T13:29:18.285864Z","iopub.status.idle":"2022-07-05T13:29:18.308200Z","shell.execute_reply.started":"2022-07-05T13:29:18.285814Z","shell.execute_reply":"2022-07-05T13:29:18.306947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[train_data[\"Embarked\"].isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.310005Z","iopub.execute_input":"2022-07-05T13:29:18.310620Z","iopub.status.idle":"2022-07-05T13:29:18.329304Z","shell.execute_reply.started":"2022-07-05T13:29:18.310583Z","shell.execute_reply":"2022-07-05T13:29:18.327727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embarked_by_Sex = train_data.groupby([\"Embarked\",\"Sex\"]).count()[\"Age\"]\nembarked_by_Sex","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.331303Z","iopub.execute_input":"2022-07-05T13:29:18.332552Z","iopub.status.idle":"2022-07-05T13:29:18.354534Z","shell.execute_reply.started":"2022-07-05T13:29:18.332499Z","shell.execute_reply":"2022-07-05T13:29:18.352563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!DOCTYPE html>\n<html>\n<head>\n<title>\n</title>\n<meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n</head>\n<body>\n<h1 style=\"color:Red;\">\nEmbarked Variable :\n</h1>\n<h2 style=\"color:DodgerBlue;\">\nEmbarked column is categorical variable. There is 2 value missing in embarked column. since there is only 2 value missing shown above.both passenger holding same ticket number and both are female.Since Both holding same ticket we know both embarked from same port. Mode value for embarked for female is S but this is not necessarily true. let check if we find something on internet\n</h2>\n</body>\n</html> .","metadata":{}},{"cell_type":"markdown","source":"<!DOCTYPE html>\n<html>\n<head>\n<title>\n</title>\n<meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n</head>\n<body>\n<h3 style=\"color:DodgerBlue;\">\nWhen I search about Stone, Mrs. George Nelson (Martha Evelyn), on google. i found some information . you can check on this <a href=\" https://www.encyclopedia-titanica.org/titanic-survivor/martha-evelyn-stone.html\">link</a> From below image is clear that Martha Evelyn embarked from S port\n</h3>\n<a href=\"https://imgbb.com/\"><img src=\"https://i.ibb.co/0qS1818/test-1.jpg\" alt=\"test-1\" border=\"0\"></a><br />\n</body>\n</html> .\n","metadata":{}},{"cell_type":"code","source":"# let impute missing data in Embarked column with S\ntrain_data['Embarked'] = train_data['Embarked'].fillna('S')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.357401Z","iopub.execute_input":"2022-07-05T13:29:18.358517Z","iopub.status.idle":"2022-07-05T13:29:18.366697Z","shell.execute_reply.started":"2022-07-05T13:29:18.358447Z","shell.execute_reply":"2022-07-05T13:29:18.364721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fare value column\ntest_data[test_data[\"Fare\"].isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.388906Z","iopub.execute_input":"2022-07-05T13:29:18.389587Z","iopub.status.idle":"2022-07-05T13:29:18.407050Z","shell.execute_reply.started":"2022-07-05T13:29:18.389549Z","shell.execute_reply":"2022-07-05T13:29:18.405374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!DOCTYPE html>\n<html>\n<head>\n<title>\n</title>\n<meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n</head>\n<body>\n<h1 style=\"color:Red;\">\nFare column :\n</h1>\n<h2 style=\"color:DodgerBlue;\">\nIn Fare colum only one value missing let us check which row has missing value. We assume that fare is related to family size and Pclass. as missing value sex is male , Pclass is 3 and no family(Parch and SibSp). so we can group by above feature like sex=Male; Parch = 0; SibSp= 0; Pclass = 3 and impute median value .\n</h2>\n</body>\n</html>","metadata":{}},{"cell_type":"code","source":"median_fare = test_data.groupby([\"Pclass\",\"Parch\",\"SibSp\"]).Fare.median()[3][0][0]\ntest_data[\"Fare\"] = test_data[\"Fare\"] .fillna(median_fare)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.427951Z","iopub.execute_input":"2022-07-05T13:29:18.428360Z","iopub.status.idle":"2022-07-05T13:29:18.439338Z","shell.execute_reply.started":"2022-07-05T13:29:18.428326Z","shell.execute_reply":"2022-07-05T13:29:18.437995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!DOCTYPE html>\n<html>\n<head>\n<title>\n</title>\n<meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n</head>\n<body>\n<h1 style=\"color:Red;\">\nCabin column :\n</h1>\n<h2 style=\"color:DodgerBlue\">\nCabin Feature is Categorical value. There is lots of missing value in dataset. but as there are some cabin which have high survival rate and some has not.Let us explore more for Cabin Feature what can we do on this. Look at this blueprint of titanic.\n</h2>\n<a href=\"https://ibb.co/mvqwTpkk\"><img src=\"https://i.ibb.co/TK8S2y6/Titanic-side-plan.webp\" alt=\"Titanic-side-plan\" border=\"1\" width=\"1500\" height=\"600\"></a>\n<ul style=\"color:DodgerBlue\">\n<li>On the Boat Deck there were 6 rooms labeled as T, U, W, X, Y, Z but only the T cabin is present in the dataset</li>\n<li>A, B and C decks were only for 1st class passengers</li>\n<li>D and E decks were for all classes</li>\n<li>F and G decks were for both 2nd and 3rd class passengers</li>\n<li>From going A to G, distance to the staircase increases which might be a factor of survival</li>\n</ul>  \n</body>\n</html> ","metadata":{}},{"cell_type":"code","source":"# let us create new coulmn : first letter of the cabin column as deck ... for missing value as M deck (train_data)\ntrain_data[\"Deck\"] =  train_data['Cabin'].apply(lambda s: s[0] if pd.notnull(s) else 'M')\ntrain_data_decks =  train_data.groupby(['Deck', 'Pclass']).count().drop(columns=['Survived', 'Sex', 'Age', 'SibSp', 'Parch', \n                                                                        'Fare', 'Embarked', 'Cabin', 'PassengerId', 'Ticket']).rename(columns={'Name': 'Count'}).transpose()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.463212Z","iopub.execute_input":"2022-07-05T13:29:18.463663Z","iopub.status.idle":"2022-07-05T13:29:18.479863Z","shell.execute_reply.started":"2022-07-05T13:29:18.463627Z","shell.execute_reply":"2022-07-05T13:29:18.478089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let us create new coulmn : first letter of the cabin column as deck ... for missing value as M deck (test data)\ntest_data[\"Deck\"] =  test_data['Cabin'].apply(lambda s: s[0] if pd.notnull(s) else 'M')\ntest_data_decks =  test_data.groupby(['Deck', 'Pclass']).count().drop(columns=['Sex', 'Age', 'SibSp', 'Parch', \n                                                                        'Fare', 'Embarked', 'Cabin', 'PassengerId', 'Ticket']).rename(columns={'Name': 'Count'}).transpose()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.506347Z","iopub.execute_input":"2022-07-05T13:29:18.506744Z","iopub.status.idle":"2022-07-05T13:29:18.525716Z","shell.execute_reply.started":"2022-07-05T13:29:18.506714Z","shell.execute_reply":"2022-07-05T13:29:18.523835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_pclass_count_per(df):\n    \n    # Creating a dictionary for every passenger class count in every deck\n    deck_counts = {'A': {}, 'B': {}, 'C': {}, 'D': {}, 'E': {}, 'F': {}, 'G': {}, 'M': {}, 'T': {}}\n    decks = df.columns.levels[0]\n    \n    for deck in decks:\n        for pclass in range(1,4):\n            try:\n                count = df[deck][pclass][0]\n                deck_counts[deck][pclass] = count \n            except KeyError:\n                deck_counts[deck][pclass] = 0\n    \n    df_decks = pd.DataFrame(deck_counts)    \n    deck_percentages = {}\n\n    # Creating a dictionary for every passenger class percentage in every deck\n    for col in df_decks.columns:\n        deck_percentages[col] = [(count / df_decks[col].sum()) * 100 for count in df_decks[col]]\n        \n    return deck_counts, deck_percentages","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.546394Z","iopub.execute_input":"2022-07-05T13:29:18.547020Z","iopub.status.idle":"2022-07-05T13:29:18.557425Z","shell.execute_reply.started":"2022-07-05T13:29:18.546983Z","shell.execute_reply":"2022-07-05T13:29:18.555680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_pclass_dist(percentages):\n    \n    df_percentages = pd.DataFrame(percentages).transpose()\n    deck_names = ('A', 'B', 'C', 'D', 'E', 'F', 'G', 'M', 'T')\n    bar_count = np.arange(len(deck_names))  \n    bar_width = 0.85\n    \n    pclass1 = df_percentages[0]\n    pclass2 = df_percentages[1]\n    pclass3 = df_percentages[2]\n    \n    plt.figure(figsize=(20, 10))\n    plt.bar(bar_count, pclass1, color='#b5ffb9', edgecolor='white', width=bar_width, label='Passenger Class 1')\n    plt.bar(bar_count, pclass2, bottom=pclass1, color='#f9bc86', edgecolor='white', width=bar_width, label='Passenger Class 2')\n    plt.bar(bar_count, pclass3, bottom=pclass1 + pclass2, color='#a3acff', edgecolor='white', width=bar_width, label='Passenger Class 3')\n\n    plt.xlabel('Deck', size=15, labelpad=20)\n    plt.ylabel('Passenger Class Percentage', size=15, labelpad=20)\n    plt.xticks(bar_count, deck_names)    \n    plt.tick_params(axis='x', labelsize=15)\n    plt.tick_params(axis='y', labelsize=15)\n    \n    plt.legend(loc='upper left', bbox_to_anchor=(1, 1), prop={'size': 15})\n    plt.title('Passenger Class Distribution in Decks', size=18, y=1.05)   \n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.581045Z","iopub.execute_input":"2022-07-05T13:29:18.581917Z","iopub.status.idle":"2022-07-05T13:29:18.595548Z","shell.execute_reply.started":"2022-07-05T13:29:18.581876Z","shell.execute_reply":"2022-07-05T13:29:18.593383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let do first for train data\ntrain_deck_count, train_deck_per = get_pclass_count_per(train_data_decks)\ndisplay_pclass_dist(train_deck_per)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:18.610634Z","iopub.execute_input":"2022-07-05T13:29:18.611778Z","iopub.status.idle":"2022-07-05T13:29:19.009337Z","shell.execute_reply.started":"2022-07-05T13:29:18.611736Z","shell.execute_reply":"2022-07-05T13:29:19.007593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let do first for test data\ntest_deck_count, test_deck_per = get_pclass_count_per(test_data_decks)\ndisplay_pclass_dist(test_deck_per)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:19.012650Z","iopub.execute_input":"2022-07-05T13:29:19.013253Z","iopub.status.idle":"2022-07-05T13:29:19.701756Z","shell.execute_reply.started":"2022-07-05T13:29:19.013214Z","shell.execute_reply":"2022-07-05T13:29:19.700520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!DOCTYPE html>\n<html>\n<head>\n<title>\n</title>\n<meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n</head>\n<body>\n<h1 style=\"color:Red;\">\nIntutation from Above graphs (train and test data):\n</h1>\n<ul style=\"color:DodgerBlue; list-style-type: square; padding: 10px; font-size:16px;\"><ul style=\"color:DodgerBlue; font-size:16px;\">\n<li>100% Passanger from A,B,C decks are 1st class passanger</li>\n<li>For D deck 87% passanger around are 1st class passanger. others are 2nd class passanger</li>\n<li>For E in test data it belong to 1st class passanger. For G in test data it belong to 3rd class Passanger 100%</li>\n<li>For E Deck in train data nearly 80% belong to 1st class passanger.around 10% each belong to 2nd and 3rd class passanger </li>\n<li>There is one person on the boat deck in T cabin and he is a 1st class passenger. T cabin passenger has the closest resemblance to A deck passengers so he is grouped with A deck</li>\n<li>Passengers labeled as M are the missing values in Cabin feature. I don't think it is possible to find those passengers' real Deck so I decided to use M like a deck</li>\n<li> 100% of G deck are 3rd class passanger</li>\n</ul>  \n</body>\n</html> ","metadata":{}},{"cell_type":"code","source":"# let us change passanger in T deck is changed to A (train Data)\nidx = train_data[train_data['Deck'] == 'T'].index\ntrain_data.loc[idx, 'Deck'] = 'A'","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:19.703728Z","iopub.execute_input":"2022-07-05T13:29:19.704867Z","iopub.status.idle":"2022-07-05T13:29:19.714499Z","shell.execute_reply.started":"2022-07-05T13:29:19.704816Z","shell.execute_reply":"2022-07-05T13:29:19.713565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let us change passanger in T deck is changed to A (test Data)\nidx = test_data[test_data['Deck'] == 'T'].index\ntest_data.loc[idx, 'Deck'] = 'A'","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:19.716595Z","iopub.execute_input":"2022-07-05T13:29:19.717537Z","iopub.status.idle":"2022-07-05T13:29:19.728506Z","shell.execute_reply.started":"2022-07-05T13:29:19.717428Z","shell.execute_reply":"2022-07-05T13:29:19.727593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_decks_survived = train_data.groupby(['Deck', 'Survived']).count().drop(columns=['Sex', 'Age', 'SibSp', 'Parch', 'Fare', \n                                                                                   'Embarked', 'Pclass', 'Cabin', 'PassengerId', 'Ticket']).rename(columns={'Name':'Count'}).transpose()\n\ntest_data_decks_survived = test_data.groupby(['Deck']).count().drop(columns=['Sex', 'Age', 'SibSp', 'Parch', 'Fare', \n                                                                                   'Embarked', 'Pclass', 'Cabin', 'PassengerId', 'Ticket']).rename(columns={'Name':'Count'}).transpose()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:19.729608Z","iopub.execute_input":"2022-07-05T13:29:19.730672Z","iopub.status.idle":"2022-07-05T13:29:19.756418Z","shell.execute_reply.started":"2022-07-05T13:29:19.730637Z","shell.execute_reply":"2022-07-05T13:29:19.754392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_survived_dist(df):\n    \n    # Creating a dictionary for every survival count in every deck\n    surv_counts = {'A':{}, 'B':{}, 'C':{}, 'D':{}, 'E':{}, 'F':{}, 'G':{}, 'M':{}}\n    decks = df.columns.levels[0]    \n\n    for deck in decks:\n        for survive in range(0, 2):\n            surv_counts[deck][survive] = df[deck][survive][0]\n            \n    df_surv = pd.DataFrame(surv_counts)\n    surv_percentages = {}\n\n    for col in df_surv.columns:\n        surv_percentages[col] = [(count / df_surv[col].sum()) * 100 for count in df_surv[col]]\n        \n    return surv_counts, surv_percentages\n\ndef display_surv_dist(percentages):\n    \n    df_survived_percentages = pd.DataFrame(percentages).transpose()\n    deck_names = ('A', 'B', 'C', 'D', 'E', 'F', 'G', 'M')\n    bar_count = np.arange(len(deck_names))  \n    bar_width = 0.85    \n\n    not_survived = df_survived_percentages[0]\n    survived = df_survived_percentages[1]\n    \n    plt.figure(figsize=(20, 10))\n    plt.bar(bar_count, not_survived, color='#b5ffb9', edgecolor='white', width=bar_width, label=\"Not Survived\")\n    plt.bar(bar_count, survived, bottom=not_survived, color='#f9bc86', edgecolor='white', width=bar_width, label=\"Survived\")\n \n    plt.xlabel('Deck', size=15, labelpad=20)\n    plt.ylabel('Survival Percentage', size=15, labelpad=20)\n    plt.xticks(bar_count, deck_names)    \n    plt.tick_params(axis='x', labelsize=15)\n    plt.tick_params(axis='y', labelsize=15)\n    \n    plt.legend(loc='upper left', bbox_to_anchor=(1, 1), prop={'size': 15})\n    plt.title('Survival Percentage in Decks', size=18, y=1.05)\n    \n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:19.760714Z","iopub.execute_input":"2022-07-05T13:29:19.761071Z","iopub.status.idle":"2022-07-05T13:29:19.775640Z","shell.execute_reply.started":"2022-07-05T13:29:19.761042Z","shell.execute_reply":"2022-07-05T13:29:19.774364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_surv_count, train_surv_per = get_survived_dist(train_data_decks_survived)\ndisplay_surv_dist(train_surv_per)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:19.777447Z","iopub.execute_input":"2022-07-05T13:29:19.778560Z","iopub.status.idle":"2022-07-05T13:29:20.075271Z","shell.execute_reply.started":"2022-07-05T13:29:19.778445Z","shell.execute_reply":"2022-07-05T13:29:20.073976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# since ABC Deck is all First class Citizen let change deck as ABC\ntrain_data['Deck'] = train_data['Deck'].replace(['A', 'B', 'C'], 'ABC')\ntest_data['Deck'] = test_data['Deck'].replace(['A', 'B', 'C'], 'ABC')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:20.077637Z","iopub.execute_input":"2022-07-05T13:29:20.078026Z","iopub.status.idle":"2022-07-05T13:29:20.091036Z","shell.execute_reply.started":"2022-07-05T13:29:20.077992Z","shell.execute_reply":"2022-07-05T13:29:20.089155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# D and E decks are labeled as DE because both of them have similar passenger class distribution and same survival rate\ntrain_data['Deck'] = train_data['Deck'].replace(['D', 'E'], 'DE')\ntest_data['Deck'] = test_data['Deck'].replace(['D', 'E'], 'DE')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:20.093406Z","iopub.execute_input":"2022-07-05T13:29:20.093932Z","iopub.status.idle":"2022-07-05T13:29:20.108754Z","shell.execute_reply.started":"2022-07-05T13:29:20.093897Z","shell.execute_reply":"2022-07-05T13:29:20.106971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# F and G decks are labeled as DE because both of them have similar passenger class distribution and same survival rate\ntrain_data['Deck'] = train_data['Deck'].replace(['F', 'G'], 'FG')\ntest_data['Deck'] = test_data['Deck'].replace(['F', 'G'], 'FG')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:20.113547Z","iopub.execute_input":"2022-07-05T13:29:20.114849Z","iopub.status.idle":"2022-07-05T13:29:20.130972Z","shell.execute_reply.started":"2022-07-05T13:29:20.114805Z","shell.execute_reply":"2022-07-05T13:29:20.129296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let check value counts for deck column in train data\ntrain_data[\"Deck\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:20.133301Z","iopub.execute_input":"2022-07-05T13:29:20.134914Z","iopub.status.idle":"2022-07-05T13:29:20.151911Z","shell.execute_reply.started":"2022-07-05T13:29:20.134825Z","shell.execute_reply":"2022-07-05T13:29:20.150447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let check value counts for deck column in test data\ntest_data[\"Deck\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:20.153665Z","iopub.execute_input":"2022-07-05T13:29:20.154135Z","iopub.status.idle":"2022-07-05T13:29:20.169056Z","shell.execute_reply.started":"2022-07-05T13:29:20.154096Z","shell.execute_reply":"2022-07-05T13:29:20.168067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# we extract Deck feature from Cabin feature so we will drop cabin column\ntrain_data.drop(['Cabin'], inplace=True, axis=1)\ntest_data.drop(['Cabin'], inplace=True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:20.170114Z","iopub.execute_input":"2022-07-05T13:29:20.171045Z","iopub.status.idle":"2022-07-05T13:29:20.182327Z","shell.execute_reply.started":"2022-07-05T13:29:20.171007Z","shell.execute_reply":"2022-07-05T13:29:20.181299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let check null value in train data\ntrain_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:20.183896Z","iopub.execute_input":"2022-07-05T13:29:20.185001Z","iopub.status.idle":"2022-07-05T13:29:20.195216Z","shell.execute_reply.started":"2022-07-05T13:29:20.184963Z","shell.execute_reply":"2022-07-05T13:29:20.194365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let check null value in test data\ntest_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:20.196884Z","iopub.execute_input":"2022-07-05T13:29:20.198002Z","iopub.status.idle":"2022-07-05T13:29:20.213810Z","shell.execute_reply.started":"2022-07-05T13:29:20.197964Z","shell.execute_reply":"2022-07-05T13:29:20.212544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!DOCTYPE html>\n<html>\n<head>\n<title>\n</title>\n<meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n</head>\n<body>\n<h1 style=\"color:DodgerBlue;\">\n3 Feature Engineering</h1>\n<ul style=\"color:DodgerBlue; list-style-type: square; padding: 10px; font-size:16px;\">\n    <li>Family size : We will count total no of family member avilable in Family </li>\n    <li>IsAlone : this will indicate he/she was travelling along or not </li>\n    <li>FareBin : Bins range for Fare</li>\n    <li>AgeBin : Bins range for Age variable</li>\n    <li> Title : Title from name</li>\n    <li>Ticket_Frequency: occurence of Ticket values</li>\n    <li>Is_Married : this will categorical feature which indicate it is married or not</li>\n    <li>Family : this feature is created with the extracted surname.</li>\n</ul>\n</body>\n</html>","metadata":{}},{"cell_type":"code","source":"## feature engineering for creaing new features for train and test data\n\ndata_cleaner = [train_data, test_data]\nfor dataset in data_cleaner:\n    \n    # discreate variable Family Size\n    dataset[\"FamilySize\"] = dataset['SibSp'] + dataset['Parch'] + 1\n    \n    #for ISAlone feature \n    dataset[\"IsAlone\"] = 1 #initialize to yes/1 is alone\n    dataset['IsAlone'].loc[dataset['FamilySize'] > 1] = 0 # now update to no/0 if family size is greater than 1\n    \n    #Fare Bins : using qcut\n    dataset['FareBin'] = pd.qcut(dataset['Fare'], 4) # creating 4 bins\n    \n    #Age Bins : using qcut\n    dataset[\"AgeBin\"] = pd.qcut(dataset['Age'].astype(int), 5) # creating 5 bins\n    \n    # Title : we will use split method\n    dataset[\"Title\"] = dataset[\"Name\"].str.split(\", \", expand=True)[1].str.split(\".\", expand=True)[0]\n    \n    #cleanup rare title names\n    stat_min = 10\n    \n    title_names = (dataset['Title'].value_counts() < stat_min)\n    \n    #apply and lambda functions are quick and dirty code to find and replace with fewer lines of code: https://community.modeanalytics.com/python/tutorial/pandas-groupby-and-python-lambda-functions/\n    dataset['Title'] = dataset['Title'].apply(lambda x: 'Misc' if title_names.loc[x] == True else x)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:20.215758Z","iopub.execute_input":"2022-07-05T13:29:20.216299Z","iopub.status.idle":"2022-07-05T13:29:20.289164Z","shell.execute_reply.started":"2022-07-05T13:29:20.216266Z","shell.execute_reply":"2022-07-05T13:29:20.288009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train data Survival vs Fare bins\nfig,axis = plt.subplots(figsize = (22,9))\nsns.countplot(x='FareBin', hue='Survived', data=train_data)\n\nplt.xlabel('Fare', size=15, labelpad=20)\nplt.ylabel('Passenger Count', size=15, labelpad=20)\nplt.tick_params(axis='x', labelsize=10)\nplt.tick_params(axis='y', labelsize=15)\n\nplt.legend(['Not Survived', 'Survived'], loc='upper right', prop={'size': 15})\nplt.title('Count of Survival in {} Feature'.format('Fare'), size=15, y=1.05)\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:20.290708Z","iopub.execute_input":"2022-07-05T13:29:20.291736Z","iopub.status.idle":"2022-07-05T13:29:20.561978Z","shell.execute_reply.started":"2022-07-05T13:29:20.291693Z","shell.execute_reply":"2022-07-05T13:29:20.560598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train data Survival vs Fare bins\nfig,axis = plt.subplots(figsize = (22,9))\nsns.countplot(x='AgeBin', hue='Survived', data=train_data)\n\nplt.xlabel('Fare', size=15, labelpad=20)\nplt.ylabel('Passenger Count', size=15, labelpad=20)\nplt.tick_params(axis='x', labelsize=10)\nplt.tick_params(axis='y', labelsize=15)\n\nplt.legend(['Not Survived', 'Survived'], loc='upper right', prop={'size': 15})\nplt.title('Count of Survival in {} Feature'.format('Age'), size=15, y=1.05)\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:20.564630Z","iopub.execute_input":"2022-07-05T13:29:20.565169Z","iopub.status.idle":"2022-07-05T13:29:20.866422Z","shell.execute_reply.started":"2022-07-05T13:29:20.565134Z","shell.execute_reply":"2022-07-05T13:29:20.864604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(figsize=(20, 20), ncols=2, nrows=2)\nplt.subplots_adjust(right=1.5)\n\nsns.barplot(x=train_data['FamilySize'].value_counts().index, y=train_data['FamilySize'].value_counts().values, ax=axs[0][0])\nsns.countplot(x='FamilySize', hue='Survived', data=train_data, ax=axs[0][1])\n\naxs[0][0].set_title('Family Size Feature Value Counts', size=20, y=1.05)\naxs[0][1].set_title('Survival Counts in Family Size ', size=20, y=1.05)\n\nfamily_map = {1: 'Alone', 2: 'Small', 3: 'Small', 4: 'Small', 5: 'Medium', 6: 'Medium', 7: 'Large', 8: 'Large', 11: 'Large'}\ntrain_data['FamilySize_Grouped'] = train_data['FamilySize'].map(family_map)\n\nsns.barplot(x=train_data['FamilySize_Grouped'].value_counts().index, y=train_data['FamilySize_Grouped'].value_counts().values, ax=axs[1][0])\nsns.countplot(x='FamilySize_Grouped', hue='Survived', data=train_data, ax=axs[1][1])\n\naxs[1][0].set_title('Family Size Feature Value Counts After Grouping', size=20, y=1.05)\naxs[1][1].set_title('Survival Counts in Family Size After Grouping', size=20, y=1.05)\n\nfor i in range(2):\n    axs[i][1].legend(['Not Survived', 'Survived'], loc='upper right', prop={'size': 20})\n    for j in range(2):\n        axs[i][j].tick_params(axis='x', labelsize=20)\n        axs[i][j].tick_params(axis='y', labelsize=20)\n        axs[i][j].set_xlabel('')\n        axs[i][j].set_ylabel('')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:20.868889Z","iopub.execute_input":"2022-07-05T13:29:20.869676Z","iopub.status.idle":"2022-07-05T13:29:21.578642Z","shell.execute_reply.started":"2022-07-05T13:29:20.869627Z","shell.execute_reply":"2022-07-05T13:29:21.577658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"family_map = {1: 'Alone', 2: 'Small', 3: 'Small', 4: 'Small', 5: 'Medium', 6: 'Medium', 7: 'Large', 8: 'Large', 11: 'Large'}\ntest_data['FamilySize_Grouped'] = test_data['FamilySize'].map(family_map)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:21.580277Z","iopub.execute_input":"2022-07-05T13:29:21.580725Z","iopub.status.idle":"2022-07-05T13:29:21.589759Z","shell.execute_reply.started":"2022-07-05T13:29:21.580684Z","shell.execute_reply":"2022-07-05T13:29:21.588875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data[\"Ticket_Frequency\"] = train_data.groupby('Ticket')['Ticket'].transform('count')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:21.590950Z","iopub.execute_input":"2022-07-05T13:29:21.591725Z","iopub.status.idle":"2022-07-05T13:29:21.610973Z","shell.execute_reply.started":"2022-07-05T13:29:21.591686Z","shell.execute_reply":"2022-07-05T13:29:21.608591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data[\"Ticket_Frequency\"] = test_data.groupby('Ticket')['Ticket'].transform('count')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:21.613550Z","iopub.execute_input":"2022-07-05T13:29:21.614128Z","iopub.status.idle":"2022-07-05T13:29:21.623493Z","shell.execute_reply.started":"2022-07-05T13:29:21.614093Z","shell.execute_reply":"2022-07-05T13:29:21.621976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train data Survival vs ticket frequency\nfig,axis = plt.subplots(figsize = (22,9))\nsns.countplot(x='Ticket_Frequency', hue='Survived', data=train_data)\n\nplt.xlabel('Fare', size=15, labelpad=20)\nplt.ylabel('Passenger Count', size=15, labelpad=20)\nplt.tick_params(axis='x', labelsize=10)\nplt.tick_params(axis='y', labelsize=15)\n\nplt.legend(['Not Survived', 'Survived'], loc='upper right', prop={'size': 15})\nplt.title('Count of Survival in {} Feature'.format('Age'), size=15, y=1.05)\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:21.624607Z","iopub.execute_input":"2022-07-05T13:29:21.624920Z","iopub.status.idle":"2022-07-05T13:29:21.899079Z","shell.execute_reply.started":"2022-07-05T13:29:21.624892Z","shell.execute_reply":"2022-07-05T13:29:21.897867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['Is_Married'] = 0\ntrain_data['Is_Married'].loc[train_data['Title'] == 'Mrs'] = 1","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:21.900559Z","iopub.execute_input":"2022-07-05T13:29:21.900893Z","iopub.status.idle":"2022-07-05T13:29:21.909209Z","shell.execute_reply.started":"2022-07-05T13:29:21.900862Z","shell.execute_reply":"2022-07-05T13:29:21.907967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['Is_Married'] = 0\ntest_data['Is_Married'].loc[test_data['Title'] == 'Mrs'] = 1","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:21.910820Z","iopub.execute_input":"2022-07-05T13:29:21.911118Z","iopub.status.idle":"2022-07-05T13:29:21.922127Z","shell.execute_reply.started":"2022-07-05T13:29:21.911091Z","shell.execute_reply":"2022-07-05T13:29:21.921025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import string\ndef extract_surname(data):\n    \n    families = []\n    \n    for i in range(len(data)):\n        name = data.iloc[i]\n        if '(' in name:\n            name_no_bracket = name.split('(')[0] \n        else:\n            name_no_bracket = name\n            \n        family = name_no_bracket.split(',')[0]\n        title = name_no_bracket.split(',')[1].strip().split(' ')[0]\n        \n        for c in string.punctuation:\n            family = family.replace(c, '').strip()\n            \n        families.append(family)\n            \n    return families\n\ntrain_data[\"Family\"] = extract_surname(train_data[\"Name\"])\ntest_data[\"Family\"] = extract_surname(test_data[\"Name\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:21.923508Z","iopub.execute_input":"2022-07-05T13:29:21.924767Z","iopub.status.idle":"2022-07-05T13:29:21.951129Z","shell.execute_reply.started":"2022-07-05T13:29:21.924733Z","shell.execute_reply":"2022-07-05T13:29:21.950060Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!DOCTYPE html>\n<html>\n<head>\n<title>\n</title>\n<meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n</head>\n<body>\n<h1 style=\"color:DodgerBlue;\">\nsome more feature</h1>\n<ul style=\"color:DodgerBlue; list-style-type: square; padding: 10px; font-size:16px;\">\n    <li>Family Survival Rate : this will indicate survival rate of family(we club family using surname) </li>\n    <li>Ticket Survival Rate : this will indicate survival rate of perticular tikcets </li>\n</ul>\n</body>\n</html>","metadata":{}},{"cell_type":"code","source":"# Creating a list of families and tickets that are occuring in both training and test set\nnon_unique_families = [x for x in train_data[\"Family\"].unique() if x in test_data[\"Family\"].unique()]\nnon_unique_tickets = [x for x in train_data[\"Ticket\"].unique() if x in test_data[\"Ticket\"].unique()]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:21.952912Z","iopub.execute_input":"2022-07-05T13:29:21.953219Z","iopub.status.idle":"2022-07-05T13:29:22.083966Z","shell.execute_reply.started":"2022-07-05T13:29:21.953190Z","shell.execute_reply":"2022-07-05T13:29:22.083009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_family_survival_rate = train_data.groupby(\"Family\")[\"Survived\",\"Family\",\"FamilySize\"].median()\ndf_ticket_survival_rate = train_data.groupby(\"Ticket\")[\"Survived\",\"Ticket\", \"Ticket_Frequency\"].median()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.090672Z","iopub.execute_input":"2022-07-05T13:29:22.091241Z","iopub.status.idle":"2022-07-05T13:29:22.108360Z","shell.execute_reply.started":"2022-07-05T13:29:22.091198Z","shell.execute_reply":"2022-07-05T13:29:22.107555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"family_rates = {}\nticket_rates = {}","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.109559Z","iopub.execute_input":"2022-07-05T13:29:22.110036Z","iopub.status.idle":"2022-07-05T13:29:22.114799Z","shell.execute_reply.started":"2022-07-05T13:29:22.110005Z","shell.execute_reply":"2022-07-05T13:29:22.113509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(len(df_family_survival_rate)):\n    # Checking a family exists in both training and test set, and has members more than 1\n    if df_family_survival_rate.index[i] in non_unique_families and df_family_survival_rate.iloc[i, 1] > 1:\n        family_rates[df_family_survival_rate.index[i]] = df_family_survival_rate.iloc[i, 0]\n\n\nfor i in range(len(df_ticket_survival_rate)):\n    \n    # checking ticket avilable in both train and test set and has more than 1 member\n    if df_ticket_survival_rate.index[i] in non_unique_tickets and df_ticket_survival_rate.iloc[i,1]>1 :\n        ticket_rates[df_ticket_survival_rate.index[i]] = df_ticket_survival_rate.iloc[i,0]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.115841Z","iopub.execute_input":"2022-07-05T13:29:22.116598Z","iopub.status.idle":"2022-07-05T13:29:22.143948Z","shell.execute_reply.started":"2022-07-05T13:29:22.116564Z","shell.execute_reply":"2022-07-05T13:29:22.142734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_survival_rate = np.mean(train_data['Survived'])\n\n# family survival rate \ntrain_family_survival_rate = []\ntrain_family_survival_rate_NA = []\ntest_family_survival_rate = []\ntest_family_survival_rate_NA = []\n\nfor i in range(len(train_data)):\n    if train_data[\"Family\"][i] in family_rates :\n        train_family_survival_rate.append(family_rates[train_data[\"Family\"][i]])\n        train_family_survival_rate_NA.append(1)\n    else:\n        train_family_survival_rate.append(mean_survival_rate)\n        train_family_survival_rate_NA.append(0)\n        \nfor i in range(len(test_data)):\n    if test_data[\"Family\"].iloc[i] in family_rates :\n        test_family_survival_rate.append(family_rates[test_data[\"Family\"].iloc[i]])\n        test_family_survival_rate_NA.append(1)\n    else:\n        test_family_survival_rate.append(mean_survival_rate)\n        test_family_survival_rate_NA.append(0)\n        \ntrain_data[\"Famiily_survival_rates\"] = train_family_survival_rate\ntrain_data[\"Famiily_survival_rates_NA\"] = train_family_survival_rate_NA\ntest_data[\"Famiily_survival_rates\"] = test_family_survival_rate\ntest_data[\"Famiily_survival_rates_NA\"] = test_family_survival_rate_NA","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.145555Z","iopub.execute_input":"2022-07-05T13:29:22.145902Z","iopub.status.idle":"2022-07-05T13:29:22.170813Z","shell.execute_reply.started":"2022-07-05T13:29:22.145871Z","shell.execute_reply":"2022-07-05T13:29:22.169739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now we will do same for ticket_survival_rate\n\ntrain_ticket_survival_rate = []\ntrain_ticket_survival_rate_NA = []\ntest_ticket_survival_rate = []\ntest_ticket_survival_rate_NA = []\n\nfor i in range(len(train_data)):\n    if train_data[\"Ticket\"][i] in ticket_rates:\n        train_ticket_survival_rate.append(ticket_rates[train_data[\"Ticket\"][i]])\n        train_ticket_survival_rate_NA.append(1)\n    else:\n        train_ticket_survival_rate.append(mean_survival_rate)\n        train_ticket_survival_rate_NA.append(0)\n\nfor i in range(len(test_data)):\n    if test_data[\"Ticket\"].iloc[i] in ticket_rates:\n        test_ticket_survival_rate.append(ticket_rates[test_data[\"Ticket\"].iloc[i]])\n        test_ticket_survival_rate_NA.append(1)\n    else:\n        test_ticket_survival_rate.append(mean_survival_rate)\n        test_ticket_survival_rate_NA.append(0)\n\ntrain_data[\"Ticket_survival_rates\"] = train_ticket_survival_rate\ntrain_data[\"Ticket_survival_rates_NA\"] = train_ticket_survival_rate_NA\n\ntest_data[\"Ticket_survival_rates\"] = test_ticket_survival_rate\ntest_data[\"Ticket_survival_rates_NA\"] = test_ticket_survival_rate_NA","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.172080Z","iopub.execute_input":"2022-07-05T13:29:22.172904Z","iopub.status.idle":"2022-07-05T13:29:22.200796Z","shell.execute_reply.started":"2022-07-05T13:29:22.172855Z","shell.execute_reply":"2022-07-05T13:29:22.199737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.202366Z","iopub.execute_input":"2022-07-05T13:29:22.202938Z","iopub.status.idle":"2022-07-05T13:29:22.228533Z","shell.execute_reply.started":"2022-07-05T13:29:22.202904Z","shell.execute_reply":"2022-07-05T13:29:22.227677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!DOCTYPE html>\n<html>\n<head>\n<title>\n</title>\n<meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n</head>\n<body>\n<h1 style=\"color:DodgerBlue;\">\nFeature Transformation</h1>\n<ul style=\"color:DodgerBlue; list-style-type: square; padding: 10px; font-size:16px;\">\n    <li>Label Encoding Non-Numerical Features</li>\n    <li>One-Hot Encoding the Categorical Features </li>\n</ul>\n</body>\n</html> \n","metadata":{}},{"cell_type":"code","source":"non_numeric_features = [\"Embarked\",\"Deck\",\"Title\",\"FamilySize_Grouped\",\"Sex\",\"AgeBin\",\"FareBin\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.229781Z","iopub.execute_input":"2022-07-05T13:29:22.230302Z","iopub.status.idle":"2022-07-05T13:29:22.234387Z","shell.execute_reply.started":"2022-07-05T13:29:22.230270Z","shell.execute_reply":"2022-07-05T13:29:22.233551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder,OneHotEncoder","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.235775Z","iopub.execute_input":"2022-07-05T13:29:22.236122Z","iopub.status.idle":"2022-07-05T13:29:22.244645Z","shell.execute_reply.started":"2022-07-05T13:29:22.236094Z","shell.execute_reply":"2022-07-05T13:29:22.243625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datasets = [train_data, test_data]\nfor dataset in datasets:\n    for feature in non_numeric_features:\n        dataset[feature] = LabelEncoder().fit_transform(dataset[feature])","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.245782Z","iopub.execute_input":"2022-07-05T13:29:22.246076Z","iopub.status.idle":"2022-07-05T13:29:22.272080Z","shell.execute_reply.started":"2022-07-05T13:29:22.246043Z","shell.execute_reply":"2022-07-05T13:29:22.271196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# categorical features\ncat_features = ['Pclass', 'Sex', 'Deck', 'Embarked', 'Title', 'FamilySize_Grouped']\nencoded_features = []","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.273298Z","iopub.execute_input":"2022-07-05T13:29:22.273890Z","iopub.status.idle":"2022-07-05T13:29:22.283100Z","shell.execute_reply.started":"2022-07-05T13:29:22.273854Z","shell.execute_reply":"2022-07-05T13:29:22.282129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in datasets:\n    for feature in cat_features:\n        encoded_feat = OneHotEncoder().fit_transform(dataset[feature].values.reshape(-1,1)).toarray()\n        n = dataset[feature].nunique()\n        cols = ['{}_{}'.format(feature, n) for n in range(1, n + 1)]\n        encoded_df = pd.DataFrame(encoded_feat, columns=cols)\n        encoded_df.index = dataset.index\n        encoded_features.append(encoded_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.284364Z","iopub.execute_input":"2022-07-05T13:29:22.285335Z","iopub.status.idle":"2022-07-05T13:29:22.309112Z","shell.execute_reply.started":"2022-07-05T13:29:22.285300Z","shell.execute_reply":"2022-07-05T13:29:22.307838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.concat([train_data, *encoded_features[:6]], axis=1)\ntest_data = pd.concat([test_data, *encoded_features[6:]], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.310725Z","iopub.execute_input":"2022-07-05T13:29:22.311052Z","iopub.status.idle":"2022-07-05T13:29:22.319936Z","shell.execute_reply.started":"2022-07-05T13:29:22.311023Z","shell.execute_reply":"2022-07-05T13:29:22.318764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.321229Z","iopub.execute_input":"2022-07-05T13:29:22.321584Z","iopub.status.idle":"2022-07-05T13:29:22.355361Z","shell.execute_reply.started":"2022-07-05T13:29:22.321552Z","shell.execute_reply":"2022-07-05T13:29:22.354557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.356547Z","iopub.execute_input":"2022-07-05T13:29:22.357605Z","iopub.status.idle":"2022-07-05T13:29:22.375885Z","shell.execute_reply.started":"2022-07-05T13:29:22.357566Z","shell.execute_reply":"2022-07-05T13:29:22.374729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"drop_cols = [\"Deck\",\"Embarked\",\"Family\",\"FamilySize\",\"FamilySize_Grouped\",\"Name\",\"Parch\",\"PassengerId\",\n             \"Pclass\",\"Sex\",\"SibSp\",\"Ticket\",\"Title\",\"Ticket_survival_rates\",\"Famiily_survival_rates\",\n             \"Ticket_survival_rates_NA\",\"Famiily_survival_rates_NA\",\"Age\",\"Fare\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.377223Z","iopub.execute_input":"2022-07-05T13:29:22.378096Z","iopub.status.idle":"2022-07-05T13:29:22.383052Z","shell.execute_reply.started":"2022-07-05T13:29:22.378050Z","shell.execute_reply":"2022-07-05T13:29:22.382234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.drop(columns = drop_cols, inplace = True)\ntrain_data.drop(columns = drop_cols,inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.384259Z","iopub.execute_input":"2022-07-05T13:29:22.384805Z","iopub.status.idle":"2022-07-05T13:29:22.398834Z","shell.execute_reply.started":"2022-07-05T13:29:22.384772Z","shell.execute_reply":"2022-07-05T13:29:22.397943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.400307Z","iopub.execute_input":"2022-07-05T13:29:22.401034Z","iopub.status.idle":"2022-07-05T13:29:22.428849Z","shell.execute_reply.started":"2022-07-05T13:29:22.400996Z","shell.execute_reply":"2022-07-05T13:29:22.427741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.430242Z","iopub.execute_input":"2022-07-05T13:29:22.430610Z","iopub.status.idle":"2022-07-05T13:29:22.463772Z","shell.execute_reply.started":"2022-07-05T13:29:22.430578Z","shell.execute_reply":"2022-07-05T13:29:22.462847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!DOCTYPE html>\n<html>\n<head>\n<title>\n</title>\n<meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n</head>\n<body>\n<h1 style=\"color:DodgerBlue;\">\n5 Model Creation</h1>\n<ul style=\"color:DodgerBlue; list-style-type: square; padding: 10px; font-size:16px;\">\n    <li>Define X and Y</li>\n    <li>Scale using Standard Scaler</li>\n</body>\n</html>","metadata":{}},{"cell_type":"code","source":"# let us Define X and Y first\n\nX = train_data.iloc[:,1:]\nY = train_data[\"Survived\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.465084Z","iopub.execute_input":"2022-07-05T13:29:22.465383Z","iopub.status.idle":"2022-07-05T13:29:22.476112Z","shell.execute_reply.started":"2022-07-05T13:29:22.465357Z","shell.execute_reply":"2022-07-05T13:29:22.474708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let scale using standard scaler\n\nscaler = StandardScaler()\nX_scale = scaler.fit_transform(X)\ntest_data_scale = scaler.transform(test_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.477380Z","iopub.execute_input":"2022-07-05T13:29:22.478553Z","iopub.status.idle":"2022-07-05T13:29:22.496745Z","shell.execute_reply.started":"2022-07-05T13:29:22.478506Z","shell.execute_reply":"2022-07-05T13:29:22.495431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('X_train shape: {}'.format(X_scale.shape))\nprint('y_train shape: {}'.format(Y.shape))\nprint('X_test shape: {}'.format(test_data_scale.shape))","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.498418Z","iopub.execute_input":"2022-07-05T13:29:22.499123Z","iopub.status.idle":"2022-07-05T13:29:22.508319Z","shell.execute_reply.started":"2022-07-05T13:29:22.499078Z","shell.execute_reply":"2022-07-05T13:29:22.507293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<!DOCTYPE html>\n<html>\n<head>\n<title>\n</title>\n<meta name=\"viewport\" content=\"width=device-width, initial-scale=1\">\n</head>\n<body>\n<h1 style=\"color:DodgerBlue;\">\nXGBoost Model</h1>\n<ul style=\"color:DodgerBlue; list-style-type: square; padding: 10px; font-size:16px;\">\n    <li>Model Creation</li>\n    <li>Hyper Parameter tuning</li>\n<ul>\n</body>\n</html>","metadata":{}},{"cell_type":"code","source":"# XGBOOST Model\nimport xgboost as xgb ","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.509406Z","iopub.execute_input":"2022-07-05T13:29:22.510260Z","iopub.status.idle":"2022-07-05T13:29:22.519546Z","shell.execute_reply.started":"2022-07-05T13:29:22.510226Z","shell.execute_reply":"2022-07-05T13:29:22.518219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_model = xgb.XGBClassifier()","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.520891Z","iopub.execute_input":"2022-07-05T13:29:22.521864Z","iopub.status.idle":"2022-07-05T13:29:22.531792Z","shell.execute_reply.started":"2022-07-05T13:29:22.521828Z","shell.execute_reply":"2022-07-05T13:29:22.530549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"xgb_model.fit(X_scale,Y)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:22.533135Z","iopub.execute_input":"2022-07-05T13:29:22.533478Z","iopub.status.idle":"2022-07-05T13:29:23.035037Z","shell.execute_reply.started":"2022-07-05T13:29:22.533429Z","shell.execute_reply":"2022-07-05T13:29:23.033698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test_pred = xgb_model.predict(test_data_scale)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:23.037357Z","iopub.execute_input":"2022-07-05T13:29:23.037993Z","iopub.status.idle":"2022-07-05T13:29:23.054755Z","shell.execute_reply.started":"2022-07-05T13:29:23.037956Z","shell.execute_reply":"2022-07-05T13:29:23.053560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission --------- Submission.csv File\nsubmission_frame = pd.read_csv(\"../input/titanic/gender_submission.csv\")\nsubmission_frame[\"Survived\"] = y_test_pred\nsubmission_frame.to_csv(\"submission.csv\",index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:29:23.056592Z","iopub.execute_input":"2022-07-05T13:29:23.057174Z","iopub.status.idle":"2022-07-05T13:29:23.071089Z","shell.execute_reply.started":"2022-07-05T13:29:23.057140Z","shell.execute_reply":"2022-07-05T13:29:23.069735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Hyperparameter tunning with Gridsearchcv","metadata":{}},{"cell_type":"code","source":"# parameter grid\nparam_grid = {\n    \"max_depth\": [3, 4, 5, 7],\n    \"learning_rate\": [0.1, 0.01, 0.05],\n    \"gamma\": [0, 0.25, 1],\n    \"reg_lambda\": [0, 1, 10],\n    \"scale_pos_weight\": [1, 3, 5],\n    \"subsample\": [0.8],\n    \"colsample_bytree\": [0.5],\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:33:49.822736Z","iopub.execute_input":"2022-07-05T13:33:49.823112Z","iopub.status.idle":"2022-07-05T13:33:49.829102Z","shell.execute_reply.started":"2022-07-05T13:33:49.823083Z","shell.execute_reply":"2022-07-05T13:33:49.827995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\n\n# Init classifier\nxgb_cl = xgb.XGBClassifier(objective=\"binary:logistic\")\n\n# Init Grid Search\n#grid_cv = GridSearchCV(xgb_cl, param_grid, n_jobs=-1, cv=3, scoring=\"accuracy\")\n\n# Fit\n#_ = grid_cv.fit(X_scale,Y)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:58:14.845011Z","iopub.execute_input":"2022-07-05T13:58:14.845492Z","iopub.status.idle":"2022-07-05T13:58:18.095716Z","shell.execute_reply.started":"2022-07-05T13:58:14.845429Z","shell.execute_reply":"2022-07-05T13:58:18.094099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#grid_cv.best_params_","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:59:35.355140Z","iopub.execute_input":"2022-07-05T13:59:35.355595Z","iopub.status.idle":"2022-07-05T13:59:35.364406Z","shell.execute_reply.started":"2022-07-05T13:59:35.355552Z","shell.execute_reply":"2022-07-05T13:59:35.363157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"param_grid[\"colsample_bytree\"] = [0.5]\nparam_grid[\"subsample\"] = [0.8]\nparam_grid[\"scale_pos_weight\"] = [1]\n# Give new value ranges to other params\nparam_grid[\"gamma\"] = [1]\nparam_grid[\"max_depth\"] = [4]\nparam_grid[\"reg_lambda\"] = [0]\nparam_grid[\"learning_rate\"] = [0.05]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:59:47.746075Z","iopub.execute_input":"2022-07-05T13:59:47.746473Z","iopub.status.idle":"2022-07-05T13:59:47.752810Z","shell.execute_reply.started":"2022-07-05T13:59:47.746427Z","shell.execute_reply":"2022-07-05T13:59:47.751661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_cv_2 = GridSearchCV(xgb_cl, param_grid, \n                         cv=3, scoring=\"accuracy\", n_jobs=-1)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:59:51.411282Z","iopub.execute_input":"2022-07-05T13:59:51.411698Z","iopub.status.idle":"2022-07-05T13:59:51.416784Z","shell.execute_reply.started":"2022-07-05T13:59:51.411664Z","shell.execute_reply":"2022-07-05T13:59:51.415914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"grid_cv_2.fit(X_scale,Y)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:59:57.410878Z","iopub.execute_input":"2022-07-05T13:59:57.411270Z","iopub.status.idle":"2022-07-05T13:59:59.373721Z","shell.execute_reply.started":"2022-07-05T13:59:57.411237Z","shell.execute_reply":"2022-07-05T13:59:59.372541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test_pred =  grid_cv_2.predict(test_data_scale)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:00:02.928594Z","iopub.execute_input":"2022-07-05T14:00:02.928986Z","iopub.status.idle":"2022-07-05T14:00:02.940107Z","shell.execute_reply.started":"2022-07-05T14:00:02.928952Z","shell.execute_reply":"2022-07-05T14:00:02.939207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#submission --------- Submission.csv File\nsubmission_frame = pd.read_csv(\"../input/titanic/gender_submission.csv\")\nsubmission_frame[\"Survived\"] = y_test_pred\nsubmission_frame.to_csv(\"submission.csv\",index = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T14:00:08.196761Z","iopub.execute_input":"2022-07-05T14:00:08.197804Z","iopub.status.idle":"2022-07-05T14:00:08.209514Z","shell.execute_reply.started":"2022-07-05T14:00:08.197760Z","shell.execute_reply":"2022-07-05T14:00:08.208689Z"},"trusted":true},"execution_count":null,"outputs":[]}]}