{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"libaries required...","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-01T08:27:43.613467Z","iopub.execute_input":"2022-08-01T08:27:43.613865Z","iopub.status.idle":"2022-08-01T08:27:43.620532Z","shell.execute_reply.started":"2022-08-01T08:27:43.613814Z","shell.execute_reply":"2022-08-01T08:27:43.618759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#dataset\ndef read_data():\n    train_data = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\n    test_data = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\n    return train_data , test_data\n\ntrain_data , test_data = read_data()\ncombine = [train_data , test_data]\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:29:48.551537Z","iopub.execute_input":"2022-08-01T08:29:48.551972Z","iopub.status.idle":"2022-08-01T08:29:48.569280Z","shell.execute_reply.started":"2022-08-01T08:29:48.551938Z","shell.execute_reply":"2022-08-01T08:29:48.567922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train dataset\ntrain_data.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:31:06.891421Z","iopub.execute_input":"2022-08-01T08:31:06.891851Z","iopub.status.idle":"2022-08-01T08:31:06.909615Z","shell.execute_reply.started":"2022-08-01T08:31:06.891819Z","shell.execute_reply":"2022-08-01T08:31:06.908786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test dataset\ntest_data.head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:31:08.080828Z","iopub.execute_input":"2022-08-01T08:31:08.081848Z","iopub.status.idle":"2022-08-01T08:31:08.099277Z","shell.execute_reply.started":"2022-08-01T08:31:08.081809Z","shell.execute_reply":"2022-08-01T08:31:08.098375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"features of datasets","metadata":{}},{"cell_type":"code","source":"train_data.columns.values","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:31:50.256503Z","iopub.execute_input":"2022-08-01T08:31:50.256957Z","iopub.status.idle":"2022-08-01T08:31:50.263960Z","shell.execute_reply.started":"2022-08-01T08:31:50.256921Z","shell.execute_reply":"2022-08-01T08:31:50.263035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.columns.values","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:32:07.621945Z","iopub.execute_input":"2022-08-01T08:32:07.622550Z","iopub.status.idle":"2022-08-01T08:32:07.632168Z","shell.execute_reply.started":"2022-08-01T08:32:07.622498Z","shell.execute_reply":"2022-08-01T08:32:07.630932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()\nprint(\"----------------------------------------------------------\")\ntest_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:33:12.006264Z","iopub.execute_input":"2022-08-01T08:33:12.006707Z","iopub.status.idle":"2022-08-01T08:33:12.027909Z","shell.execute_reply.started":"2022-08-01T08:33:12.006672Z","shell.execute_reply":"2022-08-01T08:33:12.026989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check if null value present in our dataset or not\nprint(\"Train data missed values:\\n\")\nprint(train_data.isnull().sum())\nprint('\\n','_'*40 , '\\n')\nprint(\"Test data missed values:\")\nprint(test_data.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:34:15.260623Z","iopub.execute_input":"2022-08-01T08:34:15.261049Z","iopub.status.idle":"2022-08-01T08:34:15.274822Z","shell.execute_reply.started":"2022-08-01T08:34:15.261014Z","shell.execute_reply":"2022-08-01T08:34:15.273365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:34:49.570780Z","iopub.execute_input":"2022-08-01T08:34:49.571172Z","iopub.status.idle":"2022-08-01T08:34:49.608472Z","shell.execute_reply.started":"2022-08-01T08:34:49.571142Z","shell.execute_reply":"2022-08-01T08:34:49.607325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.describe(include=['O'])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:35:29.290788Z","iopub.execute_input":"2022-08-01T08:35:29.291174Z","iopub.status.idle":"2022-08-01T08:35:29.318454Z","shell.execute_reply.started":"2022-08-01T08:35:29.291144Z","shell.execute_reply":"2022-08-01T08:35:29.317522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we apply some EDA","metadata":{}},{"cell_type":"code","source":"f,ax=plt.subplots(1,2,figsize=(8,4))\ntrain_data['Survived'].replace({0:\"died\",1:\"survived\"}).value_counts().plot.pie(explode=[0,0.1],autopct='%1.1f%%',ax=ax[0],shadow=True)\nax[0].set_ylabel('')\nsns.countplot(x = train_data[\"Survived\"].replace({0:\"died\",1:\"survived\"}) , ax = ax[1])\nax[1].set_ylabel('')\nax[1].set_xlabel('')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:36:18.650486Z","iopub.execute_input":"2022-08-01T08:36:18.651440Z","iopub.status.idle":"2022-08-01T08:36:18.950408Z","shell.execute_reply.started":"2022-08-01T08:36:18.651402Z","shell.execute_reply":"2022-08-01T08:36:18.949330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Helper functions:\ndef survived_bar_plot(feature):\n    plt.figure(figsize = (6,4))\n    sns.barplot(data = train_data , x = feature , y = \"Survived\").set_title(f\"{feature} Vs Survived\")\n    plt.show()\ndef survived_table(feature):\n    return train_data[[feature, \"Survived\"]].groupby([feature], as_index=False).mean().sort_values(by='Survived', ascending=False).style.background_gradient(low=0.75,high=1)\ndef survived_hist_plot(feature):\n    plt.figure(figsize = (6,4))\n    sns.histplot(data = train_data , x = feature , hue = \"Survived\",binwidth=5,palette = sns.color_palette([\"yellow\" , \"green\"]) ,multiple = \"stack\" ).set_title(f\"{feature} Vs Survived\")\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:36:52.580528Z","iopub.execute_input":"2022-08-01T08:36:52.580930Z","iopub.status.idle":"2022-08-01T08:36:52.591538Z","shell.execute_reply.started":"2022-08-01T08:36:52.580895Z","shell.execute_reply":"2022-08-01T08:36:52.590405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"survived_table(\"Sex\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:37:27.890390Z","iopub.execute_input":"2022-08-01T08:37:27.890793Z","iopub.status.idle":"2022-08-01T08:37:27.983601Z","shell.execute_reply.started":"2022-08-01T08:37:27.890761Z","shell.execute_reply":"2022-08-01T08:37:27.982526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"survived_table(\"Pclass\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:37:49.690419Z","iopub.execute_input":"2022-08-01T08:37:49.690816Z","iopub.status.idle":"2022-08-01T08:37:49.710651Z","shell.execute_reply.started":"2022-08-01T08:37:49.690775Z","shell.execute_reply":"2022-08-01T08:37:49.709613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"survived_table(\"Embarked\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:38:18.531123Z","iopub.execute_input":"2022-08-01T08:38:18.531654Z","iopub.status.idle":"2022-08-01T08:38:18.554914Z","shell.execute_reply.started":"2022-08-01T08:38:18.531609Z","shell.execute_reply":"2022-08-01T08:38:18.552372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"survived_table(\"Parch\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:38:33.430417Z","iopub.execute_input":"2022-08-01T08:38:33.431142Z","iopub.status.idle":"2022-08-01T08:38:33.452662Z","shell.execute_reply.started":"2022-08-01T08:38:33.431107Z","shell.execute_reply":"2022-08-01T08:38:33.451369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"survived_table(\"SibSp\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:38:45.079821Z","iopub.execute_input":"2022-08-01T08:38:45.080247Z","iopub.status.idle":"2022-08-01T08:38:45.102381Z","shell.execute_reply.started":"2022-08-01T08:38:45.080213Z","shell.execute_reply":"2022-08-01T08:38:45.101317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_style(\"dark\") # to remove the grid.\nsurvived_hist_plot(\"Age\") # Note: This plot is stack plot.","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:39:01.010464Z","iopub.execute_input":"2022-08-01T08:39:01.011114Z","iopub.status.idle":"2022-08-01T08:39:01.441538Z","shell.execute_reply.started":"2022-08-01T08:39:01.011078Z","shell.execute_reply":"2022-08-01T08:39:01.440261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#finding correlation in dataset\nsns.set(rc = {'figure.figsize':(10,6)})\nsns.heatmap(train_data.corr(), annot = True, fmt='.2g',cmap= 'YlGnBu')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:39:32.880378Z","iopub.execute_input":"2022-08-01T08:39:32.880814Z","iopub.status.idle":"2022-08-01T08:39:33.419615Z","shell.execute_reply.started":"2022-08-01T08:39:32.880776Z","shell.execute_reply":"2022-08-01T08:39:33.418372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Pclass - Age - Survived\nplot , ax = plt.subplots(1 , 3 , figsize=(14,4))\nsns.histplot(data = train_data.loc[train_data[\"Pclass\"]==1] , x = \"Age\" , hue = \"Survived\",binwidth=5,ax = ax[0],palette = sns.color_palette([\"yellow\" , \"green\"]),multiple = \"stack\").set_title(\"1-Pclass\")\nsns.histplot(data = train_data.loc[train_data[\"Pclass\"]==2] , x = \"Age\" , hue = \"Survived\",binwidth=5,ax = ax[1],palette = sns.color_palette([\"yellow\" , \"green\"]),multiple = \"stack\").set_title(\"2-Pclass\")\nsns.histplot(data = train_data.loc[train_data[\"Pclass\"]==3] , x = \"Age\" , hue = \"Survived\",binwidth=5,ax = ax[2],palette = sns.color_palette([\"yellow\" , \"green\"]),multiple = \"stack\").set_title(\"3-Pclass\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:40:21.951370Z","iopub.execute_input":"2022-08-01T08:40:21.951785Z","iopub.status.idle":"2022-08-01T08:40:22.980172Z","shell.execute_reply.started":"2022-08-01T08:40:21.951751Z","shell.execute_reply":"2022-08-01T08:40:22.978774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Sex - Age - Survived\nplot , ax = plt.subplots(1 , 2 , figsize=(14,3))\nsns.histplot(data = train_data.loc[train_data[\"Sex\"]==\"male\"] , x = \"Age\" , hue = \"Survived\",binwidth=5,ax = ax[0],palette = sns.color_palette([\"yellow\" , \"green\"]),multiple = \"stack\").set_title(\"Males\")\nsns.histplot(data = train_data.loc[train_data[\"Sex\"]==\"female\"] , x = \"Age\" , hue = \"Survived\",binwidth=5,ax = ax[1],palette = sns.color_palette([\"yellow\" , \"green\"]),multiple = \"stack\").set_title(\"Females\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:41:14.165059Z","iopub.execute_input":"2022-08-01T08:41:14.165503Z","iopub.status.idle":"2022-08-01T08:41:14.852485Z","shell.execute_reply.started":"2022-08-01T08:41:14.165468Z","shell.execute_reply":"2022-08-01T08:41:14.851503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# drop sone not useful features\ntrain_data.drop(columns = [\"PassengerId\"] , inplace = True)\n\nfor dataset in combine:\n    dataset.drop(columns = [\"Ticket\" , \"Cabin\"] , inplace = True)\n    \nprint(\"Dropping features Done !!\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:41:53.310541Z","iopub.execute_input":"2022-08-01T08:41:53.310953Z","iopub.status.idle":"2022-08-01T08:41:53.322679Z","shell.execute_reply.started":"2022-08-01T08:41:53.310922Z","shell.execute_reply":"2022-08-01T08:41:53.321157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Converting Categorical Features to Numerical and Filling Missed Values:","metadata":{}},{"cell_type":"code","source":"#embarked\ntrain_data.Embarked.fillna(train_data.Embarked.dropna().max(), inplace=True)\nfor dataset in combine:\n    dataset['Embarked'] = dataset['Embarked'].dropna().map({'S':0,'C':1,'Q':2}).astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:42:25.670608Z","iopub.execute_input":"2022-08-01T08:42:25.671025Z","iopub.status.idle":"2022-08-01T08:42:25.684195Z","shell.execute_reply.started":"2022-08-01T08:42:25.670994Z","shell.execute_reply":"2022-08-01T08:42:25.683034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#sex\nfor dataset in combine:\n    dataset['Sex'] = dataset['Sex'].map( {'female': 1, 'male': 0} ).astype(int)   ","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:43:27.510837Z","iopub.execute_input":"2022-08-01T08:43:27.511272Z","iopub.status.idle":"2022-08-01T08:43:27.520422Z","shell.execute_reply.started":"2022-08-01T08:43:27.511240Z","shell.execute_reply":"2022-08-01T08:43:27.519320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#age\nguess_ages = np.zeros((2,3))\n\nfor dataset in combine:\n    for i in range(0, 2):\n        for j in range(0, 3):\n            guess_df = dataset[(dataset['Sex'] == i) & \\\n                                  (dataset['Pclass'] == j+1)]['Age'].dropna()\n\n            age_guess = guess_df.median()\n\n            # Convert random age float to nearest .5 age\n            guess_ages[i,j] = int( age_guess/0.5 + 0.5 ) * 0.5\n            \n    for i in range(0, 2):\n        for j in range(0, 3):\n            dataset.loc[ (dataset.Age.isnull()) & (dataset.Sex == i) & (dataset.Pclass == j+1),\\\n                    'Age'] = guess_ages[i,j]\n\n    dataset['Age'] = dataset['Age'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:44:07.730514Z","iopub.execute_input":"2022-08-01T08:44:07.730902Z","iopub.status.idle":"2022-08-01T08:44:07.775287Z","shell.execute_reply.started":"2022-08-01T08:44:07.730871Z","shell.execute_reply":"2022-08-01T08:44:07.774446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#fare\ntest_data.Fare.fillna(test_data.Fare.dropna().median() , inplace= True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:44:31.705447Z","iopub.execute_input":"2022-08-01T08:44:31.705850Z","iopub.status.idle":"2022-08-01T08:44:31.712034Z","shell.execute_reply.started":"2022-08-01T08:44:31.705817Z","shell.execute_reply":"2022-08-01T08:44:31.710902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data.isnull().sum())\nprint(\"-*\" * 30)\nprint(test_data.isnull().sum())","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:45:14.859966Z","iopub.execute_input":"2022-08-01T08:45:14.861162Z","iopub.status.idle":"2022-08-01T08:45:14.873807Z","shell.execute_reply.started":"2022-08-01T08:45:14.861117Z","shell.execute_reply":"2022-08-01T08:45:14.871998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:45:35.300500Z","iopub.execute_input":"2022-08-01T08:45:35.300912Z","iopub.status.idle":"2022-08-01T08:45:35.316182Z","shell.execute_reply.started":"2022-08-01T08:45:35.300882Z","shell.execute_reply":"2022-08-01T08:45:35.314772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Age Banding\ntrain_data['AgeBand'] = pd.cut(train_data['Age'], 5)\ntrain_data[['AgeBand', 'Survived']].groupby(['AgeBand'], as_index=False).mean().sort_values(by='AgeBand', ascending=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:46:11.475540Z","iopub.execute_input":"2022-08-01T08:46:11.475932Z","iopub.status.idle":"2022-08-01T08:46:11.507811Z","shell.execute_reply.started":"2022-08-01T08:46:11.475901Z","shell.execute_reply":"2022-08-01T08:46:11.506598Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:    \n    dataset.loc[ dataset['Age'] <= 16, 'Age'] = 0\n    dataset.loc[(dataset['Age'] > 16) & (dataset['Age'] <= 32), 'Age'] = 1\n    dataset.loc[(dataset['Age'] > 32) & (dataset['Age'] <= 48), 'Age'] = 2\n    dataset.loc[(dataset['Age'] > 48) & (dataset['Age'] <= 64), 'Age'] = 3\n    dataset.loc[ dataset['Age'] > 64, 'Age']\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:46:24.372665Z","iopub.execute_input":"2022-08-01T08:46:24.373063Z","iopub.status.idle":"2022-08-01T08:46:24.400789Z","shell.execute_reply.started":"2022-08-01T08:46:24.373032Z","shell.execute_reply":"2022-08-01T08:46:24.399912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.drop(['AgeBand'], axis=1 , inplace = True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:47:03.655762Z","iopub.execute_input":"2022-08-01T08:47:03.656416Z","iopub.status.idle":"2022-08-01T08:47:03.663232Z","shell.execute_reply.started":"2022-08-01T08:47:03.656369Z","shell.execute_reply":"2022-08-01T08:47:03.662267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fare Banding\ntrain_data['FareBand'] = pd.qcut(train_data['Fare'], 4)\ntrain_data[['FareBand', 'Survived']].groupby(['FareBand'], as_index=False).mean().sort_values(by='FareBand', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:47:47.419911Z","iopub.execute_input":"2022-08-01T08:47:47.420297Z","iopub.status.idle":"2022-08-01T08:47:47.442551Z","shell.execute_reply.started":"2022-08-01T08:47:47.420266Z","shell.execute_reply":"2022-08-01T08:47:47.441427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset.loc[ dataset['Fare'] <= 7.91, 'Fare'] = 0\n    dataset.loc[(dataset['Fare'] > 7.91) & (dataset['Fare'] <= 14.454), 'Fare'] = 1\n    dataset.loc[(dataset['Fare'] > 14.454) & (dataset['Fare'] <= 31), 'Fare']   = 2\n    dataset.loc[ dataset['Fare'] > 31, 'Fare'] = 3\n    dataset['Fare'] = dataset['Fare'].astype(int)\n\ntrain_data.drop(['FareBand'], axis=1 , inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:48:08.015813Z","iopub.execute_input":"2022-08-01T08:48:08.016225Z","iopub.status.idle":"2022-08-01T08:48:08.035816Z","shell.execute_reply.started":"2022-08-01T08:48:08.016196Z","shell.execute_reply":"2022-08-01T08:48:08.034559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:48:28.310741Z","iopub.execute_input":"2022-08-01T08:48:28.311729Z","iopub.status.idle":"2022-08-01T08:48:28.330384Z","shell.execute_reply.started":"2022-08-01T08:48:28.311682Z","shell.execute_reply":"2022-08-01T08:48:28.329223Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Family Size of passengers\nfor dataset in combine:\n    dataset['FamilySize'] = dataset['SibSp'] + dataset['Parch'] + 1\n\ntrain_data.drop(['Parch', 'SibSp'], axis=1 , inplace = True)\ntest_data.drop(['Parch', 'SibSp'], axis=1 , inplace = True)    \n\ntrain_data[['FamilySize', 'Survived']].groupby(['FamilySize'], as_index=False).mean().sort_values(by='Survived', ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:49:29.260956Z","iopub.execute_input":"2022-08-01T08:49:29.261428Z","iopub.status.idle":"2022-08-01T08:49:29.285139Z","shell.execute_reply.started":"2022-08-01T08:49:29.261394Z","shell.execute_reply":"2022-08-01T08:49:29.284030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create new feature of family size\nfor dataset in combine:\n    dataset['Single'] = dataset['FamilySize'].map(lambda s: 1 if s == 1 else 0)\n    dataset['SmallF'] = dataset['FamilySize'].map(lambda s: 1 if  s == 2  else 0)\n    dataset['MedF'] = dataset['FamilySize'].map(lambda s: 1 if 3 <= s <= 4 else 0)\n    dataset['LargeF'] = dataset['FamilySize'].map(lambda s: 1 if s >= 5 else 0)\n    \ntrain_data.drop(columns = [\"FamilySize\"] , inplace = True)\ntest_data.drop(columns = [\"FamilySize\"] , inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:49:44.079811Z","iopub.execute_input":"2022-08-01T08:49:44.080223Z","iopub.status.idle":"2022-08-01T08:49:44.101497Z","shell.execute_reply.started":"2022-08-01T08:49:44.080188Z","shell.execute_reply":"2022-08-01T08:49:44.100507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Name Title\nfor dataset in combine:\n    dataset['Title'] = dataset.Name.str.extract(' ([A-Za-z]+)\\.', expand=False)\n\npd.crosstab(train_data['Title'], train_data['Sex'])","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:50:04.120644Z","iopub.execute_input":"2022-08-01T08:50:04.121044Z","iopub.status.idle":"2022-08-01T08:50:04.157005Z","shell.execute_reply.started":"2022-08-01T08:50:04.121013Z","shell.execute_reply":"2022-08-01T08:50:04.155680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combine:\n    dataset['Title'] = dataset['Title'].replace(['Lady', 'Countess','Capt', 'Col',\\\n    'Don', 'Dr', 'Major', 'Rev', 'Sir', 'Jonkheer', 'Dona'], 'Rare')\n\n    dataset['Title'] = dataset['Title'].replace('Mlle', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Ms', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Mme', 'Mrs')\n    \ntrain_data[['Title', 'Survived']].groupby(['Title'], as_index=False).mean()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:50:20.440843Z","iopub.execute_input":"2022-08-01T08:50:20.441254Z","iopub.status.idle":"2022-08-01T08:50:20.468642Z","shell.execute_reply.started":"2022-08-01T08:50:20.441222Z","shell.execute_reply":"2022-08-01T08:50:20.467731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"title_mapping = {\"Mr\": 1, \"Miss\": 2, \"Mrs\": 3, \"Master\": 4, \"Rare\": 5}\nfor dataset in combine:\n    dataset['Title'] = dataset['Title'].map(title_mapping)\n    dataset['Title'] = dataset['Title'].fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:50:33.991359Z","iopub.execute_input":"2022-08-01T08:50:33.991760Z","iopub.status.idle":"2022-08-01T08:50:34.002453Z","shell.execute_reply.started":"2022-08-01T08:50:33.991728Z","shell.execute_reply":"2022-08-01T08:50:34.001282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.drop(['Name'], axis=1 , inplace = True)\ntest_data.drop(['Name'], axis=1 , inplace = True) ","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:50:48.170758Z","iopub.execute_input":"2022-08-01T08:50:48.171167Z","iopub.status.idle":"2022-08-01T08:50:48.180329Z","shell.execute_reply.started":"2022-08-01T08:50:48.171134Z","shell.execute_reply":"2022-08-01T08:50:48.179134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:50:59.789757Z","iopub.execute_input":"2022-08-01T08:50:59.790150Z","iopub.status.idle":"2022-08-01T08:50:59.806572Z","shell.execute_reply.started":"2022-08-01T08:50:59.790119Z","shell.execute_reply":"2022-08-01T08:50:59.805361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:51:18.610661Z","iopub.execute_input":"2022-08-01T08:51:18.611080Z","iopub.status.idle":"2022-08-01T08:51:18.627956Z","shell.execute_reply.started":"2022-08-01T08:51:18.611048Z","shell.execute_reply":"2022-08-01T08:51:18.626622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Apply Modeling ","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier, AdaBoostClassifier, GradientBoostingClassifier, ExtraTreesClassifier, VotingClassifier\nfrom sklearn.discriminant_analysis import LinearDiscriminantAnalysis\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.model_selection import GridSearchCV, cross_val_score, StratifiedKFold, learning_curve","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:51:57.490929Z","iopub.execute_input":"2022-08-01T08:51:57.491341Z","iopub.status.idle":"2022-08-01T08:51:57.843691Z","shell.execute_reply.started":"2022-08-01T08:51:57.491289Z","shell.execute_reply":"2022-08-01T08:51:57.842563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preparing Data For Training:\n\nY_train = train_data[\"Survived\"]\nX_train = train_data.drop(labels = [\"Survived\"],axis = 1)\nTest = test_data.drop(labels = [\"PassengerId\"],axis = 1)\nprint(f\"X_train shape is = {X_train.shape}\" )\nprint(f\"Y_train shape is = {Y_train.shape}\" )\nprint(f\"Test shape is = {Test.shape}\" )","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:52:21.790751Z","iopub.execute_input":"2022-08-01T08:52:21.791167Z","iopub.status.idle":"2022-08-01T08:52:21.802335Z","shell.execute_reply.started":"2022-08-01T08:52:21.791133Z","shell.execute_reply":"2022-08-01T08:52:21.800989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cross validate model with Kfold stratified cross val\nkfold = StratifiedKFold(n_splits=10)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:52:54.210950Z","iopub.execute_input":"2022-08-01T08:52:54.211445Z","iopub.status.idle":"2022-08-01T08:52:54.216981Z","shell.execute_reply.started":"2022-08-01T08:52:54.211409Z","shell.execute_reply":"2022-08-01T08:52:54.215774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Modeling step Test differents algorithms \nrandom_state = 2\nclassifiers = []\nclassifiers.append(SVC(random_state=random_state))\nclassifiers.append(DecisionTreeClassifier(random_state=random_state))\nclassifiers.append(AdaBoostClassifier(DecisionTreeClassifier(random_state=random_state),random_state=random_state,learning_rate=0.1))\nclassifiers.append(RandomForestClassifier(random_state=random_state))\nclassifiers.append(ExtraTreesClassifier(random_state=random_state))\nclassifiers.append(GradientBoostingClassifier(random_state=random_state))\nclassifiers.append(MLPClassifier(random_state=random_state))\nclassifiers.append(KNeighborsClassifier())\nclassifiers.append(LogisticRegression(random_state = random_state))\nclassifiers.append(LinearDiscriminantAnalysis())\n\ncv_results = []\nfor classifier in classifiers :\n    cv_results.append(cross_val_score(classifier, X_train, y = Y_train, scoring = \"accuracy\", cv = kfold, n_jobs=4))\n\ncv_means = []\ncv_std = []\nfor cv_result in cv_results:\n    cv_means.append(cv_result.mean())\n    cv_std.append(cv_result.std())\n\ncv_res = pd.DataFrame({\"CrossValMeans\":cv_means,\"CrossValerrors\": cv_std,\"Algorithm\":[\"SVC\",\"DecisionTree\",\"AdaBoost\",\n\"RandomForest\",\"ExtraTrees\",\"GradientBoosting\",\"MultipleLayerPerceptron\",\"KNeighboors\",\"LogisticRegression\",\"LinearDiscriminantAnalysis\"]})\n\ng = sns.barplot(\"CrossValMeans\",\"Algorithm\",data = cv_res, palette=\"Set3\",orient = \"h\",**{'xerr':cv_std})\ng.set_xlabel(\"Mean Accuracy\")\ng = g.set_title(\"Cross validation scores\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:53:23.216352Z","iopub.execute_input":"2022-08-01T08:53:23.216878Z","iopub.status.idle":"2022-08-01T08:53:31.151262Z","shell.execute_reply.started":"2022-08-01T08:53:23.216838Z","shell.execute_reply":"2022-08-01T08:53:31.149840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### META MODELING  WITH ADABOOST, RF, EXTRATREES and GRADIENTBOOSTING\n\n# Adaboost\nDTC = DecisionTreeClassifier()\n\nadaDTC = AdaBoostClassifier(DTC, random_state=7)\n\nada_param_grid = {\"base_estimator__criterion\" : [\"gini\", \"entropy\"],\n              \"base_estimator__splitter\" :   [\"best\", \"random\"],\n              \"algorithm\" : [\"SAMME\",\"SAMME.R\"],\n              \"n_estimators\" :[1,2],\n              \"learning_rate\":  [0.0001, 0.001, 0.01, 0.1, 0.2, 0.3,1.5]}\n\ngsadaDTC = GridSearchCV(adaDTC,param_grid = ada_param_grid, cv=kfold, scoring=\"accuracy\", n_jobs= 4, verbose = 1)\n\ngsadaDTC.fit(X_train,Y_train)\n\nada_best = gsadaDTC.best_estimator_\n\ngsadaDTC.best_score_","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:54:17.871695Z","iopub.execute_input":"2022-08-01T08:54:17.872126Z","iopub.status.idle":"2022-08-01T08:54:22.760405Z","shell.execute_reply.started":"2022-08-01T08:54:17.872089Z","shell.execute_reply":"2022-08-01T08:54:22.759138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#ExtraTrees \nExtC = ExtraTreesClassifier()\n\n\n## Search grid for optimal parameters\nex_param_grid = {\"max_depth\": [None],\n              \"max_features\": [1, 3, 10],\n              \"min_samples_split\": [2, 3, 10],\n              \"min_samples_leaf\": [1, 3, 10],\n              \"bootstrap\": [False],\n              \"n_estimators\" :[100,300],\n              \"criterion\": [\"gini\"]}\n\n\ngsExtC = GridSearchCV(ExtC,param_grid = ex_param_grid, cv=kfold, scoring=\"accuracy\", n_jobs= 4, verbose = 1)\n\ngsExtC.fit(X_train,Y_train)\n\nExtC_best = gsExtC.best_estimator_\n\n# Best score\ngsExtC.best_score_","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:54:52.212922Z","iopub.execute_input":"2022-08-01T08:54:52.213645Z","iopub.status.idle":"2022-08-01T08:56:05.311184Z","shell.execute_reply.started":"2022-08-01T08:54:52.213608Z","shell.execute_reply":"2022-08-01T08:56:05.309906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# RFC Parameters tunning \nRFC = RandomForestClassifier()\n\n\n## Search grid for optimal parameters\nrf_param_grid = {\"max_depth\": [None],\n              \"max_features\": [1, 3, 10],\n              \"min_samples_split\": [2, 3, 10],\n              \"min_samples_leaf\": [1, 3, 10],\n              \"bootstrap\": [False],\n              \"n_estimators\" :[100,300],\n              \"criterion\": [\"gini\"]}\n\n\ngsRFC = GridSearchCV(RFC,param_grid = rf_param_grid, cv=kfold, scoring=\"accuracy\", n_jobs= 4, verbose = 1)\n\ngsRFC.fit(X_train,Y_train)\n\nRFC_best = gsRFC.best_estimator_\n\n# Best score\ngsRFC.best_score_","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:56:05.313518Z","iopub.execute_input":"2022-08-01T08:56:05.314061Z","iopub.status.idle":"2022-08-01T08:57:22.162486Z","shell.execute_reply.started":"2022-08-01T08:56:05.314014Z","shell.execute_reply":"2022-08-01T08:57:22.161666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Gradient boosting tunning\n\nGBC = GradientBoostingClassifier()\ngb_param_grid = {'loss' : [\"deviance\"],\n              'n_estimators' : [100,200,300],\n              'learning_rate': [0.1, 0.05, 0.01],\n              'max_depth': [4, 8],\n              'min_samples_leaf': [100,150],\n              'max_features': [0.3, 0.1] \n              }\n\ngsGBC = GridSearchCV(GBC,param_grid = gb_param_grid, cv=kfold, scoring=\"accuracy\", n_jobs= 4, verbose = 1)\n\ngsGBC.fit(X_train,Y_train)\n\nGBC_best = gsGBC.best_estimator_\n\n# Best score\ngsGBC.best_score_\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:57:22.163610Z","iopub.execute_input":"2022-08-01T08:57:22.164590Z","iopub.status.idle":"2022-08-01T08:58:00.182016Z","shell.execute_reply.started":"2022-08-01T08:57:22.164555Z","shell.execute_reply":"2022-08-01T08:58:00.180669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### SVC classifier\nSVMC = SVC(probability=True)\nsvc_param_grid = {'kernel': ['rbf'], \n                  'gamma': [ 0.001, 0.01, 0.1, 1],\n                  'C': [1, 10, 50, 100,200,300, 1000]}\n\ngsSVMC = GridSearchCV(SVMC,param_grid = svc_param_grid, cv=kfold, scoring=\"accuracy\", n_jobs= 4, verbose = 1)\n\ngsSVMC.fit(X_train,Y_train)\n\nSVMC_best = gsSVMC.best_estimator_\n\n# Best score\ngsSVMC.best_score_","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:58:00.184544Z","iopub.execute_input":"2022-08-01T08:58:00.184901Z","iopub.status.idle":"2022-08-01T08:58:29.914074Z","shell.execute_reply.started":"2022-08-01T08:58:00.184870Z","shell.execute_reply":"2022-08-01T08:58:29.912952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_learning_curve(estimator, title, X, y, ylim=None, cv=None,\n                        n_jobs=-1, train_sizes=np.linspace(.1, 1.0, 5)):\n    \"\"\"Generate a simple plot of the test and training learning curve\"\"\"\n    plt.figure()\n    plt.title(title)\n    if ylim is not None:\n        plt.ylim(*ylim)\n    plt.xlabel(\"Training examples\")\n    plt.ylabel(\"Score\")\n    train_sizes, train_scores, test_scores = learning_curve(\n        estimator, X, y, cv=cv, n_jobs=n_jobs, train_sizes=train_sizes)\n    train_scores_mean = np.mean(train_scores, axis=1)\n    train_scores_std = np.std(train_scores, axis=1)\n    test_scores_mean = np.mean(test_scores, axis=1)\n    test_scores_std = np.std(test_scores, axis=1)\n    plt.grid()\n\n    plt.fill_between(train_sizes, train_scores_mean - train_scores_std,\n                     train_scores_mean + train_scores_std, alpha=0.1,\n                     color=\"r\")\n    plt.fill_between(train_sizes, test_scores_mean - test_scores_std,\n                     test_scores_mean + test_scores_std, alpha=0.1, color=\"g\")\n    plt.plot(train_sizes, train_scores_mean, 'o-', color=\"r\",\n             label=\"Training score\")\n    plt.plot(train_sizes, test_scores_mean, 'o-', color=\"g\",\n             label=\"Cross-validation score\")\n\n    plt.legend(loc=\"best\")\n    return plt\n\ng = plot_learning_curve(gsRFC.best_estimator_,\"RF mearning curves\",X_train,Y_train,cv=kfold)\ng = plot_learning_curve(gsExtC.best_estimator_,\"ExtraTrees learning curves\",X_train,Y_train,cv=kfold)\ng = plot_learning_curve(gsSVMC.best_estimator_,\"SVC learning curves\",X_train,Y_train,cv=kfold)\ng = plot_learning_curve(gsadaDTC.best_estimator_,\"AdaBoost learning curves\",X_train,Y_train,cv=kfold)\ng = plot_learning_curve(gsGBC.best_estimator_,\"GradientBoosting learning curves\",X_train,Y_train,cv=kfold)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:58:29.915974Z","iopub.execute_input":"2022-08-01T08:58:29.916451Z","iopub.status.idle":"2022-08-01T08:58:50.980323Z","shell.execute_reply.started":"2022-08-01T08:58:29.916408Z","shell.execute_reply":"2022-08-01T08:58:50.979156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_Survived_RFC = pd.Series(RFC_best.predict(Test), name=\"RFC\")\ntest_Survived_ExtC = pd.Series(ExtC_best.predict(Test), name=\"ExtC\")\ntest_Survived_SVMC = pd.Series(SVMC_best.predict(Test), name=\"SVC\")\ntest_Survived_AdaC = pd.Series(ada_best.predict(Test), name=\"Ada\")\ntest_Survived_GBC = pd.Series(GBC_best.predict(Test), name=\"GBC\")\n\n\n\n# Concatenate all classifier results\nensemble_results = pd.concat([test_Survived_RFC,test_Survived_ExtC,test_Survived_AdaC,test_Survived_GBC, test_Survived_SVMC],axis=1)\n\n\ng= sns.heatmap(ensemble_results.corr(),annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:58:50.981683Z","iopub.execute_input":"2022-08-01T08:58:50.982032Z","iopub.status.idle":"2022-08-01T08:58:51.423281Z","shell.execute_reply.started":"2022-08-01T08:58:50.981991Z","shell.execute_reply":"2022-08-01T08:58:51.422199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concatenate all classifier results\nensemble_results = pd.concat([test_Survived_RFC,test_Survived_ExtC,test_Survived_AdaC,test_Survived_GBC, test_Survived_SVMC],axis=1)\n\n\ng= sns.heatmap(ensemble_results.corr(),annot=True)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:58:51.424626Z","iopub.execute_input":"2022-08-01T08:58:51.424980Z","iopub.status.idle":"2022-08-01T08:58:51.768623Z","shell.execute_reply.started":"2022-08-01T08:58:51.424948Z","shell.execute_reply":"2022-08-01T08:58:51.767203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"votingC = VotingClassifier(estimators=[('rfc', RFC_best), ('extc', ExtC_best),\n('svc', SVMC_best), ('adac',ada_best),('gbc',GBC_best)], voting='soft', n_jobs=4)\n\nvotingC = votingC.fit(X_train, Y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T08:58:51.769895Z","iopub.execute_input":"2022-08-01T08:58:51.770236Z","iopub.status.idle":"2022-08-01T08:58:52.554811Z","shell.execute_reply.started":"2022-08-01T08:58:51.770204Z","shell.execute_reply":"2022-08-01T08:58:52.553908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_Survived = pd.Series(votingC.predict(Test), name=\"Survived\")\n\nresults = pd.concat([test_data.PassengerId,test_Survived],axis=1)\n\nresults.to_csv(\"ensemble_python_voting.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:29:15.099257Z","iopub.execute_input":"2022-08-01T09:29:15.100285Z","iopub.status.idle":"2022-08-01T09:29:15.216371Z","shell.execute_reply.started":"2022-08-01T09:29:15.100234Z","shell.execute_reply":"2022-08-01T09:29:15.215200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### References:","metadata":{}},{"cell_type":"markdown","source":"## 1.https://www.kaggle.com/code/odaymourad/detailed-and-typical-solution-ensemble-modeling","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}