{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-27T21:34:15.516252Z","iopub.execute_input":"2022-07-27T21:34:15.516639Z","iopub.status.idle":"2022-07-27T21:34:15.536553Z","shell.execute_reply.started":"2022-07-27T21:34:15.516608Z","shell.execute_reply":"2022-07-27T21:34:15.535616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Let's start our project**","metadata":{}},{"cell_type":"code","source":"# import our lib\nimport pandas as pd\nimport numpy as np\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:15.541391Z","iopub.execute_input":"2022-07-27T21:34:15.542002Z","iopub.status.idle":"2022-07-27T21:34:15.548110Z","shell.execute_reply.started":"2022-07-27T21:34:15.541967Z","shell.execute_reply":"2022-07-27T21:34:15.546732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# read the data \ndata_train = pd.read_csv('../input/titanic/train.csv')\ndata_test = pd.read_csv('../input/titanic/test.csv')\ntrain = data_train.copy()\ntest = data_test.copy()\n\nall_data = [train , test]","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:15.586712Z","iopub.execute_input":"2022-07-27T21:34:15.587436Z","iopub.status.idle":"2022-07-27T21:34:15.608986Z","shell.execute_reply.started":"2022-07-27T21:34:15.587399Z","shell.execute_reply":"2022-07-27T21:34:15.607312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# The first thing is : split our data ","metadata":{}},{"cell_type":"markdown","source":"Before doing anything, we first need to partition our data before  it to avoid the case of **data leakage**.\nBut we will divide our data equally based on which column has the strongest correlation with the survival column\n\nso let's do that ","metadata":{}},{"cell_type":"markdown","source":"First we need to know which column has the strongest correlation with the survival column\nWe can see that the best column is the **Fare**","metadata":{}},{"cell_type":"code","source":"train.corr()['Survived'].sort_values(ascending = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:15.613487Z","iopub.execute_input":"2022-07-27T21:34:15.613873Z","iopub.status.idle":"2022-07-27T21:34:15.627280Z","shell.execute_reply.started":"2022-07-27T21:34:15.613842Z","shell.execute_reply":"2022-07-27T21:34:15.625992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"But because there are many values in this column, the division will not be exact,\nSo we need to cut the values of this column and then split the data","metadata":{}},{"cell_type":"code","source":"train['Fare_cut'] = pd.qcut(train['Fare'], 4)\ntrain[['Fare_cut','Survived']].groupby(['Fare_cut']).mean().sort_values(by='Survived', ascending= False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:15.629142Z","iopub.execute_input":"2022-07-27T21:34:15.630561Z","iopub.status.idle":"2022-07-27T21:34:15.652291Z","shell.execute_reply.started":"2022-07-27T21:34:15.630511Z","shell.execute_reply":"2022-07-27T21:34:15.650699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we're going to cut the fare column","metadata":{}},{"cell_type":"code","source":"def Fare_cut(fare) :\n    if fare <= 7.91 :\n        return 0\n    elif fare <= 14.454 :\n        return 1\n    elif fare <= 31.0 :\n        return 2\n    else :\n        return 3\n    \ntrain['Fare_cut_num'] = train['Fare'].apply(Fare_cut)\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:15.682059Z","iopub.execute_input":"2022-07-27T21:34:15.682517Z","iopub.status.idle":"2022-07-27T21:34:15.709538Z","shell.execute_reply.started":"2022-07-27T21:34:15.682483Z","shell.execute_reply":"2022-07-27T21:34:15.708318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we will split our Data using : **StratifiedShuffleSplit**","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedShuffleSplit\n\n\nsplit = StratifiedShuffleSplit(n_splits=1, test_size=0.2, random_state=42)\nfor train_index, test_index  in split.split(train, train[\"Fare_cut_num\"]):\n    train_set = train.iloc[train_index]\n    test_set = train.iloc[test_index]\n\ntrain_test_data = [train_set, test_set]    \n\nprint('The train set Size is :', train_set.shape)\nprint('The test set Size is :', test_set.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:15.748812Z","iopub.execute_input":"2022-07-27T21:34:15.749201Z","iopub.status.idle":"2022-07-27T21:34:15.761428Z","shell.execute_reply.started":"2022-07-27T21:34:15.749169Z","shell.execute_reply":"2022-07-27T21:34:15.760053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we are ready to extract information and visual data from our data without worrying about data leakage","metadata":{}},{"cell_type":"markdown","source":"But before that let's remove the original fare column from our data","metadata":{}},{"cell_type":"code","source":"for dataset in train_test_data :\n    dataset.drop(['Fare', 'Fare_cut'], axis=1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:15.795194Z","iopub.execute_input":"2022-07-27T21:34:15.795852Z","iopub.status.idle":"2022-07-27T21:34:15.804851Z","shell.execute_reply.started":"2022-07-27T21:34:15.795804Z","shell.execute_reply":"2022-07-27T21:34:15.803668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:15.863790Z","iopub.execute_input":"2022-07-27T21:34:15.864821Z","iopub.status.idle":"2022-07-27T21:34:15.882980Z","shell.execute_reply.started":"2022-07-27T21:34:15.864768Z","shell.execute_reply":"2022-07-27T21:34:15.881635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_set.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:15.902936Z","iopub.execute_input":"2022-07-27T21:34:15.903393Z","iopub.status.idle":"2022-07-27T21:34:15.921127Z","shell.execute_reply.started":"2022-07-27T21:34:15.903359Z","shell.execute_reply":"2022-07-27T21:34:15.919593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let's take a quick look at our data","metadata":{}},{"cell_type":"code","source":"train_set.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:16.686325Z","iopub.execute_input":"2022-07-27T21:34:16.687758Z","iopub.status.idle":"2022-07-27T21:34:16.707941Z","shell.execute_reply.started":"2022-07-27T21:34:16.687707Z","shell.execute_reply":"2022-07-27T21:34:16.706484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Train set : ', train_set.shape)\nprint('Test set : ', test_set.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:16.710046Z","iopub.execute_input":"2022-07-27T21:34:16.710450Z","iopub.status.idle":"2022-07-27T21:34:16.716929Z","shell.execute_reply.started":"2022-07-27T21:34:16.710417Z","shell.execute_reply":"2022-07-27T21:34:16.715452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('train info :  \\n')\nprint(train_set.info())\nprint('-'*70)\nprint('test info : \\n')\nprint(test_set.info())\nprint('-'*70)\ntrain_set.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:16.719166Z","iopub.execute_input":"2022-07-27T21:34:16.720935Z","iopub.status.idle":"2022-07-27T21:34:16.775837Z","shell.execute_reply.started":"2022-07-27T21:34:16.720775Z","shell.execute_reply":"2022-07-27T21:34:16.774451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set.describe(include=['O'])","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:16.778830Z","iopub.execute_input":"2022-07-27T21:34:16.779251Z","iopub.status.idle":"2022-07-27T21:34:16.810116Z","shell.execute_reply.started":"2022-07-27T21:34:16.779190Z","shell.execute_reply":"2022-07-27T21:34:16.808616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# exploratory data analysis","metadata":{}},{"cell_type":"markdown","source":"Let's get some information out and draw some plots","metadata":{}},{"cell_type":"markdown","source":"We notice that we have some missing values in the **Age** and **Cabin** columns and **Embarked** column","metadata":{}},{"cell_type":"code","source":"train_set.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:16.812008Z","iopub.execute_input":"2022-07-27T21:34:16.812719Z","iopub.status.idle":"2022-07-27T21:34:16.826963Z","shell.execute_reply.started":"2022-07-27T21:34:16.812678Z","shell.execute_reply":"2022-07-27T21:34:16.825736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's draw the above","metadata":{}},{"cell_type":"code","source":"sns.heatmap(train_set.isna(), yticklabels= False, cbar= False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:16.828471Z","iopub.execute_input":"2022-07-27T21:34:16.829368Z","iopub.status.idle":"2022-07-27T21:34:17.028371Z","shell.execute_reply.started":"2022-07-27T21:34:16.829331Z","shell.execute_reply":"2022-07-27T21:34:17.027138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We note that most of those who survived were in **Class 1**.","metadata":{}},{"cell_type":"code","source":"train_set.groupby(['Pclass'])['Survived'].mean().sort_values(ascending=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:17.031180Z","iopub.execute_input":"2022-07-27T21:34:17.032430Z","iopub.status.idle":"2022-07-27T21:34:17.044010Z","shell.execute_reply.started":"2022-07-27T21:34:17.032385Z","shell.execute_reply":"2022-07-27T21:34:17.042450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set.groupby(['Pclass'])['Survived'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:17.046256Z","iopub.execute_input":"2022-07-27T21:34:17.046948Z","iopub.status.idle":"2022-07-27T21:34:17.063086Z","shell.execute_reply.started":"2022-07-27T21:34:17.046907Z","shell.execute_reply":"2022-07-27T21:34:17.062056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='Pclass', data=train_set, hue='Survived')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:17.064731Z","iopub.execute_input":"2022-07-27T21:34:17.065165Z","iopub.status.idle":"2022-07-27T21:34:17.274564Z","shell.execute_reply.started":"2022-07-27T21:34:17.065128Z","shell.execute_reply":"2022-07-27T21:34:17.273689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set['Embarked'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:17.278008Z","iopub.execute_input":"2022-07-27T21:34:17.279083Z","iopub.status.idle":"2022-07-27T21:34:17.289806Z","shell.execute_reply.started":"2022-07-27T21:34:17.279044Z","shell.execute_reply":"2022-07-27T21:34:17.288167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We see that most of the passengers departed from Port **S** = Southampton","metadata":{}},{"cell_type":"code","source":"train_set['Embarked'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:17.291595Z","iopub.execute_input":"2022-07-27T21:34:17.292080Z","iopub.status.idle":"2022-07-27T21:34:17.304329Z","shell.execute_reply.started":"2022-07-27T21:34:17.292033Z","shell.execute_reply":"2022-07-27T21:34:17.302807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='Embarked', data=train_set)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:17.306541Z","iopub.execute_input":"2022-07-27T21:34:17.308052Z","iopub.status.idle":"2022-07-27T21:34:17.478712Z","shell.execute_reply.started":"2022-07-27T21:34:17.307985Z","shell.execute_reply":"2022-07-27T21:34:17.477486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Is there a relationship between the port of departure and survival? Let's see that \n\nWe can see that the percentage of survivors from Port C is a bit large","metadata":{}},{"cell_type":"code","source":"train_set.groupby(['Embarked'])['Survived'].mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:17.480461Z","iopub.execute_input":"2022-07-27T21:34:17.481878Z","iopub.status.idle":"2022-07-27T21:34:17.492093Z","shell.execute_reply.started":"2022-07-27T21:34:17.481820Z","shell.execute_reply":"2022-07-27T21:34:17.490993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='Embarked', data=train_set, hue='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:17.493345Z","iopub.execute_input":"2022-07-27T21:34:17.494148Z","iopub.status.idle":"2022-07-27T21:34:17.703807Z","shell.execute_reply.started":"2022-07-27T21:34:17.494107Z","shell.execute_reply":"2022-07-27T21:34:17.702294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's see some relations with the Embarked column\n\nWe see that most of the passengers from Port **C** were in Class **1**, so their survival rate was great","metadata":{}},{"cell_type":"code","source":"train_set.groupby('Pclass')['Embarked'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:17.708359Z","iopub.execute_input":"2022-07-27T21:34:17.708729Z","iopub.status.idle":"2022-07-27T21:34:17.721795Z","shell.execute_reply.started":"2022-07-27T21:34:17.708699Z","shell.execute_reply":"2022-07-27T21:34:17.719918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='Embarked', data=train_set, hue='Pclass')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:17.723577Z","iopub.execute_input":"2022-07-27T21:34:17.725573Z","iopub.status.idle":"2022-07-27T21:34:17.952376Z","shell.execute_reply.started":"2022-07-27T21:34:17.725528Z","shell.execute_reply":"2022-07-27T21:34:17.950739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's get some information about the males and females in the ship","metadata":{}},{"cell_type":"code","source":"train_set['Sex'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:17.954090Z","iopub.execute_input":"2022-07-27T21:34:17.954528Z","iopub.status.idle":"2022-07-27T21:34:17.967278Z","shell.execute_reply.started":"2022-07-27T21:34:17.954490Z","shell.execute_reply":"2022-07-27T21:34:17.965855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We see that most of those who did not survive were males","metadata":{}},{"cell_type":"code","source":"print('Survived : \\n')\nprint( train_set[train_set['Survived']==1]['Sex'].value_counts())\nprint('---------------------------------------------------------------')\n\nprint('Died : \\n')\nprint( train_set[train_set['Survived']==0]['Sex'].value_counts())\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:17.968779Z","iopub.execute_input":"2022-07-27T21:34:17.969256Z","iopub.status.idle":"2022-07-27T21:34:17.986263Z","shell.execute_reply.started":"2022-07-27T21:34:17.969195Z","shell.execute_reply":"2022-07-27T21:34:17.985349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='Sex', data=train_set, hue='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:17.987615Z","iopub.execute_input":"2022-07-27T21:34:17.988177Z","iopub.status.idle":"2022-07-27T21:34:18.199070Z","shell.execute_reply.started":"2022-07-27T21:34:17.988146Z","shell.execute_reply":"2022-07-27T21:34:18.197333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's see also the number of males and females in each class","metadata":{}},{"cell_type":"code","source":"print('ALL : \\n')\nprint(pd.crosstab(train['Pclass'], train['Embarked']))\n\nprint('-'*70)\n\n#survived :\nprint('#survived : \\n')\nprint(pd.crosstab(train[train['Survived']==1]['Pclass'], train['Embarked']))\n\nprint('-'*70)\n\n#died\nprint('#died : \\n')\nprint(pd.crosstab(train[train['Survived']==0]['Pclass'], train['Embarked']))","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:18.201395Z","iopub.execute_input":"2022-07-27T21:34:18.201997Z","iopub.status.idle":"2022-07-27T21:34:18.251875Z","shell.execute_reply.started":"2022-07-27T21:34:18.201965Z","shell.execute_reply":"2022-07-27T21:34:18.250601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.displot(x='Sex', data=train_set, hue='Survived', col='Pclass')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:18.253613Z","iopub.execute_input":"2022-07-27T21:34:18.254335Z","iopub.status.idle":"2022-07-27T21:34:18.926661Z","shell.execute_reply.started":"2022-07-27T21:34:18.254296Z","shell.execute_reply":"2022-07-27T21:34:18.925213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's not forget the age column, it is also important","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize =(10,5))\nplt.subplot(1,2,1)\ntrain_set[train_set['Survived']== 1]['Age'].hist(bins=20, alpha=0.7)\n\nplt.subplot(1,2,2)\ntrain_set[train_set['Survived']== 0]['Age'].hist(bins=20, alpha=0.5, color='r')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:18.928399Z","iopub.execute_input":"2022-07-27T21:34:18.928766Z","iopub.status.idle":"2022-07-27T21:34:19.229359Z","shell.execute_reply.started":"2022-07-27T21:34:18.928733Z","shell.execute_reply":"2022-07-27T21:34:19.228277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We saw previously that the number of survivors and those who died in Class 2 are approximately equal,\nAnd we see in this diagram that the elderly in Class 2 are more than in Class 1 and Class 3.\n\nThis may be the reason that the number of survivors in Class 2 is small compared to Class 1 .\nBut we can't say that yet","metadata":{}},{"cell_type":"code","source":"\nplt.figure(figsize =(18,5))\n\nplt.subplot(1,3,1)\nplt.title('Class 1')\nplt.xlabel('Age')\nplt.ylabel('count')\ntrain_set[train_set['Survived']== 1]['Age'].hist(bins=20, alpha=0.5)\nplt.axis([0,80,0,50])\n\nplt.subplot(1,3,2)\nplt.title('Class 2')\nplt.xlabel('Age')\nplt.ylabel('count')\ntrain_set[train_set['Pclass']== 2]['Age'].hist(bins=20, alpha=0.5)\nplt.axis([0,80,0,50])\n\nplt.subplot(1,3,3)\nplt.title('Class 3')\nplt.xlabel('Age')\nplt.ylabel('count')\ntrain_set[train_set['Pclass']== 3]['Age'].hist(bins=20, alpha=0.5)\nplt.axis([0,80,0,50])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:19.231313Z","iopub.execute_input":"2022-07-27T21:34:19.231662Z","iopub.status.idle":"2022-07-27T21:34:19.773101Z","shell.execute_reply.started":"2022-07-27T21:34:19.231630Z","shell.execute_reply":"2022-07-27T21:34:19.771767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"But we see that almost half of the elderly did not survive","metadata":{}},{"cell_type":"code","source":"train_set[train_set['Age'] >= 45]['Survived'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:19.775213Z","iopub.execute_input":"2022-07-27T21:34:19.775616Z","iopub.status.idle":"2022-07-27T21:34:19.787220Z","shell.execute_reply.started":"2022-07-27T21:34:19.775581Z","shell.execute_reply":"2022-07-27T21:34:19.785967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='SibSp', data=train_set, hue='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:19.789056Z","iopub.execute_input":"2022-07-27T21:34:19.789544Z","iopub.status.idle":"2022-07-27T21:34:20.036315Z","shell.execute_reply.started":"2022-07-27T21:34:19.789487Z","shell.execute_reply":"2022-07-27T21:34:20.035056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='Parch', data=train_set, hue='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.037807Z","iopub.execute_input":"2022-07-27T21:34:20.038632Z","iopub.status.idle":"2022-07-27T21:34:20.249837Z","shell.execute_reply.started":"2022-07-27T21:34:20.038580Z","shell.execute_reply":"2022-07-27T21:34:20.248433Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now let's find out more about the **Fare**","metadata":{}},{"cell_type":"code","source":"train_set.groupby('Fare_cut_num')['Survived'].mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.258077Z","iopub.execute_input":"2022-07-27T21:34:20.258500Z","iopub.status.idle":"2022-07-27T21:34:20.268999Z","shell.execute_reply.started":"2022-07-27T21:34:20.258466Z","shell.execute_reply":"2022-07-27T21:34:20.267849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can note that the higher the ticket price, the higher the survival rate","metadata":{}},{"cell_type":"code","source":"sns.countplot(x='Fare_cut_num', data=train_set, hue='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.270572Z","iopub.execute_input":"2022-07-27T21:34:20.271458Z","iopub.status.idle":"2022-07-27T21:34:20.447467Z","shell.execute_reply.started":"2022-07-27T21:34:20.271409Z","shell.execute_reply":"2022-07-27T21:34:20.446045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The most expensive tickets were sold out at Port **C** = \"Cherbourg\"\n\nThis explains the large number of survivors who left from Port **C** :'Cherbourg'","metadata":{}},{"cell_type":"code","source":"pd.crosstab(train['Fare_cut_num'], train['Embarked'])","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.449194Z","iopub.execute_input":"2022-07-27T21:34:20.449634Z","iopub.status.idle":"2022-07-27T21:34:20.474980Z","shell.execute_reply.started":"2022-07-27T21:34:20.449600Z","shell.execute_reply":"2022-07-27T21:34:20.473699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"***Ok that's enough, let's start preparing the data to put in the model***","metadata":{}},{"cell_type":"markdown","source":"## Let's clean our data","metadata":{}},{"cell_type":"code","source":"train_set.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.476580Z","iopub.execute_input":"2022-07-27T21:34:20.477715Z","iopub.status.idle":"2022-07-27T21:34:20.489635Z","shell.execute_reply.started":"2022-07-27T21:34:20.477666Z","shell.execute_reply":"2022-07-27T21:34:20.488536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To deal with the missing data we have three options:\n\n1. drop()   to delete columns (only when we have a lot of missing data)\n2. fillna() Replace missing values with a value\n3. dropns() Delete rows with missing values (only when we have a few missing data)\n\nbut I will use all them ","metadata":{}},{"cell_type":"code","source":"train_set.groupby('Pclass')['Age'].median()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.491340Z","iopub.execute_input":"2022-07-27T21:34:20.491959Z","iopub.status.idle":"2022-07-27T21:34:20.505464Z","shell.execute_reply.started":"2022-07-27T21:34:20.491926Z","shell.execute_reply":"2022-07-27T21:34:20.504576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_data(dataset) :\n    # We will replace the missing values in the age column with the average for each class\n    dataset['Age'] = dataset.groupby(['Pclass'])['Age'].transform(lambda x : round(x.fillna(x.median()),2))\n    # We will remove rows with missing values in the ُEmbarked column.\n    dataset.dropna(subset=['Embarked'], axis=0, inplace=True)\n    # We will remove the cabin column\n    dataset.drop('Cabin', axis=1, inplace= True)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.507146Z","iopub.execute_input":"2022-07-27T21:34:20.507501Z","iopub.status.idle":"2022-07-27T21:34:20.518630Z","shell.execute_reply.started":"2022-07-27T21:34:20.507470Z","shell.execute_reply":"2022-07-27T21:34:20.517582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clean_data(train_set)\nclean_data(test_set)\n\ntrain_set.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.521276Z","iopub.execute_input":"2022-07-27T21:34:20.521702Z","iopub.status.idle":"2022-07-27T21:34:20.562107Z","shell.execute_reply.started":"2022-07-27T21:34:20.521669Z","shell.execute_reply":"2022-07-27T21:34:20.560606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We don't have any false values\ntrain_set.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.564299Z","iopub.execute_input":"2022-07-27T21:34:20.564845Z","iopub.status.idle":"2022-07-27T21:34:20.603775Z","shell.execute_reply.started":"2022-07-27T21:34:20.564797Z","shell.execute_reply":"2022-07-27T21:34:20.602081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We will create a function to convert text values into numeric","metadata":{}},{"cell_type":"code","source":"def cat_to_num(dataset) :\n    #We will convert text values into numeric values\n    sex = pd.get_dummies(dataset['Sex'], drop_first= True)\n    embarked = pd.get_dummies(dataset['Embarked'], drop_first= True)\n    # Now we're going to combine it all into one data set\n    dataset = pd.concat([dataset, sex, embarked], axis=1)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.605358Z","iopub.execute_input":"2022-07-27T21:34:20.606116Z","iopub.status.idle":"2022-07-27T21:34:20.613698Z","shell.execute_reply.started":"2022-07-27T21:34:20.606075Z","shell.execute_reply":"2022-07-27T21:34:20.612090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set = cat_to_num(train_set)\ntest_set = cat_to_num(test_set)\ntrain_set.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.615773Z","iopub.execute_input":"2022-07-27T21:34:20.616388Z","iopub.status.idle":"2022-07-27T21:34:20.644711Z","shell.execute_reply.started":"2022-07-27T21:34:20.616355Z","shell.execute_reply":"2022-07-27T21:34:20.643570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We will create a function to remove the columns we don't need","metadata":{}},{"cell_type":"code","source":"train_set.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.646278Z","iopub.execute_input":"2022-07-27T21:34:20.646643Z","iopub.status.idle":"2022-07-27T21:34:20.654215Z","shell.execute_reply.started":"2022-07-27T21:34:20.646610Z","shell.execute_reply":"2022-07-27T21:34:20.653374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_col(dataset) :\n    dataset.drop(['PassengerId','Sex','Ticket','Embarked'], axis=1, inplace=True)\n    \nremove_col(train_set)\nremove_col(test_set)\n\ntrain_set.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.655820Z","iopub.execute_input":"2022-07-27T21:34:20.656351Z","iopub.status.idle":"2022-07-27T21:34:20.680287Z","shell.execute_reply.started":"2022-07-27T21:34:20.656319Z","shell.execute_reply":"2022-07-27T21:34:20.679088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We will create a function to add new columns","metadata":{}},{"cell_type":"code","source":"def add_col(dataset) :\n    #Combining the 'Parch' column of the and the 'SibSp' column of + 1, we get the size of the family\n    dataset['Family_size'] = dataset['Parch'] + dataset['SibSp'] + 1\n    \n    #From the previous column, we can create a new column if the family size is only one\n    dataset['Is_alone'] = 0\n    dataset.loc[(dataset['Family_size']==1)  ,'Is_alone'] = 1\n    \n    dataset['Big_family'] = 0\n    dataset.loc[(dataset['Family_size']>= 5)  ,'Big_family'] = 1\n    \n\n    \n    \n    return dataset\n    \n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.681902Z","iopub.execute_input":"2022-07-27T21:34:20.683321Z","iopub.status.idle":"2022-07-27T21:34:20.691304Z","shell.execute_reply.started":"2022-07-27T21:34:20.683257Z","shell.execute_reply":"2022-07-27T21:34:20.689588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set = add_col(train_set)\ntest_set = add_col(test_set)\n\ntrain_set.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.693161Z","iopub.execute_input":"2022-07-27T21:34:20.694411Z","iopub.status.idle":"2022-07-27T21:34:20.722102Z","shell.execute_reply.started":"2022-07-27T21:34:20.694325Z","shell.execute_reply":"2022-07-27T21:34:20.721220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def add_title(dataset) :\n    #We will create a new column with the title of the name (Mr., Mrs.....)\n    dataset['Title'] = dataset.Name.str.extract(' ([A-Za-z]+)\\.', expand=False)\n    \n    # to reduce the values in this column\n    dataset['Title'] = dataset['Title'].replace(['Lady', 'Countess','Capt', 'Col',\\\n \t'Don', 'Dr', 'Major', 'Rev', 'Sir', 'Jonkheer', 'Dona'], 'Rare')\n\n    dataset['Title'] = dataset['Title'].replace('Mlle', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Ms', 'Miss')\n    dataset['Title'] = dataset['Title'].replace('Mme', 'Mrs')\n    \n    dataset['Is_marred'] = 0\n    dataset.loc[dataset['Title']=='Mrs', 'Is_marred'] = 1\n    \n    return dataset\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.723633Z","iopub.execute_input":"2022-07-27T21:34:20.723978Z","iopub.status.idle":"2022-07-27T21:34:20.733293Z","shell.execute_reply.started":"2022-07-27T21:34:20.723947Z","shell.execute_reply":"2022-07-27T21:34:20.732004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set = add_title(train_set)\ntest_set = add_title(test_set)\ntrain_set.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.734462Z","iopub.execute_input":"2022-07-27T21:34:20.735178Z","iopub.status.idle":"2022-07-27T21:34:20.776144Z","shell.execute_reply.started":"2022-07-27T21:34:20.735148Z","shell.execute_reply":"2022-07-27T21:34:20.774814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_set.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.778394Z","iopub.execute_input":"2022-07-27T21:34:20.779560Z","iopub.status.idle":"2022-07-27T21:34:20.797516Z","shell.execute_reply.started":"2022-07-27T21:34:20.779508Z","shell.execute_reply":"2022-07-27T21:34:20.796284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# to convert text values into numeric values in title column\ndef title_to_num(title) :\n    if title == 'Mr' :\n        return 1\n    elif title == 'Miss' :\n        return 2\n    elif title == 'Mrs' :\n        return 3\n    elif title == 'Master' :\n        return 4\n    elif title == 'other' :\n        return 5\n    else :\n        return 0","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.799150Z","iopub.execute_input":"2022-07-27T21:34:20.799956Z","iopub.status.idle":"2022-07-27T21:34:20.807941Z","shell.execute_reply.started":"2022-07-27T21:34:20.799908Z","shell.execute_reply":"2022-07-27T21:34:20.807027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(data=train_set, x='Family_size',hue='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:20.809733Z","iopub.execute_input":"2022-07-27T21:34:20.810562Z","iopub.status.idle":"2022-07-27T21:34:21.107993Z","shell.execute_reply.started":"2022-07-27T21:34:20.810516Z","shell.execute_reply":"2022-07-27T21:34:21.106613Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set['Title'] = train_set['Title'].apply(title_to_num)\ntest_set['Title'] = test_set['Title'].apply(title_to_num)\ntrain_set.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:21.109784Z","iopub.execute_input":"2022-07-27T21:34:21.110152Z","iopub.status.idle":"2022-07-27T21:34:21.134907Z","shell.execute_reply.started":"2022-07-27T21:34:21.110121Z","shell.execute_reply":"2022-07-27T21:34:21.133941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#We will remove the  we no longer need\ndef remove_col_2(dataset) :\n    dataset.drop(['Name', 'Parch','Family_size','SibSp'], axis = 1, inplace = True)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:21.136169Z","iopub.execute_input":"2022-07-27T21:34:21.137482Z","iopub.status.idle":"2022-07-27T21:34:21.142335Z","shell.execute_reply.started":"2022-07-27T21:34:21.137447Z","shell.execute_reply":"2022-07-27T21:34:21.141429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"remove_col_2(train_set)\nremove_col_2(test_set)\n\ntrain_set.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:21.143471Z","iopub.execute_input":"2022-07-27T21:34:21.144183Z","iopub.status.idle":"2022-07-27T21:34:21.168346Z","shell.execute_reply.started":"2022-07-27T21:34:21.144151Z","shell.execute_reply":"2022-07-27T21:34:21.167329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_set.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:21.169973Z","iopub.execute_input":"2022-07-27T21:34:21.170619Z","iopub.status.idle":"2022-07-27T21:34:21.192885Z","shell.execute_reply.started":"2022-07-27T21:34:21.170580Z","shell.execute_reply":"2022-07-27T21:34:21.191508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:21.194598Z","iopub.execute_input":"2022-07-27T21:34:21.194944Z","iopub.status.idle":"2022-07-27T21:34:21.213735Z","shell.execute_reply.started":"2022-07-27T21:34:21.194915Z","shell.execute_reply":"2022-07-27T21:34:21.212344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Now we will reduce the values of the age column","metadata":{}},{"cell_type":"code","source":"# First, we need to divide this data into equal parts to break it into pieces\n\ntrain_set['Age_cut'] = pd.cut(train_set['Age'], 5)\ntrain_set[['Age_cut', 'Survived']].groupby(['Age_cut'], as_index=False).mean().sort_values(by='Age_cut', ascending=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:21.215129Z","iopub.execute_input":"2022-07-27T21:34:21.216654Z","iopub.status.idle":"2022-07-27T21:34:21.238863Z","shell.execute_reply.started":"2022-07-27T21:34:21.216526Z","shell.execute_reply":"2022-07-27T21:34:21.237341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# now we will cut our data 'Age' into pieces\ndef age_cut(dataset) :\n    dataset.loc[ dataset['Age'] <= 16, 'Age'] = 0\n    dataset.loc[(dataset['Age'] > 16) & (dataset['Age'] <= 32), 'Age'] = 1\n    dataset.loc[(dataset['Age'] > 32) & (dataset['Age'] <= 48), 'Age'] = 2\n    dataset.loc[(dataset['Age'] > 48) & (dataset['Age'] <= 64), 'Age'] = 3\n    dataset.loc[ dataset['Age'] > 64, 'Age'] = 4\nage_cut(train_set)\nage_cut(test_set)\ntrain_set.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:21.240336Z","iopub.execute_input":"2022-07-27T21:34:21.241147Z","iopub.status.idle":"2022-07-27T21:34:21.576620Z","shell.execute_reply.started":"2022-07-27T21:34:21.241109Z","shell.execute_reply":"2022-07-27T21:34:21.575668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in [train_set] :\n    dataset.drop(['Age_cut'], axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:21.578040Z","iopub.execute_input":"2022-07-27T21:34:21.578660Z","iopub.status.idle":"2022-07-27T21:34:21.593934Z","shell.execute_reply.started":"2022-07-27T21:34:21.578621Z","shell.execute_reply":"2022-07-27T21:34:21.592781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_set.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:21.601945Z","iopub.execute_input":"2022-07-27T21:34:21.602586Z","iopub.status.idle":"2022-07-27T21:34:21.616780Z","shell.execute_reply.started":"2022-07-27T21:34:21.602551Z","shell.execute_reply":"2022-07-27T21:34:21.615824Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedShuffleSplit\n\nx = train_set.drop(['Survived'], axis=1)\ny = train_set['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:21.618042Z","iopub.execute_input":"2022-07-27T21:34:21.618688Z","iopub.status.idle":"2022-07-27T21:34:21.630620Z","shell.execute_reply.started":"2022-07-27T21:34:21.618654Z","shell.execute_reply":"2022-07-27T21:34:21.629596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nfrom sklearn.svm import SVC, LinearSVC\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.linear_model import Perceptron\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.tree import DecisionTreeClassifier","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:21.632112Z","iopub.execute_input":"2022-07-27T21:34:21.632810Z","iopub.status.idle":"2022-07-27T21:34:21.643534Z","shell.execute_reply.started":"2022-07-27T21:34:21.632767Z","shell.execute_reply":"2022-07-27T21:34:21.642527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# I got this code from this website : https://www.projectpro.io/recipes/plot-learning-curve-in-python\n# We can plot the learning rate from this function : \n\nfrom sklearn.model_selection import learning_curve\ndef draw_learning_curve(model, x, y) :\n    train_sizes, train_scores, test_scores = learning_curve(model, x, y, cv=10, scoring='accuracy', n_jobs=-1, train_sizes=np.linspace(0.01, 1.0, 50))\n\n    train_mean = np.mean(train_scores, axis=1)\n    train_std = np.std(train_scores, axis=1)\n\n    test_mean = np.mean(test_scores, axis=1)\n    test_std = np.std(test_scores, axis=1)\n\n    plt.subplots(1, figsize=(5,5))\n    plt.plot(train_sizes, train_mean, '--', color=\"#111111\",  label=\"Training score\")\n    plt.plot(train_sizes, test_mean, color=\"#111111\", label=\"Cross-validation score\")\n\n    plt.fill_between(train_sizes, train_mean - train_std, train_mean + train_std, color=\"#DDDDDD\")\n    plt.fill_between(train_sizes, test_mean - test_std, test_mean + test_std, color=\"#DDDDDD\")\n\n    plt.title(\"Learning Curve\")\n    plt.xlabel(\"Training Set Size\"), plt.ylabel(\"Accuracy Score\"), plt.legend(loc=\"best\")\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:21.644916Z","iopub.execute_input":"2022-07-27T21:34:21.645510Z","iopub.status.idle":"2022-07-27T21:34:21.657047Z","shell.execute_reply.started":"2022-07-27T21:34:21.645469Z","shell.execute_reply":"2022-07-27T21:34:21.656074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import cross_val_score\n\nmodels = {'LogisticRegression' : LogisticRegression(),\n          'SVC': SVC(),\n          'KNeighborsClassifier' : KNeighborsClassifier(),\n          'GaussianNB' : GaussianNB(), \n          'SGDClassifier' : SGDClassifier(), \n          'DecisionTreeClassifier' : DecisionTreeClassifier(),\n          'RandomForestClassifier' : RandomForestClassifier()\n         \n         }\n\n\nfor name, model in models.items() :\n    scores = cross_val_score(model,x, y, cv=5)\n    print(f'{name} Score is : {np.mean(scores)} \\n')\n    \n# We can plot the learning rate from this function : \n    draw_learning_curve(model, x, y )\n    print('------------------------------------------------')\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:34:21.658743Z","iopub.execute_input":"2022-07-27T21:34:21.659457Z","iopub.status.idle":"2022-07-27T21:35:32.364967Z","shell.execute_reply.started":"2022-07-27T21:34:21.659412Z","shell.execute_reply":"2022-07-27T21:35:32.363124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\n\nparam_grid = [\n              {'n_estimators' : [2,3,5,6,7,9,16,20], 'max_features':[2,4,6,8], 'max_depth' : [3,4,5,6,7]}\n    \n             ]\n\nrandom_forest = RandomForestClassifier(random_state=42)\ngrid_search = GridSearchCV(random_forest, param_grid, cv=5, return_train_score=True)\ngrid_search.fit(x,y)\n\nprint(' best params  :', grid_search.best_params_)\nprint('best scor : ',grid_search.best_score_)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:35:32.369628Z","iopub.execute_input":"2022-07-27T21:35:32.370088Z","iopub.status.idle":"2022-07-27T21:35:55.667772Z","shell.execute_reply.started":"2022-07-27T21:35:32.370048Z","shell.execute_reply":"2022-07-27T21:35:55.666512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfull_data_train = pd.concat([train_set, test_set], axis=0)\nX = full_data_train.drop(['Survived'], axis=1).copy()\nY = full_data_train['Survived']\n\nx_train ,x_test, y_train ,y_test = train_test_split(X,Y,test_size=0.3, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:35:55.669661Z","iopub.execute_input":"2022-07-27T21:35:55.671129Z","iopub.status.idle":"2022-07-27T21:35:55.685197Z","shell.execute_reply.started":"2022-07-27T21:35:55.671079Z","shell.execute_reply":"2022-07-27T21:35:55.684326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\nrandom_forest = RandomForestClassifier( n_estimators= 20,max_features = 6, max_depth = 3, random_state=42 )\nrandom_forest.fit(x_train ,y_train)\n\ny_pred = random_forest.predict(x_test)\nconfusion_matrix(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:35:55.686841Z","iopub.execute_input":"2022-07-27T21:35:55.687416Z","iopub.status.idle":"2022-07-27T21:35:55.750321Z","shell.execute_reply.started":"2022-07-27T21:35:55.687380Z","shell.execute_reply":"2022-07-27T21:35:55.749180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def all_function(new_data) :\n    clean_data(new_data)\n    new_data['Fare'] = new_data['Fare'].fillna(new_data['Fare'].mean())\n    \n    new_data = cat_to_num(new_data)\n    \n    #remove_col(new_data)\n    new_data.drop(['Sex','Ticket','Embarked'], axis=1, inplace=True)\n    \n    new_data = add_col(new_data)\n    \n    new_data = add_title(new_data)\n    new_data['Title'] = new_data['Title'].apply(title_to_num)\n    \n    remove_col_2(new_data)\n    \n    age_cut(new_data)\n    return new_data\n    \n    \nready_data_test = all_function(data_test)\nready_data_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:35:55.751819Z","iopub.execute_input":"2022-07-27T21:35:55.753058Z","iopub.status.idle":"2022-07-27T21:35:55.808268Z","shell.execute_reply.started":"2022-07-27T21:35:55.753011Z","shell.execute_reply":"2022-07-27T21:35:55.806807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = full_data_train.drop(\"Survived\", axis=1)\nY_train = full_data_train[\"Survived\"]\nX_test  = ready_data_test.drop(\"PassengerId\", axis=1).copy()\nX_train.shape, Y_train.shape, X_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:35:55.810286Z","iopub.execute_input":"2022-07-27T21:35:55.811621Z","iopub.status.idle":"2022-07-27T21:35:55.826190Z","shell.execute_reply.started":"2022-07-27T21:35:55.811565Z","shell.execute_reply":"2022-07-27T21:35:55.824528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_forest = RandomForestClassifier(n_estimators=20,max_features = 6, max_depth = 3, random_state=42)\nrandom_forest.fit(X_train, Y_train)\nY_pred = random_forest.predict(X_test)\nrandom_forest.score(X_train, Y_train)\nacc_random_forest = round(random_forest.score(X_train, Y_train) * 100, 2)\nacc_random_forest","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:35:55.828055Z","iopub.execute_input":"2022-07-27T21:35:55.828972Z","iopub.status.idle":"2022-07-27T21:35:55.909336Z","shell.execute_reply.started":"2022-07-27T21:35:55.828925Z","shell.execute_reply":"2022-07-27T21:35:55.908032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\n        \"PassengerId\": ready_data_test[\"PassengerId\"],\n        \"Survived\": Y_pred\n    })\nsubmission['Survived'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:35:55.910832Z","iopub.execute_input":"2022-07-27T21:35:55.911344Z","iopub.status.idle":"2022-07-27T21:35:55.924553Z","shell.execute_reply.started":"2022-07-27T21:35:55.911295Z","shell.execute_reply":"2022-07-27T21:35:55.922728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:35:55.926652Z","iopub.execute_input":"2022-07-27T21:35:55.927098Z","iopub.status.idle":"2022-07-27T21:35:55.936948Z","shell.execute_reply.started":"2022-07-27T21:35:55.927063Z","shell.execute_reply":"2022-07-27T21:35:55.935224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-27T21:35:55.938674Z","iopub.execute_input":"2022-07-27T21:35:55.939600Z","iopub.status.idle":"2022-07-27T21:35:55.954640Z","shell.execute_reply.started":"2022-07-27T21:35:55.939559Z","shell.execute_reply":"2022-07-27T21:35:55.953282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}