{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T17:35:15.205172Z","iopub.execute_input":"2022-08-11T17:35:15.205636Z","iopub.status.idle":"2022-08-11T17:35:15.234815Z","shell.execute_reply.started":"2022-08-11T17:35:15.205544Z","shell.execute_reply":"2022-08-11T17:35:15.233841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data loading and preprocessing","metadata":{}},{"cell_type":"code","source":"pip install pymystem3","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:15.236625Z","iopub.execute_input":"2022-08-11T17:35:15.236971Z","iopub.status.idle":"2022-08-11T17:35:28.501586Z","shell.execute_reply.started":"2022-08-11T17:35:15.236940Z","shell.execute_reply":"2022-08-11T17:35:28.500239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport plotly.express as px\nimport plotly.graph_objects as go\nfrom plotly.subplots import make_subplots\nimport plotly.graph_objects as go\nfrom pymystem3 import Mystem\nfrom collections import Counter\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.ensemble import RandomForestClassifier, GradientBoostingClassifier\nfrom sklearn.tree import DecisionTreeClassifier","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:28.503391Z","iopub.execute_input":"2022-08-11T17:35:28.503831Z","iopub.status.idle":"2022-08-11T17:35:30.385315Z","shell.execute_reply.started":"2022-08-11T17:35:28.503778Z","shell.execute_reply":"2022-08-11T17:35:30.384151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data overview","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/titanic/train.csv')\ntest = pd.read_csv('/kaggle/input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:30.388572Z","iopub.execute_input":"2022-08-11T17:35:30.389055Z","iopub.status.idle":"2022-08-11T17:35:30.415545Z","shell.execute_reply.started":"2022-08-11T17:35:30.389007Z","shell.execute_reply":"2022-08-11T17:35:30.414666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def data_exploration(name_set):\n    \n    '''This function outputs data using three methods: head, info, describe'''\n    \n    display(name_set.head(5))\n    print()\n    print()\n    name_set.info()\n    print()\n    print()\n    display(name_set.describe())\n    \n    \ndef find_duplicates(name_set):\n    \n    '''Function check for obvious duplicates'''\n    \n    if name_set.duplicated().sum() == 0:\n        print(f'No duplicates found in the dataset.')\n        print('--------------------------------------------')    \n        print()\n    else:\n        print(f'{name_set.duplicated().sum()} duplicates found in the dataset.')    \n        print('--------------------------------------------')    \n        print()\n        \ndef unique_value_check(name_set):\n    \n    '''Checking unique values'''        \n    \n    print('Checking for unique values')\n    print('-----------------------------------------------')\n    print()\n    for column in name_set.columns:\n        if column not in ('PassengerId', 'Name', 'Ticket', 'Cabin'):\n            print(f'{column}: {sorted(name_set.loc[name_set[column].notna(), column].unique())}')\n            print()\n        else:\n            display(name_set.groupby(column).agg({column: 'count'}).rename(columns={column: 'count'}) \\\n                                            .sort_values('count', ascending=False).head(5))\n            print()\n            \n            \ndef empty_values(height, data):\n    \n    '''The function displays the percentage of empty values as a graph'''\n    \n    empty_data = (round(data.isna().sum() / data.shape[0] * 100, 2).to_frame('percent') \\\n                                                                   .query(\"percent > 0\") \\\n                                                                   .sort_values('percent', ascending=True))    \n    fig = px.bar(empty_data,  \n                 orientation='h',\n                 height=height,\n                 text=[f\"{i}%\" for i in empty_data['percent']])\n        \n    \n    fig.update_layout(plot_bgcolor='#F8F8F6',                  \n                      font_color='#012623',\n                      title_font_size=20,\n                      title_font_family='Arial Black',\n                      xaxis_title='',\n                      yaxis_title='',\n                      bargap=0.2,\n                      showlegend=False,\n                      title_text='Percentage of empty values in the dataset', title_x=0,\n                      margin=dict(l=0, r=20, t=100, b=20))\n    \n    fig.show()\n    \n    \ndef scatter(column, facet_col, title):\n    fig = px.scatter(train, x=column, facet_col=facet_col, facet_row='Survived')\n    fig.update_coloraxes(showscale=False)\n    fig.update_traces(marker=dict(color='#19D3F3',\n                             line=dict(width=1, color='#636EFA')))\n    \n    fig.update_layout(plot_bgcolor='#F8F8F6',\n                      height=700,\n                      width=700,                      \n                      title_text=title,\n                      title_x=0.5,\n                      title_y=0.89,\n                      margin=dict(l=0, r=20, t=130, b=50))\n    fig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:30.417347Z","iopub.execute_input":"2022-08-11T17:35:30.417826Z","iopub.status.idle":"2022-08-11T17:35:30.435664Z","shell.execute_reply.started":"2022-08-11T17:35:30.417758Z","shell.execute_reply":"2022-08-11T17:35:30.434504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Train dataset","metadata":{}},{"cell_type":"code","source":"data_exploration(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:30.437175Z","iopub.execute_input":"2022-08-11T17:35:30.437522Z","iopub.status.idle":"2022-08-11T17:35:30.518697Z","shell.execute_reply.started":"2022-08-11T17:35:30.437492Z","shell.execute_reply":"2022-08-11T17:35:30.517417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The data set **_train_** contains:\n- **_PassengerId_** - identification number of passenger; \n- **_Survived_** - the passenger survived or not: 0 = No, 1 = Yes;\n- **_Pclass_** - ticket class: 1 = 1st, 2 = 2nd, 3 = 3rd;\n- **_Name_** - name of passenger;\n- **_Sex_** - sex: male, female;\n- **_Age_** - age in years;\n- **_SibSp_** - of siblings / spouses aboard the Titanic;\n- **_Parch_** - of parents / children aboard the Titanic\n- **_Ticket_** - ticket number\n- **_Fare_** - passenger fare\n- **_Cabin_** - cabin number\n- **_Embarked_** - port of embarkation: C = Cherbourg, Q = Queenstown, S = Southampton.\n\nWhat we can say about the data in **_train_** dataset:\n- from the describe() method we see, that the median value of **_Survived_** is zero, indicating that more people did not survive;\n- 3rd class tickets predominate;\n- the median age is 28 years old, this indicates that young people predominated aboard the Titanic;\n- there were not a lot of siblings or spouses as well as parents or children aboard the Titanic;\n- the prevailing fare was more than 14(I think in pounds).","metadata":{}},{"cell_type":"markdown","source":"#### Test dataset","metadata":{}},{"cell_type":"markdown","source":"Column names and values are the same as in the **_train_** dataset.\n\nWhat we can say about the data in **_test_** dataset:\n- 3rd class tickets predominate;\n- the median age is 27 years old, this indicates that young people predominated aboard the Titanic;\n- there were not a lot of siblings or spouses as well as parents or children aboard the Titanic;\n- the prevailing fare was more than 14(I think in pounds).","metadata":{"execution":{"iopub.status.busy":"2022-08-09T15:11:44.797198Z","iopub.execute_input":"2022-08-09T15:11:44.797513Z","iopub.status.idle":"2022-08-09T15:11:44.804915Z","shell.execute_reply.started":"2022-08-09T15:11:44.797491Z","shell.execute_reply":"2022-08-09T15:11:44.803448Z"}}},{"cell_type":"markdown","source":"### Checking for duplicates and gaps","metadata":{}},{"cell_type":"markdown","source":"#### Train dataset","metadata":{}},{"cell_type":"code","source":"find_duplicates(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:30.520315Z","iopub.execute_input":"2022-08-11T17:35:30.521232Z","iopub.status.idle":"2022-08-11T17:35:30.530646Z","shell.execute_reply.started":"2022-08-11T17:35:30.521195Z","shell.execute_reply":"2022-08-11T17:35:30.529704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check columns for uniqueness.","metadata":{}},{"cell_type":"code","source":"unique_value_check(train)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:30.532218Z","iopub.execute_input":"2022-08-11T17:35:30.532765Z","iopub.status.idle":"2022-08-11T17:35:30.573832Z","shell.execute_reply.started":"2022-08-11T17:35:30.532735Z","shell.execute_reply":"2022-08-11T17:35:30.572316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"empty_values(300, train)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:30.581264Z","iopub.execute_input":"2022-08-11T17:35:30.582193Z","iopub.status.idle":"2022-08-11T17:35:31.694515Z","shell.execute_reply.started":"2022-08-11T17:35:30.582144Z","shell.execute_reply":"2022-08-11T17:35:31.693264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"What we can say about the **_train_** dataset:\n- consists of 891 rows and 12 columns;\n- no obvious duplicates were found in the dataset;\n- **_Sex_** is a categorical feature;\n- gaps were found, especially in columns: **_Cabin_**, **_Age_**, **_Embarked_**;\n- there are duplicate values in the columns **_Ticket_** and **_Cabin_**.","metadata":{}},{"cell_type":"markdown","source":"#### Test dataset","metadata":{}},{"cell_type":"code","source":"find_duplicates(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:31.695650Z","iopub.execute_input":"2022-08-11T17:35:31.695982Z","iopub.status.idle":"2022-08-11T17:35:31.705057Z","shell.execute_reply.started":"2022-08-11T17:35:31.695952Z","shell.execute_reply":"2022-08-11T17:35:31.703997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check columns for uniqueness.","metadata":{}},{"cell_type":"code","source":"unique_value_check(test)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:31.706611Z","iopub.execute_input":"2022-08-11T17:35:31.707267Z","iopub.status.idle":"2022-08-11T17:35:31.748436Z","shell.execute_reply.started":"2022-08-11T17:35:31.707221Z","shell.execute_reply":"2022-08-11T17:35:31.747451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"empty_values(300, test)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:31.749513Z","iopub.execute_input":"2022-08-11T17:35:31.750242Z","iopub.status.idle":"2022-08-11T17:35:31.828215Z","shell.execute_reply.started":"2022-08-11T17:35:31.750201Z","shell.execute_reply":"2022-08-11T17:35:31.827006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"What we can say about the **_test_** dataset:\n- consists of 418 rows and 11 columns;\n- no obvious duplicates were found in the dataset;\n- **_Sex_** is a categorical feature;\n- gaps were found, especially in columns: **_Cabin_**, **_Age_**, **_Embarked_**;\n- there are duplicate values in the columns **_Ticket_** and **_Cabin_**.","metadata":{}},{"cell_type":"markdown","source":"### Duplicate analysis in the Cabin and Ticket columns","metadata":{}},{"cell_type":"code","source":"train.groupby('Cabin').agg({'Cabin': 'count'}).rename(columns={'Cabin': 'count'}) \\\n     .sort_values('count', ascending=False).head(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:31.829476Z","iopub.execute_input":"2022-08-11T17:35:31.829897Z","iopub.status.idle":"2022-08-11T17:35:31.844050Z","shell.execute_reply.started":"2022-08-11T17:35:31.829847Z","shell.execute_reply":"2022-08-11T17:35:31.843237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.query('Cabin == \"C23 C25 C27\"')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:31.845315Z","iopub.execute_input":"2022-08-11T17:35:31.845846Z","iopub.status.idle":"2022-08-11T17:35:31.866140Z","shell.execute_reply.started":"2022-08-11T17:35:31.845815Z","shell.execute_reply":"2022-08-11T17:35:31.864828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.query('Cabin == \"G6\"')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:31.867877Z","iopub.execute_input":"2022-08-11T17:35:31.868308Z","iopub.status.idle":"2022-08-11T17:35:31.890006Z","shell.execute_reply.started":"2022-08-11T17:35:31.868275Z","shell.execute_reply":"2022-08-11T17:35:31.888780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.query('Cabin == \"C22 C26\"')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:31.892628Z","iopub.execute_input":"2022-08-11T17:35:31.893002Z","iopub.status.idle":"2022-08-11T17:35:31.910247Z","shell.execute_reply.started":"2022-08-11T17:35:31.892971Z","shell.execute_reply":"2022-08-11T17:35:31.909445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.query('Cabin == \"F2\"')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:31.911647Z","iopub.execute_input":"2022-08-11T17:35:31.912259Z","iopub.status.idle":"2022-08-11T17:35:31.936133Z","shell.execute_reply.started":"2022-08-11T17:35:31.912218Z","shell.execute_reply":"2022-08-11T17:35:31.934518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"It looks like one ticket was purchased for the whole family. The cabin number can also refer to a single family or a set of cabins.\n\nBut why is the total cost of the ticket listed for each family member and not a portion of that cost? Create a new column and specify for each family member the cost of the ticket.","metadata":{}},{"cell_type":"code","source":"join_data_set = train.append(test, sort=False)\nduplicate_ticket = join_data_set['Ticket'].value_counts().reset_index() \\\n                                          .rename(columns={'Ticket': 'count', 'index': 'ticket'}) \\\n                                          .query('count > 1') \\\n                                          .sort_values('ticket')\n\ndic = dict(zip(duplicate_ticket['ticket'], duplicate_ticket['count']))\njoin_data_set['passengers_in_one_ticket'] = join_data_set['Ticket'].map(dic)\njoin_data_set.loc[join_data_set['passengers_in_one_ticket'].isna(), 'passengers_in_one_ticket'] = 1\njoin_data_set['correct_fare'] = join_data_set['Fare'] / join_data_set['passengers_in_one_ticket']    \ndisplay(join_data_set['correct_fare'].describe().to_frame())","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:31.939543Z","iopub.execute_input":"2022-08-11T17:35:31.941074Z","iopub.status.idle":"2022-08-11T17:35:31.969902Z","shell.execute_reply.started":"2022-08-11T17:35:31.941040Z","shell.execute_reply":"2022-08-11T17:35:31.968943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we have real fares for every passenger. The maximum fare has changed from 512 to 221.\n\nNow we need to add the correct fare information to the two data sets.","metadata":{}},{"cell_type":"code","source":"train['correct_fare'] = join_data_set.loc[join_data_set['Survived'].notna(), 'correct_fare']\ntest['correct_fare'] = join_data_set.loc[join_data_set['Survived'].isna(), 'correct_fare']","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:31.971093Z","iopub.execute_input":"2022-08-11T17:35:31.971407Z","iopub.status.idle":"2022-08-11T17:35:31.978997Z","shell.execute_reply.started":"2022-08-11T17:35:31.971379Z","shell.execute_reply":"2022-08-11T17:35:31.977957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Gap analysis","metadata":{}},{"cell_type":"markdown","source":"Let's look at the percentage of empty values in the **_Cabin_** column.","metadata":{}},{"cell_type":"code","source":"print(f\"Number of rows with gaps in the Cabins column in train dataset is {round(train.loc[train['Cabin'].isna(), 'Cabin'].shape[0] / train.shape[0] * 100, 2)}%.\")\nprint(f\"Number of rows with gaps in the Cabins column in test dataset is {round(test.loc[test['Cabin'].isna(), 'Cabin'].shape[0] / test.shape[0] * 100, 2)}%.\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:31.980331Z","iopub.execute_input":"2022-08-11T17:35:31.981005Z","iopub.status.idle":"2022-08-11T17:35:31.993515Z","shell.execute_reply.started":"2022-08-11T17:35:31.980967Z","shell.execute_reply":"2022-08-11T17:35:31.992401Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Which class has more gaps?","metadata":{}},{"cell_type":"code","source":"def gaps_analysis(column, method):\n    if method == 'isna':\n        for i, dataset in enumerate([train, test]): \n            dataset_list=['train', 'test']\n            print(dataset_list[i])\n            data = dataset[dataset[column].isna()].groupby('Pclass').agg({'Pclass': 'count'}) \\\n                                                  .rename(columns={'Pclass': 'count'}).reset_index()\n            data['%'] = round(data['count'] / dataset.groupby('Pclass').agg({'Pclass': 'count'}) \\\n                                                     .rename(columns={'Pclass': 'count'}).reset_index()['count'] * 100, 2)\n            display(data)\n    if method == 'notna':\n        for i, dataset in enumerate([train, test]): \n            dataset_list=['train', 'test']\n            print(dataset_list[i])\n            data = dataset[dataset[column].notna()].groupby('Pclass').agg({'Pclass': 'count'}) \\\n                                                  .rename(columns={'Pclass': 'count'}).reset_index()\n            data['%'] = round(data['count'] / dataset.groupby('Pclass').agg({'Pclass': 'count'}) \\\n                                                     .rename(columns={'Pclass': 'count'}).reset_index()['count'] * 100, 2)\n            display(data)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:31.994990Z","iopub.execute_input":"2022-08-11T17:35:31.995588Z","iopub.status.idle":"2022-08-11T17:35:32.005629Z","shell.execute_reply.started":"2022-08-11T17:35:31.995555Z","shell.execute_reply":"2022-08-11T17:35:32.004753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gaps_analysis('Cabin', 'isna')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:32.006977Z","iopub.execute_input":"2022-08-11T17:35:32.007983Z","iopub.status.idle":"2022-08-11T17:35:32.044614Z","shell.execute_reply.started":"2022-08-11T17:35:32.007950Z","shell.execute_reply":"2022-08-11T17:35:32.043856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The 3rd and 2nd class has more gaps. Which class of data is better filled?","metadata":{}},{"cell_type":"code","source":"gaps_analysis('Cabin', 'notna')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:32.045689Z","iopub.execute_input":"2022-08-11T17:35:32.046903Z","iopub.status.idle":"2022-08-11T17:35:32.078034Z","shell.execute_reply.started":"2022-08-11T17:35:32.046855Z","shell.execute_reply":"2022-08-11T17:35:32.076873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The 1st class is better filled. But I think it is better to remove **_Cabin_** column, too many gaps.","metadata":{}},{"cell_type":"code","source":"train.drop('Cabin', inplace=True, axis=1)\ntest.drop('Cabin', inplace=True, axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:32.084838Z","iopub.execute_input":"2022-08-11T17:35:32.085184Z","iopub.status.idle":"2022-08-11T17:35:32.092548Z","shell.execute_reply.started":"2022-08-11T17:35:32.085152Z","shell.execute_reply":"2022-08-11T17:35:32.091522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's look at the percentage of empty values in the **_Age_** column.","metadata":{}},{"cell_type":"code","source":"print(f\"Number of rows with gaps in the Age column in train dataset is {round(train.loc[train['Age'].isna(), 'Age'].shape[0] / train.shape[0] * 100, 2)}%.\")\nprint(f\"Number of rows with gaps in the Age column in test dataset is {round(test.loc[test['Age'].isna(), 'Age'].shape[0] / test.shape[0] * 100, 2)}%.\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:32.093885Z","iopub.execute_input":"2022-08-11T17:35:32.094344Z","iopub.status.idle":"2022-08-11T17:35:32.104569Z","shell.execute_reply.started":"2022-08-11T17:35:32.094300Z","shell.execute_reply":"2022-08-11T17:35:32.103789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Which class has more gaps?","metadata":{}},{"cell_type":"code","source":"gaps_analysis('Age', 'isna')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:32.105713Z","iopub.execute_input":"2022-08-11T17:35:32.106195Z","iopub.status.idle":"2022-08-11T17:35:32.139194Z","shell.execute_reply.started":"2022-08-11T17:35:32.106162Z","shell.execute_reply":"2022-08-11T17:35:32.138094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The 3rd class has more gaps. Which class of data is better filled?","metadata":{}},{"cell_type":"code","source":"gaps_analysis('Age', 'notna')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:32.140580Z","iopub.execute_input":"2022-08-11T17:35:32.140930Z","iopub.status.idle":"2022-08-11T17:35:32.174361Z","shell.execute_reply.started":"2022-08-11T17:35:32.140896Z","shell.execute_reply":"2022-08-11T17:35:32.173286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The 1st and 2nd class is better filled. Classes 1st and 2nd are better filled. I have two ways: remove rows with gaps, or fill gaps with the median value. ","metadata":{}},{"cell_type":"markdown","source":"### Text lematization in the Name column","metadata":{}},{"cell_type":"code","source":"m = Mystem() \nlist_of_name = train['Name'].unique()\nlist_of_name = ' '.join(list_of_name)\nlemmas = m.lemmatize(list_of_name)\nsorted_lemmas = Counter(lemmas)\nsorted_lemmas = sorted(sorted_lemmas.items(), key=lambda x: -x[1])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:32.176260Z","iopub.execute_input":"2022-08-11T17:35:32.176693Z","iopub.status.idle":"2022-08-11T17:35:39.292280Z","shell.execute_reply.started":"2022-08-11T17:35:32.176651Z","shell.execute_reply":"2022-08-11T17:35:39.290847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, elem in enumerate(sorted_lemmas):\n    if elem[0].isalpha():\n        print(i, elem)\n    if i == 11:\n        break","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:39.294206Z","iopub.execute_input":"2022-08-11T17:35:39.295187Z","iopub.status.idle":"2022-08-11T17:35:39.301473Z","shell.execute_reply.started":"2022-08-11T17:35:39.295148Z","shell.execute_reply":"2022-08-11T17:35:39.300356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- Mr. is found 521 times;\n- Miss is found 182 times;\n- Mrs. is found 129 times;\n- Master is found 40 times;","metadata":{}},{"cell_type":"markdown","source":"## EDA","metadata":{}},{"cell_type":"markdown","source":"### Data review on the train dataset","metadata":{}},{"cell_type":"code","source":"age_of_suvivors = train[train['Survived'] == 1].groupby(['Age', 'Name']).agg({'Age': 'count'}) \\\n    .rename(columns={'Age': 'count'}).reset_index() \\\n    .sort_values('Age', ascending=True)\n\n\np_class_more_survived = train[train['Survived'] == 1].groupby('Pclass').agg({'Pclass': 'count'}) \\\n                                                     .rename(columns={'Pclass': 'count'}).reset_index() \\\n                                                     .sort_values('Pclass', ascending=True)\n\n\npopular_port = train.groupby('Embarked').agg({'Embarked': 'count'}).rename(columns={'Embarked': 'count'}).reset_index() \\\n                    .sort_values('count', ascending=False)\nif popular_port['Embarked'][2] == 'S':\n    port = 'Southampton'\n    quantity = popular_port['count'][2]\nif popular_port['Embarked'][1] == 'C':\n    port = 'Cherbourg'\n    quantity = popular_port['count'][1]\nif popular_port['Embarked'][0] == 'Q':\n    port = 'Queenstown'\n    quantity = popular_port['count'][0]\n\n\nfig = make_subplots(rows=5, cols=3, column_widths=[0.33, 0.33, 0.33], horizontal_spacing=0.05,\n                    specs=[[{'type': 'indicator'}, {'type': 'indicator'}, {'type': 'indicator'}],\n                           [{'type': 'indicator'}, {'type': 'indicator'},\n                               {'type': 'indicator'}],\n                           [{'type': 'indicator'}, {'type': 'indicator'},\n                               {'type': 'indicator'}],\n                           [{'type': 'indicator'}, {'type': 'indicator'},\n                               {'type': 'indicator'}],\n                           [{'type': 'indicator'}, {'type': 'indicator'}, {'type': 'indicator'}]])\n\n\n# Plot 1.1-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=round(train[train['Survived'] == 1]['Survived'].count(\n                           ) / train['Survived'].count() * 100, 2),\n                           number={'prefix': \"% \"},\n                           delta={'position': \"top\", 'reference': 100},\n                           title={'text': 'Survived'}), row=1, col=1)\n\n# Plot 1.2-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=round(train[(train['Survived'] == 1) & (\n                               train['Sex'] == 'male')]['Survived'].count() / train['Survived'].count() * 100, 2),\n                           number={'prefix': \"% \"},\n                           delta={'position': \"top\", 'reference': round(train[train['Sex'] == 'male']['Sex'].count() / train['Survived'].count() * 100, 2)},\n                           title={'text': 'Men Survived'}), row=1, col=2)\n\n# Plot 1.3-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=round(train[(train['Survived'] == 1) & (\n                               train['Sex'] == 'female')]['Survived'].count() / train['Survived'].count() * 100, 2),\n                           delta={'position': \"top\", 'reference': round(train[train['Sex'] == 'female']['Sex'].count() / train['Survived'].count() * 100, 2)},\n                           number={'prefix': \"% \"},\n                           title={'text': 'Women Survived'}), row=1, col=3)\n\n# Plot 2.1-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=round(p_class_more_survived['count'][0] / train['Survived'].count() * 100, 2),\n                           delta={'position': \"top\", 'reference': round(train[train['Pclass'] == 1]['Pclass'].count() / train['Survived'].count() * 100, 2)},\n                           number={'prefix': \"% \"},\n                           title={'text': f'More survived were in {p_class_more_survived[\"Pclass\"][0]} Pclass'}),\n              row=2, col=1)\n\n# Plot 2.2-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=round(p_class_more_survived['count'][1] / train['Survived'].count() * 100, 2),\n                           delta={'position': \"top\", 'reference': round(train[train['Pclass'] == 2]['Pclass'].count() / train['Survived'].count() * 100, 2)},\n                           number={'prefix': \"% \"},\n                           title={'text': f'More unsurvived were in {p_class_more_survived[\"Pclass\"][1]} Pclass'}),\n              row=2, col=2)\n\n# Plot 2.3-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=round(p_class_more_survived['count'][2] / train['Survived'].count() * 100, 2),\n                           delta={'position': \"top\", 'reference': round(train[train['Pclass'] == 3]['Pclass'].count() / train['Survived'].count() * 100, 2)},\n                           number={'prefix': \"% \"},\n                           title={'text': f'More unsurvived were in {p_class_more_survived[\"Pclass\"][2]} Pclass'}),\n              row=2, col=3)\n\n# Plot 3.1-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=age_of_suvivors['Age'].min(),\n                           number={'prefix': 'The Youngest (y.o.): '},\n                           delta={'position': \"top\", 'reference': None},\n                           title={'text': f\"{age_of_suvivors['Name'][0]}\"}),\n              row=3, col=1)\n\n# Plot 3.2-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=age_of_suvivors['Age'].max(),\n                           number={'prefix': 'The Oldest (y.o.): '},\n                           delta={'position': \"top\", 'reference': None},\n                           title={'text': f\"{age_of_suvivors['Name'][64]}\"}),\n              row=3, col=2)\n\n# Plot 3.3-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=age_of_suvivors['Age'].median(),\n                           number={'prefix': 'y.o. '},\n                           delta={'position': \"top\", 'reference': None},\n                           title={'text': 'Median age on board'}),\n              row=3, col=3)\n\n# Plot 4.1-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=sorted_lemmas[8][1],\n                           number={'prefix': f'{sorted_lemmas[8][0]} (count): '},\n                           delta={'position': \"top\", 'reference': None},\n                           title={'text': 'Common male name on board'}),\n              row=4, col=1)\n\n# Plot 4.2-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=sorted_lemmas[18][1],\n                           number={'prefix': f'{sorted_lemmas[18][0]} (count): '},\n                           delta={'position': \"top\", 'reference': None},\n                           title={'text': 'Common female name on board'}),\n              row=4, col=2)\n\n# Plot 4.3-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=sorted_lemmas[3][1],\n                           number={'prefix': f'{sorted_lemmas[3][0]} (count): '},\n                           delta={'position': \"top\", 'reference': None},\n                           title={'text': 'Common Title of Man'}),\n              row=4, col=3)\n\n# Plot 5.1-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=quantity,\n                           number={'prefix': f'{port} (count): '},\n                           delta={'position': \"top\", 'reference': None},\n                           title={'text': 'Popular Port of Embarkation'}),\n              row=5, col=1)\n\n# Plot 5.2-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=train['correct_fare'].max(),\n                           number={'prefix': '£ '},\n                           delta={'position': \"top\", 'reference': None},\n                           title={'text': 'The most expensive ticket'}),\n              row=5, col=2)\n\n# Plot 5.3-------------------------------------------------\nfig.add_trace(go.Indicator(mode='number+delta',\n                           value=train['correct_fare'].min(),\n                           number={'prefix': '£ '},\n                           delta={'position': \"top\", 'reference': None},\n                           title={'text': 'The cheapest ticket'}),\n              row=5, col=3)\n\n# first vertical column\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0, y0=0.85, x1=0.3, y1=1,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0, y0=0.79, x1=0.3, y1=0.64,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0, y0=0.58, x1=0.3, y1=0.43,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0, y0=0.37, x1=0.3, y1=0.22,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0, y0=0.16, x1=0.3, y1=0.01,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\n# second vertical column\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0.35, y0=0.85, x1=0.655, y1=1,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0.35, y0=0.79, x1=0.655, y1=0.64,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0.35, y0=0.58, x1=0.655, y1=0.43,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0.35, y0=0.37, x1=0.655, y1=0.22,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0.35, y0=0.16, x1=0.655, y1=0.01,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\n# third vertical column\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0.705, y0=0.85, x1=1, y1=1,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0.705, y0=0.79, x1=1, y1=0.64,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0.705, y0=0.58, x1=1, y1=0.43,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0.705, y0=0.37, x1=1, y1=0.22,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\nfig.add_shape(type=\"rect\", xref=\"paper\", yref=\"paper\", x0=0.705, y0=0.16, x1=1, y1=0.01,\n              line=dict(color='#636EFA', width=3), fillcolor='#636EFA', opacity=0.08)\n\n\nfig.update_traces(delta_font_size=14,\n                  number_font_size=18,\n                  title_font_size=14,\n                  selector=dict(type='indicator'))\n\nfig.update_layout(plot_bgcolor='#F8F8F6',\n                  height=850,\n                  width=950,\n                  title_text='Some of the interesting information in the train dataset',\n                  title_font_size=20,\n                  title_font_family='Arial Black',\n                  title_x=0,\n                  title_y=0.93,\n                  bargap=0.2,\n                  margin=dict(l=0, r=20, t=130, b=80))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:39.303204Z","iopub.execute_input":"2022-08-11T17:35:39.303822Z","iopub.status.idle":"2022-08-11T17:35:39.624077Z","shell.execute_reply.started":"2022-08-11T17:35:39.303769Z","shell.execute_reply":"2022-08-11T17:35:39.623238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's see how many passengers with popular names and titles survived.","metadata":{}},{"cell_type":"code","source":"Master = []\nMary = []\nWilliam = []\nfor i in train['Name']:\n    if 'Master' in i:\n        Master.append(i)\n    if 'Mary' in i:\n        Mary.append(i)\n    if 'William' in i:\n        William.append(i)\n\n\nfor name in (Master, Mary, William):       \n    survived = (train[train['Name'].isin(name) & (train['Survived'] == 1)].groupby('Pclass').agg({'Survived': 'count'}) \\\n                                                                          .reset_index())\n    if len(name) == 40:\n        print(f'{survived[\"Survived\"].sum()} of the {len(name)} passengers who named Master survived.')\n    if len(name) == 20:\n        print(f'{survived[\"Survived\"].sum()} of the {len(name)} passengers who named Mary survived.')\n    if len(name) == 69:\n        print(f'{survived[\"Survived\"].sum()} of the {len(name)} passengers who named William survived.')        \n    \n    display(survived)        ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:39.625715Z","iopub.execute_input":"2022-08-11T17:35:39.626060Z","iopub.status.idle":"2022-08-11T17:35:39.662077Z","shell.execute_reply.started":"2022-08-11T17:35:39.626028Z","shell.execute_reply":"2022-08-11T17:35:39.661242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Women and Masters survived more.","metadata":{}},{"cell_type":"markdown","source":"### Survived","metadata":{}},{"cell_type":"code","source":"data_pie = train.groupby('Survived').agg({'Survived': 'count'}).rename(\n    columns={'Survived': 'count'}).reset_index()\nfig = px.pie(data_pie, values='count')\nfig.update_traces(textinfo='value+percent', showlegend=True,\n                  legendgrouptitle_text='Survived:')\nfig.update_layout(plot_bgcolor='#F8F8F6',\n                  height=330,\n                  width=500,\n                  title_text='Survived',\n                  title_font_size=20,\n                  title_font_family='Arial Black',\n                  title_x=0,\n                  margin=dict(l=0, r=20, t=100, b=40))\nfig.update_traces(insidetextfont_size=10, selector=dict(type='pie'))\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:39.663228Z","iopub.execute_input":"2022-08-11T17:35:39.664014Z","iopub.status.idle":"2022-08-11T17:35:39.761418Z","shell.execute_reply.started":"2022-08-11T17:35:39.663981Z","shell.execute_reply":"2022-08-11T17:35:39.760167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Pclass","metadata":{}},{"cell_type":"code","source":"fig = make_subplots(rows=1, cols=2, specs=[[{'type': 'domain'}, {'type': 'bar'}]],\n                    subplot_titles=('Passenger classes in general', 'Which passenger class has more survivors'))\n\ndata_pie = train.groupby('Pclass').agg({'Pclass': 'count'}).rename(\n    columns={'Pclass': 'count'}).reset_index()\nfig.add_trace(go.Pie(labels=data_pie['Pclass'], values=data_pie['count'], textinfo='value+percent',\n                     legendgroup='1', legendgrouptitle_text='Pclass:'), 1, 1)\n\nfor i in train['Survived'].unique():\n    data_bar = train[train['Survived'] == i].groupby('Pclass').agg({'Pclass': 'count'}) \\\n                                            .rename(columns={'Pclass': 'count'}).reset_index()\n    fig.add_trace(go.Bar(x=data_bar['Pclass'], y=data_bar['count'], showlegend=True,\n                         texttemplate=\"%{y}\", name=i.astype(str), legendgroup='2',\n                         legendgrouptitle_text='Survived:'), row=1, col=2)\n\nfig.update_layout(plot_bgcolor='#F8F8F6',\n                  height=400,\n                  width=800,\n                  title_text=f'Analysis of Pclass',\n                  title_font_size=20,\n                  title_font_family='Arial Black',\n                  title_x=0,\n                  title_y=0.91,\n                  yaxis_title='Count of passengers',\n                  xaxis_title='Passenger classes',\n                  margin=dict(l=0, r=20, t=130, b=80))\n\nfig.update_annotations(yshift=20)\nfig.update_traces(insidetextfont_size=10, selector=dict(type='pie'))\n\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:39.763534Z","iopub.execute_input":"2022-08-11T17:35:39.764130Z","iopub.status.idle":"2022-08-11T17:35:39.851576Z","shell.execute_reply.started":"2022-08-11T17:35:39.764083Z","shell.execute_reply":"2022-08-11T17:35:39.850269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- class 3 is the most large, but it has a small number of survivors; \n- class 1 has the largest number of survivors.\n\nThis feature is important for predicting survivors.","metadata":{}},{"cell_type":"markdown","source":"### Sex","metadata":{}},{"cell_type":"code","source":"fig = make_subplots(rows=2, cols=2, specs=[[{'type': 'domain'}, {'type': 'bar'}],\n                                           [{\"colspan\": 2}, None]],\n                    subplot_titles=('Sex in general', 'Who survived',\n                                    'Sex by passenger class'))\n\ndata_pie = train.groupby('Sex').agg({'Sex': 'count'}).rename(columns={'Sex': 'count'}).reset_index()\n\nfig.add_trace(go.Pie(labels=data_pie['Sex'], values=data_pie['count'], textinfo='value+percent',\n                     legendgroup='1', legendgrouptitle_text='Sex:'), 1, 1)\n\nfor i in train['Survived'].unique():\n    data_bar = train[train['Survived'] == i].groupby('Sex').agg({'Sex': 'count'}) \\\n                                            .rename(columns={'Sex': 'count'}).reset_index()\n    fig.add_trace(go.Bar(x=data_bar['Sex'], y=data_bar['count'], showlegend=True,\n                         texttemplate=\"%{y}\", name=i.astype(str), legendgroup='2',\n                         legendgrouptitle_text='Survived:'), row=1, col=2)\n\nfor i in train['Sex'].unique():\n    data_bar = train[train['Sex'] == i].groupby('Pclass').agg({'Pclass': 'count'}) \\\n                                       .rename(columns={'Pclass': 'count'}).reset_index()\n    \n    fig.add_trace(go.Bar(x=data_bar['Pclass'], y=data_bar['count'], showlegend=True,\n                         texttemplate=\"%{y}\", name=i, legendgroup='3',\n                         legendgrouptitle_text='Sex:'), row=2, col=1)\n\nfig.update_layout(plot_bgcolor='#F8F8F6',\n                  height=700,\n                  width=800,\n                  title_text=f'Analysis of Sex',\n                  title_font_size=20,\n                  title_font_family='Arial Black',\n                  xaxis_title='Sex',\n                  title_x=0,\n                  title_y=0.94,\n                  margin=dict(l=0, r=20, t=130, b=80))\n\nfig.update_annotations(yshift=20)\nfig.update_traces(insidetextfont_size=10, selector=dict(type='pie'))\nfig.update_xaxes(title='Passenger classes', row=2, col=1)\nfig.update_yaxes(title='Count of passengers')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:39.853603Z","iopub.execute_input":"2022-08-11T17:35:39.854070Z","iopub.status.idle":"2022-08-11T17:35:39.944272Z","shell.execute_reply.started":"2022-08-11T17:35:39.854025Z","shell.execute_reply":"2022-08-11T17:35:39.943132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- more women survived than men;\n- there were more men than women in all passenger classes, but the third class had the most.\n\nThis feature is important for predicting survivors.","metadata":{}},{"cell_type":"markdown","source":"### Age","metadata":{}},{"cell_type":"code","source":"def age_or_fare_plots(column, y_title):\n    fig = make_subplots(rows=5, cols=2, specs=[[{\"colspan\": 2}, None],\n                                               [{'type': 'box'}, {'type': 'box'}],\n                                               [{'type': 'box'}, {'type': 'box'}],\n                                               [{'type': 'box'}, {'type': 'box'}],\n                                               [{'type': 'box'}, {'type': 'box'}]],\n                        subplot_titles=(f'{y_title} in general according to survivors',\n                                        f'{y_title} in general', \n                                        f'{y_title} according to survivors',\n                                        f'{y_title} of male according to survivors',\n                                        f'{y_title} of female according to survivors',\n                                        'Survivors by 1st class',\n                                        'Survivors by 2nd class',\n                                        'Survivors by 3rd class'))\n\n    # Plot 1.1-------------------------------------------------\n    for i in train['Survived'].unique():\n        data_box = train[train['Survived'] == i]\n        fig.add_trace(go.Histogram(x=data_box[column], nbinsx=40, showlegend=True, name=i.astype(str), \n                                   legendgroup='1', legendgrouptitle_text='Survived:'), row=1, col=1)\n\n    # Plot 2.1-------------------------------------------------\n    fig.add_trace(\n        go.Box(y=train[column], showlegend=False, name=''), row=2, col=1)\n\n    # Plot 2.2-------------------------------------------------\n    for i in train['Survived'].unique():\n        data_box = train[train['Survived'] == i]\n        fig.add_trace(go.Box(y=data_box[column], showlegend=True, name=i.astype(str),\n                             legendgroup='2', legendgrouptitle_text='Survived:'), row=2, col=2)\n\n    # Plot 3.1-------------------------------------------------\n    for i in train['Survived'].unique():         \n        data_box = train[(train['Survived'] == i) & (train['Sex'] == 'male')]\n        fig.add_trace(go.Histogram(x=data_box[column], nbinsx=35, showlegend=True, name=i.astype(str), \n                                   legendgroup='3', legendgrouptitle_text='Survived:'), row=3, col=1)\n\n    # Plot 3.2-------------------------------------------------\n    for i in train['Survived'].unique():         \n        data_box = train[(train['Survived'] == i) & (train['Sex'] == 'female')]\n        fig.add_trace(go.Histogram(x=data_box[column], nbinsx=30, showlegend=True, name=i.astype(str), \n                                   legendgroup='4', legendgrouptitle_text='Survived:'), row=3, col=2)\n\n    # Plot 4.1-------------------------------------------------\n    for i in train['Survived'].unique():         \n        data_box = train[(train['Survived'] == i) & (train['Pclass'] == 1)]\n        fig.add_trace(go.Histogram(x=data_box[column], nbinsx=35, showlegend=True, name=i.astype(str), \n                                   legendgroup='5', legendgrouptitle_text='Survived:'), row=4, col=1)\n\n    # Plot 4.2-------------------------------------------------\n    for i in train['Survived'].unique():         \n        data_box = train[(train['Survived'] == i) & (train['Pclass'] == 2)]\n        fig.add_trace(go.Histogram(x=data_box[column], nbinsx=30, showlegend=True, name=i.astype(str), \n                                   legendgroup='6', legendgrouptitle_text='Survived:'), row=4, col=2)\n\n    # Plot 5.1-------------------------------------------------\n    for i in train['Survived'].unique():         \n        data_box = train[(train['Survived'] == i) & (train['Pclass'] == 3)]\n        fig.add_trace(go.Histogram(x=data_box[column], nbinsx=35, showlegend=True, name=i.astype(str), \n                                   legendgroup='7', legendgrouptitle_text='Survived:'), row=5, col=1)    \n\n    fig.update_layout(plot_bgcolor='#F8F8F6',\n                      height=1500,\n                      width=800,\n                      bargap=0.2,\n                      title_text=f'Analysis of Age',\n                      title_font_size=20,\n                      title_font_family='Arial Black',                      \n                      title_x=0,\n                      title_y=0.98,\n                      margin=dict(l=0, r=20, t=130, b=80))\n\n    fig.update_annotations(yshift=20)\n    fig.update_traces(insidetextfont_size=10, selector=dict(type='pie'))\n    fig.update_yaxes(title=y_title)\n    fig.update_xaxes(title='Age')\n    fig.update_yaxes(title='Count of passengers')\n    fig.update_yaxes(title='Age', row=2, col=1)\n    fig.update_xaxes(title='', row=2, col=1)\n    fig.update_xaxes(title='Survived or not', row=2, col=2)\n    fig.update_yaxes(title='Age', row=2, col=2)\n    fig.update_yaxes(title='Age', row=2, col=1)    \n    fig.show()\n\nage_or_fare_plots('Age', 'Age')\nscatter('Age', 'Sex', 'Analysis of Age by Sex and survivors')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:39.945947Z","iopub.execute_input":"2022-08-11T17:35:39.946318Z","iopub.status.idle":"2022-08-11T17:35:40.349910Z","shell.execute_reply.started":"2022-08-11T17:35:39.946286Z","shell.execute_reply":"2022-08-11T17:35:40.348662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- most losses are between 20 and 40 years old;\n- more children under 2 years old and 4 to 6 years old are saved than 2 to 4 and 6 years old.","metadata":{"execution":{"iopub.status.busy":"2022-08-09T15:24:26.602717Z","iopub.execute_input":"2022-08-09T15:24:26.603050Z","iopub.status.idle":"2022-08-09T15:24:26.609472Z","shell.execute_reply.started":"2022-08-09T15:24:26.603025Z","shell.execute_reply":"2022-08-09T15:24:26.608322Z"}}},{"cell_type":"markdown","source":"### Siblings / Spouses","metadata":{}},{"cell_type":"code","source":"def relatives_plots(column, x_title, short_name):\n    fig = make_subplots(rows=3, cols=2, specs=[[{'type': 'bar'}, {'type': 'bar'}],\n                                               [{'type': 'bar'}, {'type': 'bar'}],\n                                               [{'type': 'bar'}, {'type': 'bar'}]],\n                        subplot_titles=(f'{x_title} in general',\n                                        f'{x_title} by Sex',\n                                        f'Male survivors with {short_name}',\n                                        f'Female survivors with {short_name}',\n                                        f'Survivors with {short_name} by pclass',\n                                        f'Unsurviving with {short_name} by pclass'))\n\n    # Plot 1.1-------------------------------------------------\n    fig.add_trace(go.Histogram(x=train[column], nbinsx=0, showlegend=False, texttemplate=\"%{y}\"), row=1, col=1)\n\n    # Plot 1.2-------------------------------------------------\n    for i in train['Sex'].unique():\n        data_bar = train[train['Sex'] == i].groupby(column).agg({column: 'count'}) \\\n                                           .rename(columns={column: 'count'}).reset_index()\n        fig.add_trace(go.Bar(x=data_bar[column], y=data_bar['count'], showlegend=True,\n                             texttemplate=\"%{y}\", name=i, legendgroup='1',\n                             legendgrouptitle_text='Sex:'), row=1, col=2)\n\n    # Plot 2.1-------------------------------------------------\n    for i in train['Survived'].unique():\n        data_bar = train[(train['Survived'] == i) & (train['Sex'] == 'male')].groupby(column) \\\n                  .agg({column: 'count'}).rename(columns={column: 'count'}).reset_index()\n        fig.add_trace(go.Bar(x=data_bar[column], y=data_bar['count'], showlegend=True,\n                             texttemplate=\"%{y}\", name=i.astype(str), legendgroup='2',\n                             legendgrouptitle_text='Survived:'), row=2, col=1)\n\n    # Plot 2.2-------------------------------------------------\n    for i in train['Survived'].unique():\n        data_bar = train[(train['Survived'] == i) & (train['Sex'] == 'female')].groupby(column) \\\n                   .agg({column: 'count'}).rename(columns={column: 'count'}).reset_index()\n        fig.add_trace(go.Bar(x=data_bar[column], y=data_bar['count'], showlegend=True,\n                             texttemplate=\"%{y}\", name=i.astype(str), legendgroup='3',\n                             legendgrouptitle_text='Survived:'), row=2, col=2)\n\n    # Plot 3.1-------------------------------------------------\n    for i in train['Pclass'].unique():\n        data_bar = train[(train['Pclass'] == i) & (train['Survived'] == 1)].groupby(column) \\\n                  .agg({column: 'count'}).rename(columns={column: 'count'}).reset_index()\n        fig.add_trace(go.Bar(x=data_bar[column], y=data_bar['count'], showlegend=True,\n                             texttemplate=\"%{y}\", name=i.astype(str), legendgroup='4',\n                             legendgrouptitle_text='Passenger classes:'), row=3, col=1)\n\n    # Plot 3.2-------------------------------------------------\n    for i in train['Pclass'].unique():\n        data_bar = train[(train['Pclass'] == i) & (train['Survived'] == 0)].groupby(column) \\\n                  .agg({column: 'count'}).rename(columns={column: 'count'}).reset_index()\n        fig.add_trace(go.Bar(x=data_bar[column], y=data_bar['count'], showlegend=True,\n                             texttemplate=\"%{y}\", name=i.astype(str), legendgroup='5',\n                             legendgrouptitle_text='Passenger classes:'), row=3, col=2)\n\n    fig.update_layout(plot_bgcolor='#F8F8F6',\n                      height=1000,\n                      width=800,\n                      bargap=0.2,\n                      title_text=f'Analysis of {x_title}',\n                      title_font_size=20,\n                      title_font_family='Arial Black',\n                      title_x=0,\n                      title_y=0.96,\n                      margin=dict(l=0, r=20, t=130, b=100))\n\n    fig.update_annotations(yshift=20)\n    fig.update_traces(insidetextfont_size=10, selector=dict(type='pie'))\n    fig.update_yaxes(title='Count of passengers')\n    fig.update_xaxes(title=x_title)\n    fig.show()\n    \nrelatives_plots('SibSp', 'Siblings / Spouses', 'SibSp')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:40.351965Z","iopub.execute_input":"2022-08-11T17:35:40.352364Z","iopub.status.idle":"2022-08-11T17:35:40.526518Z","shell.execute_reply.started":"2022-08-11T17:35:40.352329Z","shell.execute_reply":"2022-08-11T17:35:40.525413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- the number of passengers with siblings was less;\n- passengers with one sibling survived more often than those with two or more.\n\nThis features can be taken into account in forecasting, but it will have to be modified a little.","metadata":{}},{"cell_type":"markdown","source":"### Parents / Children","metadata":{}},{"cell_type":"code","source":"relatives_plots('Parch', 'Parents / Children', 'Parch')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:40.527981Z","iopub.execute_input":"2022-08-11T17:35:40.528446Z","iopub.status.idle":"2022-08-11T17:35:40.694354Z","shell.execute_reply.started":"2022-08-11T17:35:40.528400Z","shell.execute_reply":"2022-08-11T17:35:40.693147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- the number of passengers with parents / children was less;\n- more passengers without parents / children survived.\n\nThis features can be taken into account in forecasting, but it will have to be modified a little.","metadata":{}},{"cell_type":"markdown","source":"### Fare","metadata":{}},{"cell_type":"code","source":"fig = make_subplots(rows=2, cols=2, column_widths=[0.7, 0.3], specs=[[{\"colspan\": 2}, None],\n                                           [{'type': 'bar'}, {'type': 'box'}]],                                           \n                        subplot_titles=('Fare in general', 'Fare by survivors', 'Fare')) \n\n\n# Plot 1.1-------------------------------------------------       \nfig.add_trace(go.Histogram(x=train['correct_fare'], nbinsx=65, showlegend=False, texttemplate=\"%{y}\"), row=1, col=1)\n                                   \n\n# Plot 2.1-------------------------------------------------\nfor i in train['Survived'].unique():\n    data_bar = train[train['Survived'] == i]\n    fig.add_trace(go.Histogram(x=data_bar['correct_fare'], showlegend=True, name=i.astype(str), nbinsx=55,\n                               texttemplate=\"%{y}\", legendgroup='1', legendgrouptitle_text='Survived:'), row=2, col=1)\n\n# Plot 2.1-------------------------------------------------\nfig.add_trace(go.Box(y=train['correct_fare'], showlegend=False, name=''), row=2, col=2)\n\n\nfig.update_layout(plot_bgcolor='#F8F8F6',\n                      height=700,\n                      width=800,\n                      bargap=0.2,\n                      title_text=f'Analysis of Fare',\n                      title_font_size=20,\n                      title_font_family='Arial Black',\n                      title_x=0,\n                      title_y=0.95,\n                      margin=dict(l=0, r=20, t=130, b=50))\n\nfig.update_annotations(yshift=20)\nfig.update_traces(insidetextfont_size=10, selector=dict(type='pie'))\nfig.update_yaxes(title='Count of passengers')\nfig.update_xaxes(title='x_title')\nfig.show()\n\nscatter('correct_fare', 'Pclass', 'Analysis of Fare by class and survivors')\nscatter('correct_fare', 'Sex', 'Analysis of Fare by Sex and survivors')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:40.696052Z","iopub.execute_input":"2022-08-11T17:35:40.697061Z","iopub.status.idle":"2022-08-11T17:35:41.048210Z","shell.execute_reply.started":"2022-08-11T17:35:40.697024Z","shell.execute_reply":"2022-08-11T17:35:41.047078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- there were passengers in first class who traveled with zero ticket value, I suppose they were servants;\n- again we see that passengers with purchased tickets to 3rd class survived far less than others;\n- the difference between 3rd and 2nd class is not great;","metadata":{}},{"cell_type":"markdown","source":"### Embarked","metadata":{}},{"cell_type":"code","source":"fig = make_subplots(rows=2, cols=2, specs=[[{'type': 'domain'}, {'type': 'bar'}],\n                                           [{'type': 'bar'}, {'type': 'bar'}]],\n                    subplot_titles=('Embarkation in general', 'Who survived',\n                                    'Sex by Embarkation', 'Pclass by Embarkation'))\n\ndata_pie = train.groupby('Embarked').agg({'Embarked': 'count'}).rename(\n    columns={'Embarked': 'count'}).reset_index()\n\nfig.add_trace(go.Pie(labels=data_pie['Embarked'], values=data_pie['count'], textinfo='value+percent',\n                     legendgroup='1', legendgrouptitle_text='Embarked:'), 1, 1)\n\nfor i in train['Survived'].unique():\n    data_bar = train[train['Survived'] == i].groupby('Embarked').agg({'Embarked': 'count'}) \\\n                                            .rename(columns={'Embarked': 'count'}).reset_index()\n    fig.add_trace(go.Bar(x=data_bar['Embarked'], y=data_bar['count'], showlegend=True,\n                         texttemplate=\"%{y}\", name=i.astype(str), legendgroup='2',\n                         legendgrouptitle_text='Survived:'), row=1, col=2)\n    \nfor i in train['Sex'].unique():\n    data_bar = train[train['Sex'] == i].groupby('Embarked').agg({'Embarked': 'count'}) \\\n        .rename(columns={'Embarked': 'count'}).reset_index()\n    fig.add_trace(go.Bar(x=data_bar['Embarked'], y=data_bar['count'], showlegend=True,\n                         texttemplate=\"%{y}\", name=i, legendgroup='3',\n                         legendgrouptitle_text='Sex:'), row=2, col=1)    \n    \nfor i in train['Pclass'].unique():\n    data_bar = train[train['Pclass'] == i].groupby('Embarked').agg({'Embarked': 'count'}) \\\n        .rename(columns={'Embarked': 'count'}).reset_index()\n    fig.add_trace(go.Bar(x=data_bar['Embarked'], y=data_bar['count'], showlegend=True,\n                         texttemplate=\"%{y}\", name=i.astype(str), legendgroup='4',\n                         legendgrouptitle_text='Pclass:'), row=2, col=2)\n    \nfig.update_layout(plot_bgcolor='#F8F8F6',\n                  height=800,\n                  width=800,\n                  title_text=f'Analysis of Embarkation',\n                  title_font_size=20,\n                  title_font_family='Arial Black',\n                  title_x=0,\n                  title_y=0.95,                                    \n                  margin=dict(l=0, r=20, t=130, b=80))\n\nfig.update_annotations(yshift=20)\nfig.update_traces(insidetextfont_size=10, selector=dict(type='pie'))\nfig.update_xaxes(title='Embarkation')\nfig.update_yaxes(title='Count of passengers')\nfig.update_yaxes(title='Age', row=3, col=1)\nfig.show()\n\nscatter('Age', 'Embarked', 'Analysis Age by Embarked and survivors')\nscatter('correct_fare', 'Embarked', 'Analysis Fare by Embarked and survivors')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:41.050013Z","iopub.execute_input":"2022-08-11T17:35:41.050709Z","iopub.status.idle":"2022-08-11T17:35:41.544975Z","shell.execute_reply.started":"2022-08-11T17:35:41.050664Z","shell.execute_reply":"2022-08-11T17:35:41.544121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- most passengers boarded the ship in the port of Southampton, England;\n\n- the passengers who boarded the ship in the port of Southampton, England, had the most deaths;\n\n- there were more survivors among passengers who boarded the ship in the port of Cherbourg, France, than among those who boarded the ship in Queenstown, Ireland","metadata":{}},{"cell_type":"markdown","source":"## Prediction of survivors","metadata":{}},{"cell_type":"markdown","source":"### Changing a categorical feature into a quantitative one","metadata":{}},{"cell_type":"code","source":"for dataset in (train, test):    \n    dataset['Sex'] = dataset['Sex'].replace({'male': 0, 'female': 1})\n    dataset['Embarked'] = dataset['Embarked'].replace({'S': 0, 'C': 1, 'Q': 2})\n    dataset['has_siblings'] = dataset['SibSp'].replace({3: 2 , 4: 2, 5: 2, 8: 2})\n    dataset['has_parents_children'] = dataset['Parch'].replace({2: 1 , 3: 1, 4: 1, 5: 1, 6:1})\n    \ndef passenger_title(name):\n    if 'Mr.' in name:\n        return 0\n    if 'Mrs.' in name:\n        return 1\n    if 'Miss' in name:\n        return 2\n    if 'Master' in name:\n        return 3\n    return 4\n\ntrain['passenger_title'] = train['Name'].apply(passenger_title)\ntest['passenger_title'] = test['Name'].apply(passenger_title)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:41.546191Z","iopub.execute_input":"2022-08-11T17:35:41.546857Z","iopub.status.idle":"2022-08-11T17:35:41.572057Z","shell.execute_reply.started":"2022-08-11T17:35:41.546815Z","shell.execute_reply":"2022-08-11T17:35:41.570914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Filling in the blanks in the Age column with data","metadata":{}},{"cell_type":"code","source":"def fill_age(name_set):\n    if name_set == 'train':\n        age_median = train.groupby(['Survived', 'Sex', 'Pclass', 'Embarked']).agg({'Age': 'median'}).reset_index()\n    \n        for i in range(len(age_median)):\n            train.loc[(train['Survived'] == age_median['Survived'][i]) & \n                         (train['Sex'] == age_median['Sex'][i]) & \n                         (train['Pclass'] == age_median['Pclass'][i]) & \n                         (train['Embarked'] == age_median['Embarked'][i]) & \n                         train['Age'].isna(), 'Age'] = age_median['Age'][i]\n    if name_set == 'test':\n        age_median = test.groupby(['Sex', 'Pclass', 'Embarked']).agg({'Age': 'median'}).reset_index()\n    \n        for i in range(len(age_median)):\n            test.loc[(test['Embarked'] == age_median['Embarked'][i]) & \n                         (test['Sex'] == age_median['Sex'][i]) & \n                         (test['Pclass'] == age_median['Pclass'][i]) &                      \n                         test['Age'].isna(), 'Age'] = age_median['Age'][i]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:41.573708Z","iopub.execute_input":"2022-08-11T17:35:41.574698Z","iopub.status.idle":"2022-08-11T17:35:41.588102Z","shell.execute_reply.started":"2022-08-11T17:35:41.574632Z","shell.execute_reply":"2022-08-11T17:35:41.586534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for data_set in ('train', 'test'):\n    fill_age(data_set)\n    \ndisplay(train[train['Age'].isna()])\ndisplay(test[test['Age'].isna()])","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:41.589962Z","iopub.execute_input":"2022-08-11T17:35:41.590738Z","iopub.status.idle":"2022-08-11T17:35:41.733775Z","shell.execute_reply.started":"2022-08-11T17:35:41.590687Z","shell.execute_reply":"2022-08-11T17:35:41.732751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Remove  or  fix gaps in the columns","metadata":{}},{"cell_type":"markdown","source":"Let's fix in **_test_** dataset.","metadata":{}},{"cell_type":"code","source":"test[test['Fare'].isna()]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:41.735303Z","iopub.execute_input":"2022-08-11T17:35:41.735815Z","iopub.status.idle":"2022-08-11T17:35:41.752736Z","shell.execute_reply.started":"2022-08-11T17:35:41.735757Z","shell.execute_reply":"2022-08-11T17:35:41.751702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.query('Pclass == 3 and Sex == 0 and Embarked == 0').sort_values(\n    'Age', ascending=False).head(10)['correct_fare'].median()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:41.754109Z","iopub.execute_input":"2022-08-11T17:35:41.755004Z","iopub.status.idle":"2022-08-11T17:35:41.771043Z","shell.execute_reply.started":"2022-08-11T17:35:41.754969Z","shell.execute_reply":"2022-08-11T17:35:41.769745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.loc[test['PassengerId'] == 1044, 'Fare'] = 7.55\ntest.loc[test['PassengerId'] == 1044, 'correct_fare'] = 7.55\ntest.loc[test['PassengerId'] == 1044]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:41.772828Z","iopub.execute_input":"2022-08-11T17:35:41.774214Z","iopub.status.idle":"2022-08-11T17:35:41.796166Z","shell.execute_reply.started":"2022-08-11T17:35:41.774169Z","shell.execute_reply":"2022-08-11T17:35:41.795303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And remove gaps in **_train_** dataset.","metadata":{}},{"cell_type":"code","source":"train = train.loc[train['Embarked'].notna()]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:41.797493Z","iopub.execute_input":"2022-08-11T17:35:41.797949Z","iopub.status.idle":"2022-08-11T17:35:41.803431Z","shell.execute_reply.started":"2022-08-11T17:35:41.797917Z","shell.execute_reply":"2022-08-11T17:35:41.802566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Feature correlation","metadata":{}},{"cell_type":"code","source":"fig = px.imshow(round(train.corr(), 2), text_auto=True)\nfig.update_layout(plot_bgcolor='#F8F8F6',\n                  height=800,\n                  width=800,\n                  title_text=f'Feature correlation',\n                  title_font_size=20,\n                  title_font_family='Arial Black',\n                  title_x=0,\n                  title_y=0.93,                                  \n                  margin=dict(l=0, r=20, t=130, b=80))\nfig.update_coloraxes(showscale=False)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:41.804688Z","iopub.execute_input":"2022-08-11T17:35:41.805211Z","iopub.status.idle":"2022-08-11T17:35:41.896632Z","shell.execute_reply.started":"2022-08-11T17:35:41.805179Z","shell.execute_reply":"2022-08-11T17:35:41.895199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- because of the addition of new columns, multicolinear features appeared, but we will use the new columns to calculate survival prediction;\n- gender correlates more with survival than other features;\n- passenger class is more strongly correlated with fare and correct_fare, which is not surprising.","metadata":{}},{"cell_type":"markdown","source":"Let's get ready to predict.","metadata":{}},{"cell_type":"code","source":"# denote the target feature\ny_train = train['Survived']\n    \nfeatures = ['Sex', 'Pclass', 'correct_fare', 'Age', 'Embarked', 'has_parents_children', \n                'has_siblings', 'passenger_title']\nX_train = pd.get_dummies(train[features])\nX_test = pd.get_dummies(test[features])\n    \n# standardize the data using the StandartScaler method\nscaler = StandardScaler()\nscaler.fit(X_train)\n\n# convert training and validation datasets\nX_train_st = scaler.transform(X_train)\nX_test_st = scaler.transform(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:41.898261Z","iopub.execute_input":"2022-08-11T17:35:41.898698Z","iopub.status.idle":"2022-08-11T17:35:41.921297Z","shell.execute_reply.started":"2022-08-11T17:35:41.898657Z","shell.execute_reply":"2022-08-11T17:35:41.920157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### LogisticRegression","metadata":{}},{"cell_type":"code","source":"lr_model = LogisticRegression(solver='liblinear', random_state=0)\n         \n# train the model\nlr_model.fit(X_train_st, y_train)\n\n# calculate the coefficient of determination (how the prediction relates to the actual values of the target variable)\nprint(f'R2 in LogisticRegression model is: {round(lr_model.score(X_train_st, y_train), 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:41.924184Z","iopub.execute_input":"2022-08-11T17:35:41.924700Z","iopub.status.idle":"2022-08-11T17:35:41.937692Z","shell.execute_reply.started":"2022-08-11T17:35:41.924626Z","shell.execute_reply":"2022-08-11T17:35:41.936511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### RandomForestClassifier","metadata":{}},{"cell_type":"code","source":"rfc_model = RandomForestClassifier(n_estimators = 200, random_state = 0)\n         \n# train the model\nrfc_model.fit(X_train_st, y_train)\n\n# calculate the coefficient of determination (how the prediction relates to the actual values of the target variable)\nprint(f'R2 in RandomForestClassifier is: {round(rfc_model.score(X_train_st, y_train), 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:41.939212Z","iopub.execute_input":"2022-08-11T17:35:41.939842Z","iopub.status.idle":"2022-08-11T17:35:42.393891Z","shell.execute_reply.started":"2022-08-11T17:35:41.939781Z","shell.execute_reply":"2022-08-11T17:35:42.392762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### DecisionTreeClassifier","metadata":{}},{"cell_type":"code","source":"dtc_model = DecisionTreeClassifier(random_state=0)\n         \n# train the model\ndtc_model.fit(X_train_st, y_train)\n\n# calculate the coefficient of determination (how the prediction relates to the actual values of the target variable)\nprint(f'R2 in DecisionTreeClassifier is: {round(dtc_model.score(X_train_st, y_train), 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:42.395199Z","iopub.execute_input":"2022-08-11T17:35:42.395553Z","iopub.status.idle":"2022-08-11T17:35:42.405927Z","shell.execute_reply.started":"2022-08-11T17:35:42.395521Z","shell.execute_reply":"2022-08-11T17:35:42.404858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### GradientBoostingClassifier","metadata":{}},{"cell_type":"code","source":"gbc_model = GradientBoostingClassifier(n_estimators = 300, random_state = 0)\n         \n# train the model\ngbc_model.fit(X_train_st, y_train)\n\n# calculate the coefficient of determination (how the prediction relates to the actual values of the target variable)\nprint(f'R2 in GradientBoostingClassifier is: {round(gbc_model.score(X_train_st, y_train), 2)}')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:42.407563Z","iopub.execute_input":"2022-08-11T17:35:42.407966Z","iopub.status.idle":"2022-08-11T17:35:42.786759Z","shell.execute_reply.started":"2022-08-11T17:35:42.407933Z","shell.execute_reply":"2022-08-11T17:35:42.785487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"RandomForestClassifier and DecisionTreeClassifier showed the best rsults. I choose the RandomForestClassifier model prediction. Let's save the prediction to a file.","metadata":{}},{"cell_type":"code","source":"predictions = rfc_model.predict(X_test_st)\n\n#output = pd.DataFrame({'PassengerId': test.PassengerId, 'Survived': predictions})\n#output.to_csv('submission.csv', index=False)\n#print(\"Your submission was successfully saved!\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T17:35:42.788771Z","iopub.execute_input":"2022-08-11T17:35:42.789738Z","iopub.status.idle":"2022-08-11T17:35:42.841010Z","shell.execute_reply.started":"2022-08-11T17:35:42.789701Z","shell.execute_reply":"2022-08-11T17:35:42.840038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}