{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install datasist","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-18T18:25:19.884250Z","iopub.execute_input":"2022-07-18T18:25:19.884937Z","iopub.status.idle":"2022-07-18T18:25:29.288342Z","shell.execute_reply.started":"2022-07-18T18:25:19.884895Z","shell.execute_reply":"2022-07-18T18:25:29.287076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import Required libraries:\nimport numpy as np \nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import confusion_matrix\nimport seaborn as sb\nimport datasist as ds","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:29.291912Z","iopub.execute_input":"2022-07-18T18:25:29.292272Z","iopub.status.idle":"2022-07-18T18:25:29.300274Z","shell.execute_reply.started":"2022-07-18T18:25:29.292238Z","shell.execute_reply":"2022-07-18T18:25:29.299017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load Train and Test Data: \ndf_train = pd.read_csv('/kaggle/input/titanic/train.csv')\ndf_test= pd.read_csv('/kaggle/input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:29.301447Z","iopub.execute_input":"2022-07-18T18:25:29.301789Z","iopub.status.idle":"2022-07-18T18:25:29.326273Z","shell.execute_reply.started":"2022-07-18T18:25:29.301760Z","shell.execute_reply":"2022-07-18T18:25:29.325114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check structure of first 2 rows of data:\ndf_train.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:29.329955Z","iopub.execute_input":"2022-07-18T18:25:29.330355Z","iopub.status.idle":"2022-07-18T18:25:29.345143Z","shell.execute_reply.started":"2022-07-18T18:25:29.330322Z","shell.execute_reply":"2022-07-18T18:25:29.344238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:29.346402Z","iopub.execute_input":"2022-07-18T18:25:29.346749Z","iopub.status.idle":"2022-07-18T18:25:29.366334Z","shell.execute_reply.started":"2022-07-18T18:25:29.346718Z","shell.execute_reply":"2022-07-18T18:25:29.365427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get Summarized information about data:\ndf_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:29.367906Z","iopub.execute_input":"2022-07-18T18:25:29.368357Z","iopub.status.idle":"2022-07-18T18:25:29.383826Z","shell.execute_reply.started":"2022-07-18T18:25:29.368315Z","shell.execute_reply":"2022-07-18T18:25:29.382578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:29.386658Z","iopub.execute_input":"2022-07-18T18:25:29.387471Z","iopub.status.idle":"2022-07-18T18:25:29.402143Z","shell.execute_reply.started":"2022-07-18T18:25:29.387423Z","shell.execute_reply":"2022-07-18T18:25:29.401131Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#comine train and test datasets for further preprocessing:\ncombined= [df_train, df_test]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:29.404355Z","iopub.execute_input":"2022-07-18T18:25:29.405211Z","iopub.status.idle":"2022-07-18T18:25:29.410099Z","shell.execute_reply.started":"2022-07-18T18:25:29.405174Z","shell.execute_reply":"2022-07-18T18:25:29.408812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Select Appropriate data type for each feature:\nfor dataset in combined:\n    dataset[['PassengerId', 'Pclass']]= dataset[['PassengerId', 'Pclass']].astype('str')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:29.411287Z","iopub.execute_input":"2022-07-18T18:25:29.411740Z","iopub.status.idle":"2022-07-18T18:25:29.428055Z","shell.execute_reply.started":"2022-07-18T18:25:29.411712Z","shell.execute_reply":"2022-07-18T18:25:29.426948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check data type for each feature in train data:\ndf_train.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:30.664407Z","iopub.execute_input":"2022-07-18T18:25:30.664811Z","iopub.status.idle":"2022-07-18T18:25:30.672434Z","shell.execute_reply.started":"2022-07-18T18:25:30.664775Z","shell.execute_reply":"2022-07-18T18:25:30.671673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check data type for each feature in test data:\ndf_test.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:32.575838Z","iopub.execute_input":"2022-07-18T18:25:32.576226Z","iopub.status.idle":"2022-07-18T18:25:32.584594Z","shell.execute_reply.started":"2022-07-18T18:25:32.576193Z","shell.execute_reply":"2022-07-18T18:25:32.583701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Dropping features that will never be used in future analysis:\nfor dataset in combined:\n    dataset.drop(['Cabin','Ticket'],axis=1,inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:34.217156Z","iopub.execute_input":"2022-07-18T18:25:34.217815Z","iopub.status.idle":"2022-07-18T18:25:34.227485Z","shell.execute_reply.started":"2022-07-18T18:25:34.217777Z","shell.execute_reply":"2022-07-18T18:25:34.226382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check that unneeded columns has been droped from train data:\n('Cabin' or 'Ticket') in (df_train.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:35.311922Z","iopub.execute_input":"2022-07-18T18:25:35.312297Z","iopub.status.idle":"2022-07-18T18:25:35.319302Z","shell.execute_reply.started":"2022-07-18T18:25:35.312263Z","shell.execute_reply":"2022-07-18T18:25:35.318153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check that unneeded columns has been droped from test data:\n('Cabin' or 'Ticket') in (df_test.columns)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:36.817048Z","iopub.execute_input":"2022-07-18T18:25:36.817609Z","iopub.status.idle":"2022-07-18T18:25:36.823575Z","shell.execute_reply.started":"2022-07-18T18:25:36.817577Z","shell.execute_reply":"2022-07-18T18:25:36.822575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get statistcal summary for train data:\ndf_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:38.022974Z","iopub.execute_input":"2022-07-18T18:25:38.024217Z","iopub.status.idle":"2022-07-18T18:25:38.051005Z","shell.execute_reply.started":"2022-07-18T18:25:38.024165Z","shell.execute_reply":"2022-07-18T18:25:38.049880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get statistcal summary for test data:\ndf_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:39.208349Z","iopub.execute_input":"2022-07-18T18:25:39.208783Z","iopub.status.idle":"2022-07-18T18:25:39.240494Z","shell.execute_reply.started":"2022-07-18T18:25:39.208748Z","shell.execute_reply":"2022-07-18T18:25:39.239231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get description for categorical feature in train data:\ndf_train.describe(include= object)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:40.241074Z","iopub.execute_input":"2022-07-18T18:25:40.241455Z","iopub.status.idle":"2022-07-18T18:25:40.265856Z","shell.execute_reply.started":"2022-07-18T18:25:40.241423Z","shell.execute_reply":"2022-07-18T18:25:40.264726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Get description for categorical feature in train data:\ndf_test.describe(include= object)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:41.168765Z","iopub.execute_input":"2022-07-18T18:25:41.169167Z","iopub.status.idle":"2022-07-18T18:25:41.193178Z","shell.execute_reply.started":"2022-07-18T18:25:41.169130Z","shell.execute_reply":"2022-07-18T18:25:41.192127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Exploratory Data Analysis (EDA)\n- Finding trends, relationships between different feature, build new features and build a basic intution about data.\n\n- You might experience some semi-polished visualization too.\n\n- Most of our EDA will be on train dataset.","metadata":{}},{"cell_type":"code","source":"# Show out distrbution of Age:\nbins= np.arange(0,df_train['Age'].max()+5,5)\nplt.hist(df_train['Age'],bins=bins);\nplt.title('Age');","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:42.124950Z","iopub.execute_input":"2022-07-18T18:25:42.125338Z","iopub.status.idle":"2022-07-18T18:25:42.336400Z","shell.execute_reply.started":"2022-07-18T18:25:42.125305Z","shell.execute_reply":"2022-07-18T18:25:42.335526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show out distrbution of Fare:\n# We can Initially detect ouliers in Fare feature\nbins= np.arange(0,df_train['Fare'].max()+50,50)\nplt.hist(df_train['Fare'],bins=bins)\nplt.title('Fare');","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:43.176595Z","iopub.execute_input":"2022-07-18T18:25:43.177281Z","iopub.status.idle":"2022-07-18T18:25:43.326824Z","shell.execute_reply.started":"2022-07-18T18:25:43.177243Z","shell.execute_reply":"2022-07-18T18:25:43.325509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show out distrbution of Age:\n# Change x limits to focus on distribution\nbins= np.arange(0,df_train['Fare'].max()+5,5)\nplt.hist(df_train['Fare'],bins=bins);\nplt.xlim(0,200)\nplt.title('Fare');","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:44.515955Z","iopub.execute_input":"2022-07-18T18:25:44.517215Z","iopub.status.idle":"2022-07-18T18:25:44.843865Z","shell.execute_reply.started":"2022-07-18T18:25:44.517157Z","shell.execute_reply":"2022-07-18T18:25:44.842725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show out distribution of sibiling & Spouns (SibSp) feature:\n# I prefer using #Bar chart with discrete variables than #Histograms\n# It seems that most of passengers didn't have any sibiling or spoune\nbasecolor= sb.color_palette()[6]\nsb.countplot(x= df_train['SibSp'],color=basecolor);","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:45.811602Z","iopub.execute_input":"2022-07-18T18:25:45.812007Z","iopub.status.idle":"2022-07-18T18:25:45.991622Z","shell.execute_reply.started":"2022-07-18T18:25:45.811976Z","shell.execute_reply":"2022-07-18T18:25:45.990611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Show out distribution of Paraents & Child (Parch) feature:\n# I prefer using #Bar chart with discrete variables than #Histograms\n# It seems that most of passengers didn't have a parents or child\nbasecolor= sb.color_palette()[6]\nsb.countplot(x= df_train['Parch'],color=basecolor);","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:47.749864Z","iopub.execute_input":"2022-07-18T18:25:47.750760Z","iopub.status.idle":"2022-07-18T18:25:47.933681Z","shell.execute_reply.started":"2022-07-18T18:25:47.750714Z","shell.execute_reply":"2022-07-18T18:25:47.932439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# distribution of age for each survived subcatagory:\ng= sb.FacetGrid(data= df_train, col='Survived')\ng.map(plt.hist, 'Age')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:49.394292Z","iopub.execute_input":"2022-07-18T18:25:49.394983Z","iopub.status.idle":"2022-07-18T18:25:49.771749Z","shell.execute_reply.started":"2022-07-18T18:25:49.394946Z","shell.execute_reply":"2022-07-18T18:25:49.770628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distribution of age for each sex:\ng= sb.FacetGrid(data= df_train, col='Sex')\ng.map(plt.hist, 'Age')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:51.924442Z","iopub.execute_input":"2022-07-18T18:25:51.924849Z","iopub.status.idle":"2022-07-18T18:25:52.299285Z","shell.execute_reply.started":"2022-07-18T18:25:51.924813Z","shell.execute_reply":"2022-07-18T18:25:52.298278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Distribution of age for each survived and sex subcatagory:\ng= sb.FacetGrid(data= df_train, col='Sex', row='Survived')\ng.map(plt.hist, 'Age');","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:54.753157Z","iopub.execute_input":"2022-07-18T18:25:54.753544Z","iopub.status.idle":"2022-07-18T18:25:55.411051Z","shell.execute_reply.started":"2022-07-18T18:25:54.753513Z","shell.execute_reply":"2022-07-18T18:25:55.409905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting distribution of age for each  \ng= sb.FacetGrid(data=df_train, col='Pclass')\ng.map(plt.hist,'Age');","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:57.398370Z","iopub.execute_input":"2022-07-18T18:25:57.398772Z","iopub.status.idle":"2022-07-18T18:25:58.402984Z","shell.execute_reply.started":"2022-07-18T18:25:57.398735Z","shell.execute_reply":"2022-07-18T18:25:58.401737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting count of each sex for each survived and each Pclass subcategoryL\ng= sb.FacetGrid(data=df_train, row='Survived' , col='Pclass')\ng.map(sb.countplot,'Sex',order=None);\n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:25:59.890822Z","iopub.execute_input":"2022-07-18T18:25:59.891224Z","iopub.status.idle":"2022-07-18T18:26:00.632486Z","shell.execute_reply.started":"2022-07-18T18:25:59.891189Z","shell.execute_reply":"2022-07-18T18:26:00.631218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting distribution of age for each sex and class:\ng= sb.FacetGrid(data=df_train, row='Sex' , col='Pclass')\ng.map(plt.hist,'Age');\n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:02.319557Z","iopub.execute_input":"2022-07-18T18:26:02.320469Z","iopub.status.idle":"2022-07-18T18:26:03.271455Z","shell.execute_reply.started":"2022-07-18T18:26:02.320419Z","shell.execute_reply":"2022-07-18T18:26:03.270057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check whether or not Number of Parch will influence Survived:\n# It seems that survived and notsurvived have almost near number of parch\ndf_train.groupby('Survived')['Parch'].sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:04.483818Z","iopub.execute_input":"2022-07-18T18:26:04.484721Z","iopub.status.idle":"2022-07-18T18:26:04.493311Z","shell.execute_reply.started":"2022-07-18T18:26:04.484676Z","shell.execute_reply":"2022-07-18T18:26:04.492420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"g= sb.FacetGrid(data= df_train , col='Survived')\ng.map(sb.countplot, 'Parch',order=None);\n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:06.071732Z","iopub.execute_input":"2022-07-18T18:26:06.072335Z","iopub.status.idle":"2022-07-18T18:26:06.409834Z","shell.execute_reply.started":"2022-07-18T18:26:06.072289Z","shell.execute_reply":"2022-07-18T18:26:06.408688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check whether or not Number of Sibsp will influence Survived:\n# It seems that survived and notsurvived have almost near number of Sibsp\ndf_train.groupby('Survived')['SibSp'].sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:06.779064Z","iopub.execute_input":"2022-07-18T18:26:06.779450Z","iopub.status.idle":"2022-07-18T18:26:06.788022Z","shell.execute_reply.started":"2022-07-18T18:26:06.779416Z","shell.execute_reply":"2022-07-18T18:26:06.787085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ploting percentage of survived for each sex:\nsb.barplot(data= df_train, y='Survived' , x='Sex',color= basecolor)\nplt.title('Sex Vs Survived');","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:07.771060Z","iopub.execute_input":"2022-07-18T18:26:07.772237Z","iopub.status.idle":"2022-07-18T18:26:07.990248Z","shell.execute_reply.started":"2022-07-18T18:26:07.772195Z","shell.execute_reply":"2022-07-18T18:26:07.988930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting values of Sex feature from 'Male' & 'Female' into '1' & '0':\nfor dataset in combined:\n    dataset.loc[dataset['Sex']=='male','Sex']= 1\n    dataset.loc[dataset['Sex']=='female','Sex']= 0","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:09.746269Z","iopub.execute_input":"2022-07-18T18:26:09.746936Z","iopub.status.idle":"2022-07-18T18:26:09.755269Z","shell.execute_reply.started":"2022-07-18T18:26:09.746890Z","shell.execute_reply":"2022-07-18T18:26:09.754282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Make sure that values of sex feature have been modified:\ndf_train['Sex'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:11.466560Z","iopub.execute_input":"2022-07-18T18:26:11.467273Z","iopub.status.idle":"2022-07-18T18:26:11.473215Z","shell.execute_reply.started":"2022-07-18T18:26:11.467226Z","shell.execute_reply":"2022-07-18T18:26:11.472170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting Pclass and sex datatypes into int:\nfor dataset in combined:\n    dataset['Pclass']= dataset['Pclass'].astype('int')\n    dataset['Sex']= dataset['Sex'].astype('int')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:13.423981Z","iopub.execute_input":"2022-07-18T18:26:13.424359Z","iopub.status.idle":"2022-07-18T18:26:13.432216Z","shell.execute_reply.started":"2022-07-18T18:26:13.424328Z","shell.execute_reply":"2022-07-18T18:26:13.430579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['Sex'].dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:16.147673Z","iopub.execute_input":"2022-07-18T18:26:16.148673Z","iopub.status.idle":"2022-07-18T18:26:16.155608Z","shell.execute_reply.started":"2022-07-18T18:26:16.148613Z","shell.execute_reply":"2022-07-18T18:26:16.154499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compensate missing Values of age:\nfor dataset in combined:\n    for gender in range(2): # 0 for Female & 1 for Male\n        for c in range(3):  # 1 for 1stclass & 2 for 2ndclass & 3 for 3rdclass\n            mean= dataset[(dataset['Sex']== gender) & (dataset['Pclass']==c+1)]['Age'].mean()            \n            dataset.loc[(dataset['Age'].isnull()) & (dataset['Sex']==gender) & (dataset['Pclass']== c+1) ,'Age']= mean","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:18.300010Z","iopub.execute_input":"2022-07-18T18:26:18.300911Z","iopub.status.idle":"2022-07-18T18:26:18.335892Z","shell.execute_reply.started":"2022-07-18T18:26:18.300861Z","shell.execute_reply":"2022-07-18T18:26:18.334889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['Age'].isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:20.324569Z","iopub.execute_input":"2022-07-18T18:26:20.325546Z","iopub.status.idle":"2022-07-18T18:26:20.332037Z","shell.execute_reply.started":"2022-07-18T18:26:20.325507Z","shell.execute_reply":"2022-07-18T18:26:20.331098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check outliers in fare:\nimport datasist as ds\noutliers = ds.structdata.detect_outliers(df_train, 0, ['Fare'])\nlen(outliers)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:22.340686Z","iopub.execute_input":"2022-07-18T18:26:22.344002Z","iopub.status.idle":"2022-07-18T18:26:22.354280Z","shell.execute_reply.started":"2022-07-18T18:26:22.343952Z","shell.execute_reply":"2022-07-18T18:26:22.353558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop outliers from fare:\ndf_train.drop(outliers, axis=0, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:24.500262Z","iopub.execute_input":"2022-07-18T18:26:24.501363Z","iopub.status.idle":"2022-07-18T18:26:24.507218Z","shell.execute_reply.started":"2022-07-18T18:26:24.501323Z","shell.execute_reply":"2022-07-18T18:26:24.506117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check Patterns between Survived categories and fare's cost:\nsb.violinplot(data= df_train , x= 'Survived' , y='Fare', inner='quartile')","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:26.626931Z","iopub.execute_input":"2022-07-18T18:26:26.627300Z","iopub.status.idle":"2022-07-18T18:26:26.793673Z","shell.execute_reply.started":"2022-07-18T18:26:26.627270Z","shell.execute_reply":"2022-07-18T18:26:26.792407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find out mean average of fare for survived and non survived :\nmean_fare= df_train.groupby('Survived').mean()['Fare']\nprint(mean_fare)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:28.670127Z","iopub.execute_input":"2022-07-18T18:26:28.670532Z","iopub.status.idle":"2022-07-18T18:26:28.680534Z","shell.execute_reply.started":"2022-07-18T18:26:28.670497Z","shell.execute_reply":"2022-07-18T18:26:28.679384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plotting fare for survived subcatagory\nx=mean_fare.index\ny=mean_fare.values\nplt.bar(x,y)\nplt.xlabel('Survived')\nplt.ylabel('Fare')\nplt.title('Survived Vs Average Fare');\nplt.xticks([0,1],['0','1']);","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:31.129977Z","iopub.execute_input":"2022-07-18T18:26:31.130583Z","iopub.status.idle":"2022-07-18T18:26:31.272776Z","shell.execute_reply.started":"2022-07-18T18:26:31.130548Z","shell.execute_reply":"2022-07-18T18:26:31.271403Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Engineering new feature \"family\" by sum. 'sibSp' , 'Parch'\nfor dataset in combined:\n    dataset['family'] = dataset['SibSp'] + dataset['Parch'] + 1","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:33.245139Z","iopub.execute_input":"2022-07-18T18:26:33.245936Z","iopub.status.idle":"2022-07-18T18:26:33.255739Z","shell.execute_reply.started":"2022-07-18T18:26:33.245887Z","shell.execute_reply":"2022-07-18T18:26:33.254207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_family= df_train.groupby('Survived')['family'].mean()\nprint(mean_family)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:34.615572Z","iopub.execute_input":"2022-07-18T18:26:34.617018Z","iopub.status.idle":"2022-07-18T18:26:34.624797Z","shell.execute_reply.started":"2022-07-18T18:26:34.616969Z","shell.execute_reply":"2022-07-18T18:26:34.623602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x=mean_family.index\ny=mean_family.values\nplt.bar(x,y)\nplt.xlabel('Survived')\nplt.ylabel('Family number')\nplt.title('Survived Vs Average Family Number')\nplt.xticks([0,1],['0','1']);","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:35.320527Z","iopub.execute_input":"2022-07-18T18:26:35.321497Z","iopub.status.idle":"2022-07-18T18:26:35.482795Z","shell.execute_reply.started":"2022-07-18T18:26:35.321442Z","shell.execute_reply":"2022-07-18T18:26:35.481606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Engineering a new feature classyfing passengers into alone or not alone:\nfor dataset in combined:\n    dataset['isalone']=0\n    dataset.loc[dataset['family']!=1 , 'isalone']==0\n    dataset.loc[dataset['family'] ==1 , 'isalone']==1","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:36.118439Z","iopub.execute_input":"2022-07-18T18:26:36.119561Z","iopub.status.idle":"2022-07-18T18:26:36.129909Z","shell.execute_reply.started":"2022-07-18T18:26:36.119480Z","shell.execute_reply":"2022-07-18T18:26:36.128826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PassengerId = df_test['PassengerId']","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:36.467602Z","iopub.execute_input":"2022-07-18T18:26:36.468031Z","iopub.status.idle":"2022-07-18T18:26:36.473165Z","shell.execute_reply.started":"2022-07-18T18:26:36.467993Z","shell.execute_reply":"2022-07-18T18:26:36.472019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for dataset in combined:\n    dataset.drop(['PassengerId','SibSp','Parch','Name'],axis=1, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:37.350577Z","iopub.execute_input":"2022-07-18T18:26:37.351576Z","iopub.status.idle":"2022-07-18T18:26:37.359251Z","shell.execute_reply.started":"2022-07-18T18:26:37.351529Z","shell.execute_reply":"2022-07-18T18:26:37.358451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['Embarked'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:38.285798Z","iopub.execute_input":"2022-07-18T18:26:38.286568Z","iopub.status.idle":"2022-07-18T18:26:38.295166Z","shell.execute_reply.started":"2022-07-18T18:26:38.286522Z","shell.execute_reply":"2022-07-18T18:26:38.293997Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dict= {'S':1, 'Q':2 , 'C':3}\nfor dataset in combined:\n    dataset['Embarked']= dataset['Embarked'].map(dict)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:38.830127Z","iopub.execute_input":"2022-07-18T18:26:38.830523Z","iopub.status.idle":"2022-07-18T18:26:38.840440Z","shell.execute_reply.started":"2022-07-18T18:26:38.830480Z","shell.execute_reply":"2022-07-18T18:26:38.839431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:39.300915Z","iopub.execute_input":"2022-07-18T18:26:39.301568Z","iopub.status.idle":"2022-07-18T18:26:39.315568Z","shell.execute_reply.started":"2022-07-18T18:26:39.301520Z","shell.execute_reply":"2022-07-18T18:26:39.314553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:39.776468Z","iopub.execute_input":"2022-07-18T18:26:39.776870Z","iopub.status.idle":"2022-07-18T18:26:39.789374Z","shell.execute_reply.started":"2022-07-18T18:26:39.776837Z","shell.execute_reply":"2022-07-18T18:26:39.788422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:40.613254Z","iopub.execute_input":"2022-07-18T18:26:40.614495Z","iopub.status.idle":"2022-07-18T18:26:40.628562Z","shell.execute_reply.started":"2022-07-18T18:26:40.614443Z","shell.execute_reply":"2022-07-18T18:26:40.627173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:41.015781Z","iopub.execute_input":"2022-07-18T18:26:41.016175Z","iopub.status.idle":"2022-07-18T18:26:41.030123Z","shell.execute_reply.started":"2022-07-18T18:26:41.016135Z","shell.execute_reply":"2022-07-18T18:26:41.028893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compensate missing values in Fare:\ndf_test['Fare'].fillna(df_test['Fare'].mean(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:41.392521Z","iopub.execute_input":"2022-07-18T18:26:41.393203Z","iopub.status.idle":"2022-07-18T18:26:41.399299Z","shell.execute_reply.started":"2022-07-18T18:26:41.393154Z","shell.execute_reply":"2022-07-18T18:26:41.398357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Data Separation:\ny_train= df_train.loc[:,'Survived']\nx_train= df_train.iloc[:,1:]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:42.890569Z","iopub.execute_input":"2022-07-18T18:26:42.891221Z","iopub.status.idle":"2022-07-18T18:26:42.895696Z","shell.execute_reply.started":"2022-07-18T18:26:42.891185Z","shell.execute_reply":"2022-07-18T18:26:42.894874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check shape of x_train & y_train:\nx_train.shape , y_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:44.866801Z","iopub.execute_input":"2022-07-18T18:26:44.867454Z","iopub.status.idle":"2022-07-18T18:26:44.875252Z","shell.execute_reply.started":"2022-07-18T18:26:44.867406Z","shell.execute_reply":"2022-07-18T18:26:44.874493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Building Logistic Regression Model:\nlr= LogisticRegression(solver='liblinear', penalty='l2')\nlr.fit(x_train , y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:46.607461Z","iopub.execute_input":"2022-07-18T18:26:46.608178Z","iopub.status.idle":"2022-07-18T18:26:46.622156Z","shell.execute_reply.started":"2022-07-18T18:26:46.608135Z","shell.execute_reply":"2022-07-18T18:26:46.620920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Measuring Model Score:\nlr.score(x_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:48.248210Z","iopub.execute_input":"2022-07-18T18:26:48.248622Z","iopub.status.idle":"2022-07-18T18:26:48.258871Z","shell.execute_reply.started":"2022-07-18T18:26:48.248584Z","shell.execute_reply":"2022-07-18T18:26:48.258054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Ensembel classifier: \nrf= RandomForestClassifier(n_estimators=100)\nrf.fit(x_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:49.755652Z","iopub.execute_input":"2022-07-18T18:26:49.756755Z","iopub.status.idle":"2022-07-18T18:26:49.928586Z","shell.execute_reply.started":"2022-07-18T18:26:49.756711Z","shell.execute_reply":"2022-07-18T18:26:49.927407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf.score(x_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:51.688520Z","iopub.execute_input":"2022-07-18T18:26:51.688991Z","iopub.status.idle":"2022-07-18T18:26:51.724028Z","shell.execute_reply.started":"2022-07-18T18:26:51.688952Z","shell.execute_reply":"2022-07-18T18:26:51.722881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:52.644020Z","iopub.execute_input":"2022-07-18T18:26:52.644405Z","iopub.status.idle":"2022-07-18T18:26:52.657057Z","shell.execute_reply.started":"2022-07-18T18:26:52.644368Z","shell.execute_reply":"2022-07-18T18:26:52.655887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred= rf.predict(df_test)\ny_pred.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:26:53.574270Z","iopub.execute_input":"2022-07-18T18:26:53.574624Z","iopub.status.idle":"2022-07-18T18:26:53.603279Z","shell.execute_reply.started":"2022-07-18T18:26:53.574594Z","shell.execute_reply":"2022-07-18T18:26:53.602189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'PassengerId': PassengerId, 'Survived': y_pred})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T18:31:14.044719Z","iopub.execute_input":"2022-07-18T18:31:14.045163Z","iopub.status.idle":"2022-07-18T18:31:14.053635Z","shell.execute_reply.started":"2022-07-18T18:31:14.045126Z","shell.execute_reply":"2022-07-18T18:31:14.052630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}