{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Importing the libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import accuracy_score , classification_report , confusion_matrix\n","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:19.396422Z","iopub.execute_input":"2022-07-04T21:13:19.396828Z","iopub.status.idle":"2022-07-04T21:13:19.404826Z","shell.execute_reply.started":"2022-07-04T21:13:19.396797Z","shell.execute_reply":"2022-07-04T21:13:19.403596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Loading the training data\ndf = pd.read_csv(\"../input/titanic/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:19.412984Z","iopub.execute_input":"2022-07-04T21:13:19.413960Z","iopub.status.idle":"2022-07-04T21:13:19.441139Z","shell.execute_reply.started":"2022-07-04T21:13:19.413912Z","shell.execute_reply":"2022-07-04T21:13:19.439707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# printing the first 5 rows of the dataframe\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:19.754025Z","iopub.execute_input":"2022-07-04T21:13:19.755434Z","iopub.status.idle":"2022-07-04T21:13:19.787457Z","shell.execute_reply.started":"2022-07-04T21:13:19.755372Z","shell.execute_reply":"2022-07-04T21:13:19.786184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the shape\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:19.790085Z","iopub.execute_input":"2022-07-04T21:13:19.790662Z","iopub.status.idle":"2022-07-04T21:13:19.798871Z","shell.execute_reply.started":"2022-07-04T21:13:19.790599Z","shell.execute_reply":"2022-07-04T21:13:19.797203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking the datatypes and null/non-null distribution\ndf.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:19.800113Z","iopub.execute_input":"2022-07-04T21:13:19.800776Z","iopub.status.idle":"2022-07-04T21:13:19.833037Z","shell.execute_reply.started":"2022-07-04T21:13:19.800739Z","shell.execute_reply":"2022-07-04T21:13:19.831738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check the number of missing values in each column\n(df.isna().sum()/len(df))*100","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:19.834734Z","iopub.execute_input":"2022-07-04T21:13:19.835108Z","iopub.status.idle":"2022-07-04T21:13:19.850918Z","shell.execute_reply.started":"2022-07-04T21:13:19.835074Z","shell.execute_reply":"2022-07-04T21:13:19.849043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop the columns that has missing values more than 70%\ndf = df.drop(columns='Cabin', axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:19.854201Z","iopub.execute_input":"2022-07-04T21:13:19.855376Z","iopub.status.idle":"2022-07-04T21:13:19.868041Z","shell.execute_reply.started":"2022-07-04T21:13:19.855329Z","shell.execute_reply":"2022-07-04T21:13:19.867022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replacing the missing values in \"Age\" column with mean value\ndf['Age'].fillna(df['Age'].mean(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:19.870481Z","iopub.execute_input":"2022-07-04T21:13:19.871246Z","iopub.status.idle":"2022-07-04T21:13:19.880975Z","shell.execute_reply.started":"2022-07-04T21:13:19.871192Z","shell.execute_reply":"2022-07-04T21:13:19.879844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replacing the missing values in \"Age\" column with S\ndf.Embarked.fillna(\"S\", inplace  = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:19.883222Z","iopub.execute_input":"2022-07-04T21:13:19.884099Z","iopub.status.idle":"2022-07-04T21:13:19.893990Z","shell.execute_reply.started":"2022-07-04T21:13:19.884056Z","shell.execute_reply":"2022-07-04T21:13:19.892324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:19.896966Z","iopub.execute_input":"2022-07-04T21:13:19.897846Z","iopub.status.idle":"2022-07-04T21:13:19.915314Z","shell.execute_reply.started":"2022-07-04T21:13:19.897804Z","shell.execute_reply":"2022-07-04T21:13:19.913908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# getting some statistical measures about the data\ndf.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:19.917220Z","iopub.execute_input":"2022-07-04T21:13:19.917810Z","iopub.status.idle":"2022-07-04T21:13:19.959555Z","shell.execute_reply.started":"2022-07-04T21:13:19.917771Z","shell.execute_reply":"2022-07-04T21:13:19.958230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# finding the number of people survived and not survived\ndf['Survived'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:19.961495Z","iopub.execute_input":"2022-07-04T21:13:19.961967Z","iopub.status.idle":"2022-07-04T21:13:19.972285Z","shell.execute_reply.started":"2022-07-04T21:13:19.961932Z","shell.execute_reply":"2022-07-04T21:13:19.970802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the Survied precentage\n\nprint((df.groupby('Survived')['Survived'].count()/df['Survived'].count()) *100)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:19.974056Z","iopub.execute_input":"2022-07-04T21:13:19.974457Z","iopub.status.idle":"2022-07-04T21:13:19.985068Z","shell.execute_reply.started":"2022-07-04T21:13:19.974424Z","shell.execute_reply":"2022-07-04T21:13:19.983723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking the survived distribution\n\n(df.groupby('Survived')['Survived'].count()/df['Survived'].count() *100).plot.pie()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:19.987236Z","iopub.execute_input":"2022-07-04T21:13:19.988202Z","iopub.status.idle":"2022-07-04T21:13:20.168143Z","shell.execute_reply.started":"2022-07-04T21:13:19.988136Z","shell.execute_reply":"2022-07-04T21:13:20.166641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# finding the number of males and females\ndf['Sex'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:20.172796Z","iopub.execute_input":"2022-07-04T21:13:20.173251Z","iopub.status.idle":"2022-07-04T21:13:20.186752Z","shell.execute_reply.started":"2022-07-04T21:13:20.173216Z","shell.execute_reply":"2022-07-04T21:13:20.184526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# making a count plot for \"Gender\" column\n\nplt.figure(figsize=(7,5))\nsns.countplot(df['Sex'])\nplt.ylabel(\"Count\", fontsize=15)\nplt.xlabel(\"Gender\", fontsize=15)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:20.188105Z","iopub.execute_input":"2022-07-04T21:13:20.189639Z","iopub.status.idle":"2022-07-04T21:13:20.397449Z","shell.execute_reply.started":"2022-07-04T21:13:20.189571Z","shell.execute_reply":"2022-07-04T21:13:20.396230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# number of survivors Gender wise\nsns.countplot('Sex', hue='Survived', data=df)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:20.399220Z","iopub.execute_input":"2022-07-04T21:13:20.399714Z","iopub.status.idle":"2022-07-04T21:13:20.606776Z","shell.execute_reply.started":"2022-07-04T21:13:20.399669Z","shell.execute_reply":"2022-07-04T21:13:20.605519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# converting categorical Columns\n\ndf.replace({'Sex':{'male':0,'female':1}, 'Embarked':{'S':0,'C':1,'Q':2}}, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:20.607925Z","iopub.execute_input":"2022-07-04T21:13:20.609077Z","iopub.status.idle":"2022-07-04T21:13:20.624194Z","shell.execute_reply.started":"2022-07-04T21:13:20.609010Z","shell.execute_reply":"2022-07-04T21:13:20.622348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:20.627003Z","iopub.execute_input":"2022-07-04T21:13:20.628541Z","iopub.status.idle":"2022-07-04T21:13:20.655628Z","shell.execute_reply.started":"2022-07-04T21:13:20.628469Z","shell.execute_reply":"2022-07-04T21:13:20.653858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Drop unnecessary columns and save the others in X\nX = df.drop(columns = ['PassengerId','Name','Ticket','Survived'],axis=1)\ny = df['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:20.658064Z","iopub.execute_input":"2022-07-04T21:13:20.658709Z","iopub.status.idle":"2022-07-04T21:13:20.669975Z","shell.execute_reply.started":"2022-07-04T21:13:20.658657Z","shell.execute_reply":"2022-07-04T21:13:20.668886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:20.673047Z","iopub.execute_input":"2022-07-04T21:13:20.674061Z","iopub.status.idle":"2022-07-04T21:13:20.690436Z","shell.execute_reply.started":"2022-07-04T21:13:20.673986Z","shell.execute_reply":"2022-07-04T21:13:20.689059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:20.692080Z","iopub.execute_input":"2022-07-04T21:13:20.692879Z","iopub.status.idle":"2022-07-04T21:13:20.708542Z","shell.execute_reply.started":"2022-07-04T21:13:20.692831Z","shell.execute_reply":"2022-07-04T21:13:20.707173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(X)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:20.710896Z","iopub.execute_input":"2022-07-04T21:13:20.711324Z","iopub.status.idle":"2022-07-04T21:13:20.727670Z","shell.execute_reply.started":"2022-07-04T21:13:20.711290Z","shell.execute_reply":"2022-07-04T21:13:20.726243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(y)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:20.729413Z","iopub.execute_input":"2022-07-04T21:13:20.730291Z","iopub.status.idle":"2022-07-04T21:13:20.742533Z","shell.execute_reply.started":"2022-07-04T21:13:20.730182Z","shell.execute_reply":"2022-07-04T21:13:20.741343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Splitting the dataset using train test split\n\nX_train, X_test ,y_train ,y_test = train_test_split (X, y, test_size = 0.2,random_state=42)  \n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:20.744070Z","iopub.execute_input":"2022-07-04T21:13:20.744490Z","iopub.status.idle":"2022-07-04T21:13:20.756846Z","shell.execute_reply.started":"2022-07-04T21:13:20.744457Z","shell.execute_reply":"2022-07-04T21:13:20.755583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Models building","metadata":{}},{"cell_type":"code","source":"# Selection models\nlgr_model = LogisticRegression()\nrfc_model = RandomForestClassifier()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:20.758324Z","iopub.execute_input":"2022-07-04T21:13:20.759102Z","iopub.status.idle":"2022-07-04T21:13:20.777596Z","shell.execute_reply.started":"2022-07-04T21:13:20.759065Z","shell.execute_reply":"2022-07-04T21:13:20.775754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training \nlgr_model.fit(X_train, y_train)\nrfc_model.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:20.779537Z","iopub.execute_input":"2022-07-04T21:13:20.780832Z","iopub.status.idle":"2022-07-04T21:13:21.096941Z","shell.execute_reply.started":"2022-07-04T21:13:20.780779Z","shell.execute_reply":"2022-07-04T21:13:21.095924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'RandomForestClassifier Training Score   =     {rfc_model.score(X_train, y_train)*100:.2f}%')\nprint(f\"RandomForestClassifier Test Score       =     {rfc_model.score(X_test, y_test)*100:.2f}%\")\nprint(f\"LogisticRegression Training Score       =     {lgr_model.score(X_train, y_train)*100:.2f}%\")\nprint(f\"LogisticRegression Training Score       =     {lgr_model.score(X_test, y_test)*100:.2f}%\")","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:21.102443Z","iopub.execute_input":"2022-07-04T21:13:21.103184Z","iopub.status.idle":"2022-07-04T21:13:21.168909Z","shell.execute_reply.started":"2022-07-04T21:13:21.103094Z","shell.execute_reply":"2022-07-04T21:13:21.167455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### the results show Random Forest Model provided best results","metadata":{}},{"cell_type":"code","source":"#prediction\ny_preds = rfc_model.predict(X_test)\ny_preds[:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:21.172964Z","iopub.execute_input":"2022-07-04T21:13:21.173424Z","iopub.status.idle":"2022-07-04T21:13:21.202085Z","shell.execute_reply.started":"2022-07-04T21:13:21.173389Z","shell.execute_reply":"2022-07-04T21:13:21.200786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Printing the accuracy score of Randomforestclassifier\ntest_data_accuracy = accuracy_score(y_preds, y_test)\nprint(f'RandomForestClassifier Accuracy score   =     {test_data_accuracy*100:.2f}%')","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:21.204036Z","iopub.execute_input":"2022-07-04T21:13:21.204536Z","iopub.status.idle":"2022-07-04T21:13:21.214442Z","shell.execute_reply.started":"2022-07-04T21:13:21.204491Z","shell.execute_reply":"2022-07-04T21:13:21.213317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Created a common function to plot confusion matrix\ndef Plot_confusion_matrix(y_test, pred_test):\n  cm = confusion_matrix(y_test, pred_test)\n  plt.clf()\n  plt.imshow(cm, interpolation='nearest', cmap=plt.cm.Accent)\n  x_lable = ['Predicted Not survived', 'Predicted Survival']\n  y_lable = ['Actual Not Survived', 'Actual Survived']  \n  plt.title('Confusion Matrix - Test Data')\n  plt.ylabel('True label')\n  plt.xlabel('Predicted label')\n  ticks = np.arange(len(x_lable))\n  plt.xticks(ticks, x_lable, rotation=45)\n  plt.yticks(ticks, y_lable)  \n  \n  for i in range(2):\n      for j in range(2):\n          plt.text(j,i,str(cm[i][j]),fontsize=12)\n  plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:21.216024Z","iopub.execute_input":"2022-07-04T21:13:21.217040Z","iopub.status.idle":"2022-07-04T21:13:21.227548Z","shell.execute_reply.started":"2022-07-04T21:13:21.217001Z","shell.execute_reply":"2022-07-04T21:13:21.225925Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Plot_confusion_matrix(y_test, y_preds)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:21.229341Z","iopub.execute_input":"2022-07-04T21:13:21.230559Z","iopub.status.idle":"2022-07-04T21:13:21.390266Z","shell.execute_reply.started":"2022-07-04T21:13:21.230503Z","shell.execute_reply":"2022-07-04T21:13:21.388765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Testing the model on the test dataset","metadata":{}},{"cell_type":"code","source":"#Loading the testing data\ndf_test = pd.read_csv(\"../input/titanic/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:46.354239Z","iopub.execute_input":"2022-07-04T21:13:46.354685Z","iopub.status.idle":"2022-07-04T21:13:46.369276Z","shell.execute_reply.started":"2022-07-04T21:13:46.354651Z","shell.execute_reply":"2022-07-04T21:13:46.368228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# printing the first 5 rows of the dataframe\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:56.016046Z","iopub.execute_input":"2022-07-04T21:13:56.016946Z","iopub.status.idle":"2022-07-04T21:13:56.038540Z","shell.execute_reply.started":"2022-07-04T21:13:56.016898Z","shell.execute_reply":"2022-07-04T21:13:56.037213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking the shape\ndf_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:57.262991Z","iopub.execute_input":"2022-07-04T21:13:57.263832Z","iopub.status.idle":"2022-07-04T21:13:57.271615Z","shell.execute_reply.started":"2022-07-04T21:13:57.263529Z","shell.execute_reply":"2022-07-04T21:13:57.270245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Checking the datatypes and null/non-null distribution\ndf_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:58.117545Z","iopub.execute_input":"2022-07-04T21:13:58.118157Z","iopub.status.idle":"2022-07-04T21:13:58.135685Z","shell.execute_reply.started":"2022-07-04T21:13:58.118062Z","shell.execute_reply":"2022-07-04T21:13:58.134559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check the number of missing values in each column\n(df_test.isnull().sum()/len(df_test))*100","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:58.956442Z","iopub.execute_input":"2022-07-04T21:13:58.957836Z","iopub.status.idle":"2022-07-04T21:13:58.972528Z","shell.execute_reply.started":"2022-07-04T21:13:58.957769Z","shell.execute_reply":"2022-07-04T21:13:58.970898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop the columns that has null values more the 70%\ndf_test.drop(\"Cabin\", axis =1, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:13:59.761963Z","iopub.execute_input":"2022-07-04T21:13:59.762992Z","iopub.status.idle":"2022-07-04T21:13:59.770550Z","shell.execute_reply.started":"2022-07-04T21:13:59.762925Z","shell.execute_reply":"2022-07-04T21:13:59.768896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replacing the missing values in \"Age\" column with mean value\ndf_test['Age'].fillna(df_test['Age'].mean(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:14:00.842389Z","iopub.execute_input":"2022-07-04T21:14:00.843000Z","iopub.status.idle":"2022-07-04T21:14:00.850315Z","shell.execute_reply.started":"2022-07-04T21:14:00.842963Z","shell.execute_reply":"2022-07-04T21:14:00.848612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replacing the missing values in \"Fare\" column with mean value\ndf_test['Fare'].fillna(df_test['Fare'].mean(), inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:14:01.389113Z","iopub.execute_input":"2022-07-04T21:14:01.390332Z","iopub.status.idle":"2022-07-04T21:14:01.398060Z","shell.execute_reply.started":"2022-07-04T21:14:01.390251Z","shell.execute_reply":"2022-07-04T21:14:01.396313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# replacing the missing values in \"Age\" column with S\ndf_test.Embarked.fillna(\"S\", inplace  = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:14:03.181974Z","iopub.execute_input":"2022-07-04T21:14:03.183486Z","iopub.status.idle":"2022-07-04T21:14:03.190162Z","shell.execute_reply.started":"2022-07-04T21:14:03.183431Z","shell.execute_reply":"2022-07-04T21:14:03.189287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# converting categorical Columns\ndf_test.replace({'Sex':{'male':0,'female':1}, 'Embarked':{'S':0,'C':1,'Q':2}}, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:14:11.356984Z","iopub.execute_input":"2022-07-04T21:14:11.357630Z","iopub.status.idle":"2022-07-04T21:14:11.370760Z","shell.execute_reply.started":"2022-07-04T21:14:11.357591Z","shell.execute_reply":"2022-07-04T21:14:11.369230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndf_test.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:14:12.505334Z","iopub.execute_input":"2022-07-04T21:14:12.506548Z","iopub.status.idle":"2022-07-04T21:14:12.518150Z","shell.execute_reply.started":"2022-07-04T21:14:12.506503Z","shell.execute_reply":"2022-07-04T21:14:12.516704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#drop unnecessary columns and save the others in X_t\nX_t = df_test.drop(columns= ['PassengerId','Name','Ticket'])","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:14:13.316420Z","iopub.execute_input":"2022-07-04T21:14:13.317838Z","iopub.status.idle":"2022-07-04T21:14:13.325979Z","shell.execute_reply.started":"2022-07-04T21:14:13.317779Z","shell.execute_reply":"2022-07-04T21:14:13.324282Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_t.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:14:14.435239Z","iopub.execute_input":"2022-07-04T21:14:14.435654Z","iopub.status.idle":"2022-07-04T21:14:14.457900Z","shell.execute_reply.started":"2022-07-04T21:14:14.435624Z","shell.execute_reply":"2022-07-04T21:14:14.455337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#prediction\npredicted = rfc_model.predict(X_t)\npred_df = pd.DataFrame(predicted, columns=['Survived'])\npred_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:14:15.520620Z","iopub.execute_input":"2022-07-04T21:14:15.521541Z","iopub.status.idle":"2022-07-04T21:14:15.561733Z","shell.execute_reply.started":"2022-07-04T21:14:15.521501Z","shell.execute_reply":"2022-07-04T21:14:15.559580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"passengerId = df_test[[\"PassengerId\"]]\npassengerId.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:14:16.658530Z","iopub.execute_input":"2022-07-04T21:14:16.659674Z","iopub.status.idle":"2022-07-04T21:14:16.671574Z","shell.execute_reply.started":"2022-07-04T21:14:16.659631Z","shell.execute_reply":"2022-07-04T21:14:16.670292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#merged the dataframes\nlast_data = pd.concat( [passengerId,pred_df],ignore_index = False, axis = 1)\nlast_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:14:17.576485Z","iopub.execute_input":"2022-07-04T21:14:17.577278Z","iopub.status.idle":"2022-07-04T21:14:17.590404Z","shell.execute_reply.started":"2022-07-04T21:14:17.577241Z","shell.execute_reply":"2022-07-04T21:14:17.589155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"last_data.to_csv('Titanic_prediction.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-04T21:14:18.571798Z","iopub.execute_input":"2022-07-04T21:14:18.572617Z","iopub.status.idle":"2022-07-04T21:14:18.583512Z","shell.execute_reply.started":"2022-07-04T21:14:18.572570Z","shell.execute_reply":"2022-07-04T21:14:18.582253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}