{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"> # **MACHINE LEARNING FROM DISASTER**","metadata":{}},{"cell_type":"markdown","source":"**Let's make survival prediction from the titanic dataset.**","metadata":{}},{"cell_type":"code","source":"#Importing required libraries\nimport pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:23.764387Z","iopub.execute_input":"2022-07-15T07:30:23.766281Z","iopub.status.idle":"2022-07-15T07:30:24.412796Z","shell.execute_reply.started":"2022-07-15T07:30:23.766165Z","shell.execute_reply":"2022-07-15T07:30:24.410854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reading datasets\n#train data and test data\ntrain=pd.read_csv('../input/titanic/train.csv')\ntest=pd.read_csv('../input/titanic/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:24.419759Z","iopub.execute_input":"2022-07-15T07:30:24.420636Z","iopub.status.idle":"2022-07-15T07:30:24.439718Z","shell.execute_reply.started":"2022-07-15T07:30:24.420557Z","shell.execute_reply":"2022-07-15T07:30:24.437759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Viewing data and different features\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:24.441409Z","iopub.execute_input":"2022-07-15T07:30:24.442429Z","iopub.status.idle":"2022-07-15T07:30:24.470350Z","shell.execute_reply.started":"2022-07-15T07:30:24.442372Z","shell.execute_reply":"2022-07-15T07:30:24.468955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape ","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:24.474080Z","iopub.execute_input":"2022-07-15T07:30:24.474520Z","iopub.status.idle":"2022-07-15T07:30:24.484531Z","shell.execute_reply.started":"2022-07-15T07:30:24.474485Z","shell.execute_reply":"2022-07-15T07:30:24.483040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.columns ","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:24.486616Z","iopub.execute_input":"2022-07-15T07:30:24.487836Z","iopub.status.idle":"2022-07-15T07:30:24.502223Z","shell.execute_reply.started":"2022-07-15T07:30:24.487774Z","shell.execute_reply":"2022-07-15T07:30:24.500967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['Sex'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:24.503768Z","iopub.execute_input":"2022-07-15T07:30:24.504272Z","iopub.status.idle":"2022-07-15T07:30:24.520702Z","shell.execute_reply.started":"2022-07-15T07:30:24.504235Z","shell.execute_reply":"2022-07-15T07:30:24.519355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Visualization","metadata":{}},{"cell_type":"code","source":"#Visualizing survivals based on gender\ntrain['Died'] = 1 - train['Survived']\ntrain.groupby('Sex').agg('sum')[['Survived', 'Died']].plot(kind='bar',\n                                                           figsize=(10, 5),\n                                                           stacked=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:24.522348Z","iopub.execute_input":"2022-07-15T07:30:24.523729Z","iopub.status.idle":"2022-07-15T07:30:24.798373Z","shell.execute_reply.started":"2022-07-15T07:30:24.523650Z","shell.execute_reply":"2022-07-15T07:30:24.796923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##Visualizing survivals based on fare\nfigure = plt.figure(figsize=(16, 7))\nplt.hist([train[train['Survived'] == 1]['Fare'], train[train['Survived'] == 0]['Fare']], \n         stacked=True, bins = 50, label = ['Survived','Dead'])\nplt.xlabel('Fare')\nplt.ylabel('Number of passengers')\nplt.legend()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:24.799759Z","iopub.execute_input":"2022-07-15T07:30:24.800446Z","iopub.status.idle":"2022-07-15T07:30:25.315553Z","shell.execute_reply.started":"2022-07-15T07:30:24.800402Z","shell.execute_reply":"2022-07-15T07:30:25.313829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Processing training data","metadata":{}},{"cell_type":"code","source":"#Cleaning the data by removing irrelevant columns\ndf1=train.drop(['Name','Ticket','Cabin','PassengerId','Died'], axis=1)\ndf1.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.318039Z","iopub.execute_input":"2022-07-15T07:30:25.318640Z","iopub.status.idle":"2022-07-15T07:30:25.340437Z","shell.execute_reply.started":"2022-07-15T07:30:25.318580Z","shell.execute_reply":"2022-07-15T07:30:25.338676Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.isnull().sum() ","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.341774Z","iopub.execute_input":"2022-07-15T07:30:25.343098Z","iopub.status.idle":"2022-07-15T07:30:25.360850Z","shell.execute_reply.started":"2022-07-15T07:30:25.343037Z","shell.execute_reply":"2022-07-15T07:30:25.359254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Converting the categorical features 'Sex' and 'Embarked' into numerical values 0 & 1\ndf1.Sex=df1.Sex.map({'female':0, 'male':1})\ndf1.Embarked=df1.Embarked.map({'S':0, 'C':1, 'Q':2,'nan':'NaN'})\ndf1.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.363024Z","iopub.execute_input":"2022-07-15T07:30:25.363515Z","iopub.status.idle":"2022-07-15T07:30:25.388869Z","shell.execute_reply.started":"2022-07-15T07:30:25.363479Z","shell.execute_reply":"2022-07-15T07:30:25.387724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Mean age of each sex\nmean_age_men=df1[df1['Sex']==1]['Age'].mean()\nmean_age_women=df1[df1['Sex']==0]['Age'].mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.390852Z","iopub.execute_input":"2022-07-15T07:30:25.391432Z","iopub.status.idle":"2022-07-15T07:30:25.406161Z","shell.execute_reply.started":"2022-07-15T07:30:25.391389Z","shell.execute_reply":"2022-07-15T07:30:25.404937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Filling all the null values in 'Age' with respective mean age\ndf1.loc[(df1.Age.isnull()) & (df1['Sex']==0),'Age']=mean_age_women\ndf1.loc[(df1.Age.isnull()) & (df1['Sex']==1),'Age']=mean_age_men","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.411871Z","iopub.execute_input":"2022-07-15T07:30:25.412751Z","iopub.status.idle":"2022-07-15T07:30:25.429617Z","shell.execute_reply.started":"2022-07-15T07:30:25.412698Z","shell.execute_reply":"2022-07-15T07:30:25.427718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Let's check for the null values again now\ndf1.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.431814Z","iopub.execute_input":"2022-07-15T07:30:25.432212Z","iopub.status.idle":"2022-07-15T07:30:25.457719Z","shell.execute_reply.started":"2022-07-15T07:30:25.432180Z","shell.execute_reply":"2022-07-15T07:30:25.455830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Since there exist 2 null values in the Embarked column, let's drop those rows containing null values\ndf1.dropna(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.459511Z","iopub.execute_input":"2022-07-15T07:30:25.460090Z","iopub.status.idle":"2022-07-15T07:30:25.471014Z","shell.execute_reply.started":"2022-07-15T07:30:25.460034Z","shell.execute_reply":"2022-07-15T07:30:25.469185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df1.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.472766Z","iopub.execute_input":"2022-07-15T07:30:25.473231Z","iopub.status.idle":"2022-07-15T07:30:25.490945Z","shell.execute_reply.started":"2022-07-15T07:30:25.473195Z","shell.execute_reply":"2022-07-15T07:30:25.489912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Doing Feature Scaling to standardize the independent features present in the data in a fixed range\ndf1.Age = (df1.Age-min(df1.Age))/(max(df1.Age)-min(df1.Age))\ndf1.Fare = (df1.Fare-min(df1.Fare))/(max(df1.Fare)-min(df1.Fare))\ndf1.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.492981Z","iopub.execute_input":"2022-07-15T07:30:25.493688Z","iopub.status.idle":"2022-07-15T07:30:25.543073Z","shell.execute_reply.started":"2022-07-15T07:30:25.493648Z","shell.execute_reply":"2022-07-15T07:30:25.541257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Creating model","metadata":{}},{"cell_type":"code","source":"#Splitting the data for training and testing\nfrom sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(\n    df1.drop(['Survived'], axis=1),\n    df1.Survived,\n    test_size= 0.2,\n    random_state=0,\n    stratify=df1.Survived)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.545383Z","iopub.execute_input":"2022-07-15T07:30:25.545847Z","iopub.status.idle":"2022-07-15T07:30:25.627547Z","shell.execute_reply.started":"2022-07-15T07:30:25.545811Z","shell.execute_reply":"2022-07-15T07:30:25.625856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Logistic Regression**","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression\nlrmod = LogisticRegression()\nlrmod.fit(X_train, y_train)\nfrom sklearn.metrics import accuracy_score\ny_predict = lrmod.predict(X_test)\naccuracy_score(y_test, y_predict)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.629660Z","iopub.execute_input":"2022-07-15T07:30:25.630418Z","iopub.status.idle":"2022-07-15T07:30:25.680244Z","shell.execute_reply.started":"2022-07-15T07:30:25.630367Z","shell.execute_reply":"2022-07-15T07:30:25.678889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Confusion Matrix\nfrom sklearn.metrics import confusion_matrix\ncma=confusion_matrix(y_test, y_predict)\nsns.heatmap(cma,annot=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.682243Z","iopub.execute_input":"2022-07-15T07:30:25.683116Z","iopub.status.idle":"2022-07-15T07:30:25.911023Z","shell.execute_reply.started":"2022-07-15T07:30:25.683063Z","shell.execute_reply":"2022-07-15T07:30:25.909477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Processing test data","metadata":{}},{"cell_type":"code","source":"#Viewing test data\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.913921Z","iopub.execute_input":"2022-07-15T07:30:25.915614Z","iopub.status.idle":"2022-07-15T07:30:25.936975Z","shell.execute_reply.started":"2022-07-15T07:30:25.915537Z","shell.execute_reply":"2022-07-15T07:30:25.934677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Cleaning the data by removing irrelevant columns\ndf2=test.drop(['PassengerId','Name','Ticket','Cabin'], axis=1)\ndf2","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.939025Z","iopub.execute_input":"2022-07-15T07:30:25.939547Z","iopub.status.idle":"2022-07-15T07:30:25.967738Z","shell.execute_reply.started":"2022-07-15T07:30:25.939499Z","shell.execute_reply":"2022-07-15T07:30:25.966668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Converting the categorical features 'Sex' and 'Embarked' into numerical values 0 & 1\ndf2.Sex=df2.Sex.map({'female':0, 'male':1})\ndf2.Embarked=df2.Embarked.map({'S':0, 'C':1, 'Q':2,'nan':'NaN'})\ndf2.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.968978Z","iopub.execute_input":"2022-07-15T07:30:25.970058Z","iopub.status.idle":"2022-07-15T07:30:25.994166Z","shell.execute_reply.started":"2022-07-15T07:30:25.970013Z","shell.execute_reply":"2022-07-15T07:30:25.992508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Let's check for the null values\ndf2.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:25.996239Z","iopub.execute_input":"2022-07-15T07:30:25.997591Z","iopub.status.idle":"2022-07-15T07:30:26.015471Z","shell.execute_reply.started":"2022-07-15T07:30:25.997510Z","shell.execute_reply":"2022-07-15T07:30:26.014186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Finding mean age\nmean_age_men2=df2[df2['Sex']==1]['Age'].mean()\nmean_age_women2=df2[df2['Sex']==0]['Age'].mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:26.016724Z","iopub.execute_input":"2022-07-15T07:30:26.017199Z","iopub.status.idle":"2022-07-15T07:30:26.033610Z","shell.execute_reply.started":"2022-07-15T07:30:26.017153Z","shell.execute_reply":"2022-07-15T07:30:26.032088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Filling all the null values in 'Age' and 'Fare' with respective mean age and mean fare\ndf2.loc[(df2.Age.isnull()) & (df2['Sex']==0),'Age']=mean_age_women2\ndf2.loc[(df2.Age.isnull()) & (df2['Sex']==1),'Age']=mean_age_men2\ndf2['Fare']=df2['Fare'].fillna(df2['Fare'].mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:26.036016Z","iopub.execute_input":"2022-07-15T07:30:26.036673Z","iopub.status.idle":"2022-07-15T07:30:26.056884Z","shell.execute_reply.started":"2022-07-15T07:30:26.036616Z","shell.execute_reply":"2022-07-15T07:30:26.054985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df2.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:26.058635Z","iopub.execute_input":"2022-07-15T07:30:26.059814Z","iopub.status.idle":"2022-07-15T07:30:26.080893Z","shell.execute_reply.started":"2022-07-15T07:30:26.059768Z","shell.execute_reply":"2022-07-15T07:30:26.079765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Doing Feature Scaling to standardize the independent features present in the data in a fixed range\ndf2.Age = (df2.Age-min(df2.Age))/(max(df2.Age)-min(df2.Age))\ndf2.Fare = (df2.Fare-min(df2.Fare))/(max(df2.Fare)-min(df2.Fare))\ndf2.describe()","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:26.083057Z","iopub.execute_input":"2022-07-15T07:30:26.084751Z","iopub.status.idle":"2022-07-15T07:30:26.125951Z","shell.execute_reply.started":"2022-07-15T07:30:26.084696Z","shell.execute_reply":"2022-07-15T07:30:26.124915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Prediction**","metadata":{}},{"cell_type":"code","source":"prediction = lrmod.predict(df2)\nprediction","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:26.127608Z","iopub.execute_input":"2022-07-15T07:30:26.128475Z","iopub.status.idle":"2022-07-15T07:30:26.144489Z","shell.execute_reply.started":"2022-07-15T07:30:26.128429Z","shell.execute_reply":"2022-07-15T07:30:26.141883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\"PassengerId\": test[\"PassengerId\"],\n                            \"Survived\": prediction})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:26.147128Z","iopub.execute_input":"2022-07-15T07:30:26.148160Z","iopub.status.idle":"2022-07-15T07:30:26.157774Z","shell.execute_reply.started":"2022-07-15T07:30:26.148112Z","shell.execute_reply":"2022-07-15T07:30:26.156059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_df = pd.read_csv('submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:26.159470Z","iopub.execute_input":"2022-07-15T07:30:26.160846Z","iopub.status.idle":"2022-07-15T07:30:26.177890Z","shell.execute_reply.started":"2022-07-15T07:30:26.160799Z","shell.execute_reply":"2022-07-15T07:30:26.175337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualizing predicted values\nsns.countplot(x='Survived', data=prediction_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-15T07:30:26.180363Z","iopub.execute_input":"2022-07-15T07:30:26.181238Z","iopub.status.idle":"2022-07-15T07:30:26.381717Z","shell.execute_reply.started":"2022-07-15T07:30:26.181138Z","shell.execute_reply":"2022-07-15T07:30:26.380319Z"},"trusted":true},"execution_count":null,"outputs":[]}]}