{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## My First Competition Submission. ","metadata":{}},{"cell_type":"markdown","source":"![](https://images.unsplash.com/photo-1612429085511-2c2ac3aac171?ixlib=rb-1.2.1&ixid=MnwxMjA3fDB8MHxwaG90by1wYWdlfHx8fGVufDB8fHx8&auto=format&fit=crop&w=390&q=80)","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"markdown","source":"### Import Librabries","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt #data visualization\nimport seaborn as sns #data visualization\nimport plotly.express as px #data visualuzation\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T19:22:28.924035Z","iopub.execute_input":"2022-08-11T19:22:28.924352Z","iopub.status.idle":"2022-08-11T19:22:28.932369Z","shell.execute_reply.started":"2022-08-11T19:22:28.924329Z","shell.execute_reply":"2022-08-11T19:22:28.931277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check the data\ntrain_data = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:29.084317Z","iopub.execute_input":"2022-08-11T19:22:29.084793Z","iopub.status.idle":"2022-08-11T19:22:29.105625Z","shell.execute_reply.started":"2022-08-11T19:22:29.084768Z","shell.execute_reply":"2022-08-11T19:22:29.104495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Get a quick summary of the data\ntrain_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:29.226053Z","iopub.execute_input":"2022-08-11T19:22:29.226391Z","iopub.status.idle":"2022-08-11T19:22:29.258410Z","shell.execute_reply.started":"2022-08-11T19:22:29.226366Z","shell.execute_reply":"2022-08-11T19:22:29.257306Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets visualize the number of the people who survived across different age and sex\nfig = px.violin(train_data, x=\"Sex\", y=\"Age\", color=\"Survived\",points=\"all\")\n \n# Show the plot\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:29.318606Z","iopub.execute_input":"2022-08-11T19:22:29.318922Z","iopub.status.idle":"2022-08-11T19:22:29.378897Z","shell.execute_reply.started":"2022-08-11T19:22:29.318898Z","shell.execute_reply":"2022-08-11T19:22:29.377739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets visualize the number of the people across classewho survived\nfig = px.histogram(train_data, x=\"Survived\", color=\"Pclass\", barmode=\"group\")\n \n# Show the plot\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:29.406258Z","iopub.execute_input":"2022-08-11T19:22:29.407048Z","iopub.status.idle":"2022-08-11T19:22:29.464840Z","shell.execute_reply.started":"2022-08-11T19:22:29.407018Z","shell.execute_reply":"2022-08-11T19:22:29.463712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Pie digram to check the overall survival rate \nfig = px.pie(train_data, names='Survived', title='Passenger Survival',hole=0.5)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:29.592607Z","iopub.execute_input":"2022-08-11T19:22:29.593442Z","iopub.status.idle":"2022-08-11T19:22:29.632592Z","shell.execute_reply.started":"2022-08-11T19:22:29.593415Z","shell.execute_reply":"2022-08-11T19:22:29.631652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Scatter Plot For Survival Rate Visualization\nfig = px.scatter(train_data, x='Fare', y='Age', color='Survived',size='Fare')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:29.754617Z","iopub.execute_input":"2022-08-11T19:22:29.755917Z","iopub.status.idle":"2022-08-11T19:22:29.811073Z","shell.execute_reply.started":"2022-08-11T19:22:29.755876Z","shell.execute_reply":"2022-08-11T19:22:29.809864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Load and check the test data\ntest_data = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:29.840193Z","iopub.execute_input":"2022-08-11T19:22:29.840529Z","iopub.status.idle":"2022-08-11T19:22:29.859696Z","shell.execute_reply.started":"2022-08-11T19:22:29.840505Z","shell.execute_reply":"2022-08-11T19:22:29.858719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check the description of the test data table\ntest_data.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:29.928258Z","iopub.execute_input":"2022-08-11T19:22:29.928928Z","iopub.status.idle":"2022-08-11T19:22:29.951150Z","shell.execute_reply.started":"2022-08-11T19:22:29.928903Z","shell.execute_reply":"2022-08-11T19:22:29.950510Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Visualize the Port of Embarkation\nfig=px.pie(test_data,names='Embarked',hole=0.6)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:30.013621Z","iopub.execute_input":"2022-08-11T19:22:30.014082Z","iopub.status.idle":"2022-08-11T19:22:30.052929Z","shell.execute_reply.started":"2022-08-11T19:22:30.014059Z","shell.execute_reply":"2022-08-11T19:22:30.052344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Lets visualize the number of the people across Embarkaktion who survived\nfig = px.histogram(train_data, x=\"Survived\", color=\"Embarked\", barmode=\"group\")\n \n# Show the plot\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:30.098941Z","iopub.execute_input":"2022-08-11T19:22:30.099330Z","iopub.status.idle":"2022-08-11T19:22:30.160571Z","shell.execute_reply.started":"2022-08-11T19:22:30.099306Z","shell.execute_reply":"2022-08-11T19:22:30.159721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#3d Scatter Plot to Compare Survival across Sibling, Parents and Fare price for Titanic\nfig = px.scatter_3d(train_data, x='Parch', y='SibSp', color='Survived',z='Fare')\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:30.184433Z","iopub.execute_input":"2022-08-11T19:22:30.185552Z","iopub.status.idle":"2022-08-11T19:22:30.236775Z","shell.execute_reply.started":"2022-08-11T19:22:30.185512Z","shell.execute_reply":"2022-08-11T19:22:30.235980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Machine Learning","metadata":{}},{"cell_type":"code","source":"women = train_data.loc[train_data.Sex == 'female'][\"Survived\"]\nrate_women = sum(women)/len(women)\n\nprint(\"% of women who survived:\", rate_women)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:30.267129Z","iopub.execute_input":"2022-08-11T19:22:30.267771Z","iopub.status.idle":"2022-08-11T19:22:30.275270Z","shell.execute_reply.started":"2022-08-11T19:22:30.267745Z","shell.execute_reply":"2022-08-11T19:22:30.274471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"men = train_data.loc[train_data.Sex == 'male'][\"Survived\"]\nrate_men = sum(men)/len(men)\n\nprint(\"% of men who survived:\", rate_men)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:30.349563Z","iopub.execute_input":"2022-08-11T19:22:30.350336Z","iopub.status.idle":"2022-08-11T19:22:30.356051Z","shell.execute_reply.started":"2022-08-11T19:22:30.350307Z","shell.execute_reply":"2022-08-11T19:22:30.355371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We will perform the comparisons between Logistic and Random Forrest Algorithm to Identify Which Algorithm is better here to solve the problem","metadata":{}},{"cell_type":"markdown","source":"#### 1. Implementation of Random Forrest Algorithm","metadata":{}},{"cell_type":"markdown","source":"##### Import necessary libraries","metadata":{}},{"cell_type":"code","source":"#Regressions Methods\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.linear_model import LogisticRegression\n\n#To test model accuracy\nfrom sklearn.metrics import accuracy_score,precision_score,recall_score\n\n#To split the data\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:30.418656Z","iopub.execute_input":"2022-08-11T19:22:30.419200Z","iopub.status.idle":"2022-08-11T19:22:30.423636Z","shell.execute_reply.started":"2022-08-11T19:22:30.419166Z","shell.execute_reply":"2022-08-11T19:22:30.422929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Model Implementation","metadata":{}},{"cell_type":"code","source":"features = [\"Pclass\", \"Sex\", \"SibSp\", \"Parch\"] #Factors for prediction\nX = pd.get_dummies(train_data[features]) #Convert Categorical Data to Dummy Data\ny = train_data[\"Survived\"] #Prediction Data\n\n#Data Split using test_train_split fuction\nxtrain,xtest,ytrain,ytest=train_test_split(X,y,test_size=0.8)\n\nmodel = RandomForestClassifier(n_estimators=500, max_depth=15, random_state=0)\nmodel.fit(xtrain, ytrain)\npredictions = model.predict(xtest)\n\n#Print the prediction data\npredictions","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:30.473762Z","iopub.execute_input":"2022-08-11T19:22:30.474691Z","iopub.status.idle":"2022-08-11T19:22:31.112820Z","shell.execute_reply.started":"2022-08-11T19:22:30.474656Z","shell.execute_reply":"2022-08-11T19:22:31.111797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Check accuracy\ncomparation=pd.DataFrame({\"Actual Value\":ytest,\"Predicted Value\":predictions})\ncomparation.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:31.115524Z","iopub.execute_input":"2022-08-11T19:22:31.115982Z","iopub.status.idle":"2022-08-11T19:22:31.126854Z","shell.execute_reply.started":"2022-08-11T19:22:31.115936Z","shell.execute_reply":"2022-08-11T19:22:31.125805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Here we will note down the performance of Random Forest Algorithm","metadata":{}},{"cell_type":"code","source":"av=comparation[\"Actual Value\"]\npv=comparation[\"Predicted Value\"]\nacc=accuracy_score(av,pv)\npre=precision_score(av,pv)\nrec=recall_score(av,pv)\nprint(\"Accuracy score: \",acc)\nprint(\"Precision score: \",pre)\nprint(\"Recall score: \",rec)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:31.128065Z","iopub.execute_input":"2022-08-11T19:22:31.128920Z","iopub.status.idle":"2022-08-11T19:22:31.142534Z","shell.execute_reply.started":"2022-08-11T19:22:31.128884Z","shell.execute_reply":"2022-08-11T19:22:31.141298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### 2. Implementation of Logistic Regression Algorithm","metadata":{}},{"cell_type":"code","source":"l_features = [\"Pclass\", \"Sex\", \"SibSp\", \"Parch\"] #Factors for prediction\nX_l = pd.get_dummies(train_data[features]) #Convert Categorical Data to Dummy Data\ny_l = train_data[\"Survived\"] #Prediction Data\n\n#Data Split using test_train_split fuction\nx_l_train,x_l_test,y_l_train,y_l_test=train_test_split(X_l,y_l,test_size=0.8)\n\nl_model = LogisticRegression()\nl_model.fit(x_l_train, y_l_train)\nl_predictions = l_model.predict(x_l_test)\nl_predictions","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:31.144389Z","iopub.execute_input":"2022-08-11T19:22:31.145080Z","iopub.status.idle":"2022-08-11T19:22:31.166514Z","shell.execute_reply.started":"2022-08-11T19:22:31.145053Z","shell.execute_reply":"2022-08-11T19:22:31.165737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"comparation=pd.DataFrame({\"Actual Value\":ytest,\"Predicted Value\":l_predictions})\ncomparation.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:31.167535Z","iopub.execute_input":"2022-08-11T19:22:31.167909Z","iopub.status.idle":"2022-08-11T19:22:31.176119Z","shell.execute_reply.started":"2022-08-11T19:22:31.167886Z","shell.execute_reply":"2022-08-11T19:22:31.175484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Here we will note down the performance of Logistic Regression Algorithm","metadata":{}},{"cell_type":"code","source":"av=comparation[\"Actual Value\"]\npv=comparation[\"Predicted Value\"]\nacc=accuracy_score(av,pv)\npre=precision_score(av,pv)\nrec=recall_score(av,pv)\nprint(\"Accuracy score: \",acc)\nprint(\"Precision score: \",pre)\nprint(\"Recall score: \",rec)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T19:22:31.177157Z","iopub.execute_input":"2022-08-11T19:22:31.177431Z","iopub.status.idle":"2022-08-11T19:22:31.198265Z","shell.execute_reply.started":"2022-08-11T19:22:31.177409Z","shell.execute_reply":"2022-08-11T19:22:31.197181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### From the above results it is clear that Random Forest Algorithm is better suited to predict the outcome for the Titanic Problem.","metadata":{}}]}