{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib import style\nimport seaborn as sns\n\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report, plot_confusion_matrix, precision_score, recall_score\n\nfrom sklearn.preprocessing import StandardScaler\n\nfrom sklearn.linear_model import LogisticRegression\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T10:02:55.771120Z","iopub.execute_input":"2022-08-11T10:02:55.771607Z","iopub.status.idle":"2022-08-11T10:02:55.782943Z","shell.execute_reply.started":"2022-08-11T10:02:55.771564Z","shell.execute_reply":"2022-08-11T10:02:55.781870Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#importing input data\ntrain_data = pd.read_csv('../input/titanic/train.csv')\n\ntest_data = pd.read_csv('../input/titanic/test.csv') \n\nval_data = pd.read_csv('../input/titanic/gender_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:55.785076Z","iopub.execute_input":"2022-08-11T10:02:55.785745Z","iopub.status.idle":"2022-08-11T10:02:55.809733Z","shell.execute_reply.started":"2022-08-11T10:02:55.785704Z","shell.execute_reply":"2022-08-11T10:02:55.808328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let us see the train Data","metadata":{}},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:55.811490Z","iopub.execute_input":"2022-08-11T10:02:55.812180Z","iopub.status.idle":"2022-08-11T10:02:55.840194Z","shell.execute_reply.started":"2022-08-11T10:02:55.812138Z","shell.execute_reply":"2022-08-11T10:02:55.838558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lets see the Test Data","metadata":{}},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:55.842307Z","iopub.execute_input":"2022-08-11T10:02:55.843157Z","iopub.status.idle":"2022-08-11T10:02:55.870994Z","shell.execute_reply.started":"2022-08-11T10:02:55.843104Z","shell.execute_reply":"2022-08-11T10:02:55.869696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**As the test data does not have any column survive we need to add it in the data set from the kaggle inputs**","metadata":{}},{"cell_type":"code","source":"test_data['Survived'] = val_data['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:55.874348Z","iopub.execute_input":"2022-08-11T10:02:55.875030Z","iopub.status.idle":"2022-08-11T10:02:55.883601Z","shell.execute_reply.started":"2022-08-11T10:02:55.874988Z","shell.execute_reply":"2022-08-11T10:02:55.882576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:55.885123Z","iopub.execute_input":"2022-08-11T10:02:55.886389Z","iopub.status.idle":"2022-08-11T10:02:55.926375Z","shell.execute_reply.started":"2022-08-11T10:02:55.886335Z","shell.execute_reply":"2022-08-11T10:02:55.924966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preparation","metadata":{}},{"cell_type":"markdown","source":"**Both the train data and test data have some null values or missing values, we need to deal with them before working with the model**","metadata":{}},{"cell_type":"code","source":"#DATA PREP\ntrain_data.isnull().sum()#finding the amount of missing values ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:55.928045Z","iopub.execute_input":"2022-08-11T10:02:55.928401Z","iopub.status.idle":"2022-08-11T10:02:55.942364Z","shell.execute_reply.started":"2022-08-11T10:02:55.928369Z","shell.execute_reply":"2022-08-11T10:02:55.941011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"100*(train_data.isnull().sum()/len(train_data)) #finding the percentage of missing values","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:55.944366Z","iopub.execute_input":"2022-08-11T10:02:55.945081Z","iopub.status.idle":"2022-08-11T10:02:55.957406Z","shell.execute_reply.started":"2022-08-11T10:02:55.945039Z","shell.execute_reply":"2022-08-11T10:02:55.955837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Lets make a function to find the amount of missing values percentage for each column to make our work easier**","metadata":{}},{"cell_type":"code","source":"def null_percent_func(df):\n    null_percent= 100*(df.isnull().sum()/len(df))\n    null_percent= null_percent[null_percent>0].sort_values()\n    return null_percent","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:55.959248Z","iopub.execute_input":"2022-08-11T10:02:55.960388Z","iopub.status.idle":"2022-08-11T10:02:55.967029Z","shell.execute_reply.started":"2022-08-11T10:02:55.960343Z","shell.execute_reply":"2022-08-11T10:02:55.965570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_percent= null_percent_func(train_data) #taking only the missing colums","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:55.972855Z","iopub.execute_input":"2022-08-11T10:02:55.973999Z","iopub.status.idle":"2022-08-11T10:02:55.984400Z","shell.execute_reply.started":"2022-08-11T10:02:55.973939Z","shell.execute_reply":"2022-08-11T10:02:55.983009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(null_percent)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:55.986026Z","iopub.execute_input":"2022-08-11T10:02:55.986950Z","iopub.status.idle":"2022-08-11T10:02:55.994206Z","shell.execute_reply.started":"2022-08-11T10:02:55.986886Z","shell.execute_reply":"2022-08-11T10:02:55.993252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data= train_data.dropna(axis=0, subset=['Embarked']) #dropped the whole row as the percent of its low, so dropping them ","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:55.995750Z","iopub.execute_input":"2022-08-11T10:02:55.996391Z","iopub.status.idle":"2022-08-11T10:02:56.010069Z","shell.execute_reply.started":"2022-08-11T10:02:55.996344Z","shell.execute_reply":"2022-08-11T10:02:56.008944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(train_data[\"Embarked\"])\ntrain_data.iloc[60]","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.011822Z","iopub.execute_input":"2022-08-11T10:02:56.012511Z","iopub.status.idle":"2022-08-11T10:02:56.025601Z","shell.execute_reply.started":"2022-08-11T10:02:56.012449Z","shell.execute_reply":"2022-08-11T10:02:56.023903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_percent = null_percent_func(train_data) #taking only the missing colums","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.027352Z","iopub.execute_input":"2022-08-11T10:02:56.027754Z","iopub.status.idle":"2022-08-11T10:02:56.037035Z","shell.execute_reply.started":"2022-08-11T10:02:56.027716Z","shell.execute_reply":"2022-08-11T10:02:56.035400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(null_percent)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.038961Z","iopub.execute_input":"2022-08-11T10:02:56.039353Z","iopub.status.idle":"2022-08-11T10:02:56.047319Z","shell.execute_reply.started":"2022-08-11T10:02:56.039313Z","shell.execute_reply":"2022-08-11T10:02:56.046411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You can see that one of the null percentage from the output of the function is gone, it means we are progressing!","metadata":{}},{"cell_type":"markdown","source":"**As we can see Age column have 19% of missing data, so we can not delete the whole column, for this kind of case we can take the mean and take the mean values in the null places**","metadata":{}},{"cell_type":"code","source":"train_data['Age'] = train_data['Age'].fillna(train_data['Age'].median()) #taking the median values of age columns","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.048803Z","iopub.execute_input":"2022-08-11T10:02:56.049393Z","iopub.status.idle":"2022-08-11T10:02:56.060132Z","shell.execute_reply.started":"2022-08-11T10:02:56.049350Z","shell.execute_reply":"2022-08-11T10:02:56.059117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_percent = null_percent_func(train_data) #taking only the missing colums\nprint(null_percent)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.061450Z","iopub.execute_input":"2022-08-11T10:02:56.062008Z","iopub.status.idle":"2022-08-11T10:02:56.077391Z","shell.execute_reply.started":"2022-08-11T10:02:56.061973Z","shell.execute_reply":"2022-08-11T10:02:56.076059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Only one obstacle is left!","metadata":{}},{"cell_type":"code","source":"train_data=train_data.drop(['Cabin'], axis=1) #too many missing values, so it will be good if we drop the whole column","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.079205Z","iopub.execute_input":"2022-08-11T10:02:56.080068Z","iopub.status.idle":"2022-08-11T10:02:56.090128Z","shell.execute_reply.started":"2022-08-11T10:02:56.080030Z","shell.execute_reply":"2022-08-11T10:02:56.089057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.drop(['Name','Ticket'], axis = 1, inplace=True) #As Name and ticket are useless","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.091519Z","iopub.execute_input":"2022-08-11T10:02:56.092089Z","iopub.status.idle":"2022-08-11T10:02:56.103284Z","shell.execute_reply.started":"2022-08-11T10:02:56.092052Z","shell.execute_reply":"2022-08-11T10:02:56.101943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Yes we are done with our train data part, now we need to do the same things for the test data part**","metadata":{}},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.104802Z","iopub.execute_input":"2022-08-11T10:02:56.105646Z","iopub.status.idle":"2022-08-11T10:02:56.133704Z","shell.execute_reply.started":"2022-08-11T10:02:56.105569Z","shell.execute_reply":"2022-08-11T10:02:56.132453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Test pre-processing part\n100*(test_data.isnull().sum()/len(test_data))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.135276Z","iopub.execute_input":"2022-08-11T10:02:56.135755Z","iopub.status.idle":"2022-08-11T10:02:56.147399Z","shell.execute_reply.started":"2022-08-11T10:02:56.135716Z","shell.execute_reply":"2022-08-11T10:02:56.146206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here embarked is okay but Fare got some null values","metadata":{}},{"cell_type":"code","source":"null_percent = null_percent_func(test_data) #taking only the missing colums\nprint(null_percent)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.148611Z","iopub.execute_input":"2022-08-11T10:02:56.149684Z","iopub.status.idle":"2022-08-11T10:02:56.164237Z","shell.execute_reply.started":"2022-08-11T10:02:56.149644Z","shell.execute_reply":"2022-08-11T10:02:56.163104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['Fare'] = test_data['Fare'].fillna(test_data['Fare'].median()) #as for the test we can not drop any values","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.169935Z","iopub.execute_input":"2022-08-11T10:02:56.171166Z","iopub.status.idle":"2022-08-11T10:02:56.177433Z","shell.execute_reply.started":"2022-08-11T10:02:56.171118Z","shell.execute_reply":"2022-08-11T10:02:56.176392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_percent = null_percent_func(test_data) #taking only the missing colums\nprint(null_percent)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.178858Z","iopub.execute_input":"2022-08-11T10:02:56.179444Z","iopub.status.idle":"2022-08-11T10:02:56.192035Z","shell.execute_reply.started":"2022-08-11T10:02:56.179409Z","shell.execute_reply":"2022-08-11T10:02:56.191039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['Age'] =test_data['Age'].fillna(test_data['Age'].median())","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.193342Z","iopub.execute_input":"2022-08-11T10:02:56.193733Z","iopub.status.idle":"2022-08-11T10:02:56.201079Z","shell.execute_reply.started":"2022-08-11T10:02:56.193700Z","shell.execute_reply":"2022-08-11T10:02:56.199752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_percent = null_percent_func(test_data) #taking only the missing colums\nprint(null_percent)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.202542Z","iopub.execute_input":"2022-08-11T10:02:56.203119Z","iopub.status.idle":"2022-08-11T10:02:56.216843Z","shell.execute_reply.started":"2022-08-11T10:02:56.203077Z","shell.execute_reply":"2022-08-11T10:02:56.215604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data=test_data.drop(['Cabin'], axis=1) # as for the train we have done it, and we want to do the same for the test","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.219050Z","iopub.execute_input":"2022-08-11T10:02:56.219999Z","iopub.status.idle":"2022-08-11T10:02:56.232057Z","shell.execute_reply.started":"2022-08-11T10:02:56.219950Z","shell.execute_reply":"2022-08-11T10:02:56.230854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.drop(['Name','Ticket'], axis = 1, inplace=True) #As Name and ticket are useless\ntest_data","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.233372Z","iopub.execute_input":"2022-08-11T10:02:56.234238Z","iopub.status.idle":"2022-08-11T10:02:56.263231Z","shell.execute_reply.started":"2022-08-11T10:02:56.234198Z","shell.execute_reply":"2022-08-11T10:02:56.261852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dummy variables","metadata":{}},{"cell_type":"markdown","source":"**As computer is not inteligent like us, we need to make the data more simpler for computer to understand.**","metadata":{}},{"cell_type":"markdown","source":"Here unique() function helps us to ditect the amount of unique values in a column, our plan is to target the columns which have 2 3 unique values only!","metadata":{}},{"cell_type":"code","source":"#we need to make this good for regression\n\n#So, gotta work with dummy variables\n#train_data[\"Pclass\"].unique()\ntrain_data[\"Embarked\"].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.265041Z","iopub.execute_input":"2022-08-11T10:02:56.266315Z","iopub.status.idle":"2022-08-11T10:02:56.274333Z","shell.execute_reply.started":"2022-08-11T10:02:56.266266Z","shell.execute_reply":"2022-08-11T10:02:56.273149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.276041Z","iopub.execute_input":"2022-08-11T10:02:56.276918Z","iopub.status.idle":"2022-08-11T10:02:56.292277Z","shell.execute_reply.started":"2022-08-11T10:02:56.276877Z","shell.execute_reply":"2022-08-11T10:02:56.291372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['Survived'] = train_data['Survived'].apply(str)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.293767Z","iopub.execute_input":"2022-08-11T10:02:56.294141Z","iopub.status.idle":"2022-08-11T10:02:56.307301Z","shell.execute_reply.started":"2022-08-11T10:02:56.294108Z","shell.execute_reply":"2022-08-11T10:02:56.305959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.309355Z","iopub.execute_input":"2022-08-11T10:02:56.310143Z","iopub.status.idle":"2022-08-11T10:02:56.327566Z","shell.execute_reply.started":"2022-08-11T10:02:56.310093Z","shell.execute_reply":"2022-08-11T10:02:56.326265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['Pclass'] = train_data['Pclass'].apply(str)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.329401Z","iopub.execute_input":"2022-08-11T10:02:56.330173Z","iopub.status.idle":"2022-08-11T10:02:56.337809Z","shell.execute_reply.started":"2022-08-11T10:02:56.330124Z","shell.execute_reply":"2022-08-11T10:02:56.336526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.339863Z","iopub.execute_input":"2022-08-11T10:02:56.340671Z","iopub.status.idle":"2022-08-11T10:02:56.365076Z","shell.execute_reply.started":"2022-08-11T10:02:56.340617Z","shell.execute_reply":"2022-08-11T10:02:56.364104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now we will start working with the targeted columns, get_dummies() function will create 3 columns if a columns have 3 unique object. This is why we need to drop one of the columns from the three columns**","metadata":{}},{"cell_type":"code","source":"final_train = pd.get_dummies(train_data,columns=[\"Sex\"])\nfinal_train.drop('Sex_female', axis = 1, inplace=True)\nfinal_train\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.366536Z","iopub.execute_input":"2022-08-11T10:02:56.367607Z","iopub.status.idle":"2022-08-11T10:02:56.401128Z","shell.execute_reply.started":"2022-08-11T10:02:56.367554Z","shell.execute_reply":"2022-08-11T10:02:56.399844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_train = pd.get_dummies(final_train,columns=[\"Pclass\"])\nfinal_train.drop('Pclass_1', axis = 1, inplace=True)\nfinal_train","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.403294Z","iopub.execute_input":"2022-08-11T10:02:56.403784Z","iopub.status.idle":"2022-08-11T10:02:56.435503Z","shell.execute_reply.started":"2022-08-11T10:02:56.403739Z","shell.execute_reply":"2022-08-11T10:02:56.434210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_train = pd.get_dummies(final_train,columns=[\"Embarked\"])\nfinal_train.drop('Embarked_C', axis = 1, inplace=True)\nfinal_train","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.437391Z","iopub.execute_input":"2022-08-11T10:02:56.438210Z","iopub.status.idle":"2022-08-11T10:02:56.467834Z","shell.execute_reply.started":"2022-08-11T10:02:56.438156Z","shell.execute_reply":"2022-08-11T10:02:56.466571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We need the suvived columns in another variable for creating our model**","metadata":{}},{"cell_type":"code","source":"k = final_train.pop(\"Survived\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.469778Z","iopub.execute_input":"2022-08-11T10:02:56.470595Z","iopub.status.idle":"2022-08-11T10:02:56.476997Z","shell.execute_reply.started":"2022-08-11T10:02:56.470545Z","shell.execute_reply":"2022-08-11T10:02:56.475802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = k\nprint(y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.478945Z","iopub.execute_input":"2022-08-11T10:02:56.479833Z","iopub.status.idle":"2022-08-11T10:02:56.493762Z","shell.execute_reply.started":"2022-08-11T10:02:56.479781Z","shell.execute_reply":"2022-08-11T10:02:56.492461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.495860Z","iopub.execute_input":"2022-08-11T10:02:56.496670Z","iopub.status.idle":"2022-08-11T10:02:56.506200Z","shell.execute_reply.started":"2022-08-11T10:02:56.496621Z","shell.execute_reply":"2022-08-11T10:02:56.504781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TEST PART**\n","metadata":{}},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.508202Z","iopub.execute_input":"2022-08-11T10:02:56.509020Z","iopub.status.idle":"2022-08-11T10:02:56.535746Z","shell.execute_reply.started":"2022-08-11T10:02:56.508971Z","shell.execute_reply":"2022-08-11T10:02:56.534578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['Survived'] = test_data['Survived'].apply(str)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.538015Z","iopub.execute_input":"2022-08-11T10:02:56.538491Z","iopub.status.idle":"2022-08-11T10:02:56.545540Z","shell.execute_reply.started":"2022-08-11T10:02:56.538435Z","shell.execute_reply":"2022-08-11T10:02:56.544210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['Pclass'] = test_data['Pclass'].apply(str)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.547174Z","iopub.execute_input":"2022-08-11T10:02:56.548455Z","iopub.status.idle":"2022-08-11T10:02:56.558731Z","shell.execute_reply.started":"2022-08-11T10:02:56.548403Z","shell.execute_reply":"2022-08-11T10:02:56.557235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_test = pd.get_dummies(test_data,columns=[\"Sex\"])\nfinal_test.drop('Sex_female', axis = 1, inplace=True)\nfinal_test","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.562050Z","iopub.execute_input":"2022-08-11T10:02:56.562705Z","iopub.status.idle":"2022-08-11T10:02:56.594432Z","shell.execute_reply.started":"2022-08-11T10:02:56.562654Z","shell.execute_reply":"2022-08-11T10:02:56.593257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_test = pd.get_dummies(final_test,columns=[\"Pclass\"])\nfinal_test.drop('Pclass_1', axis = 1, inplace=True)\nfinal_test","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.596064Z","iopub.execute_input":"2022-08-11T10:02:56.596429Z","iopub.status.idle":"2022-08-11T10:02:56.625088Z","shell.execute_reply.started":"2022-08-11T10:02:56.596391Z","shell.execute_reply":"2022-08-11T10:02:56.623910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_test = pd.get_dummies(final_test,columns=[\"Embarked\"])\nfinal_test.drop('Embarked_C', axis = 1, inplace=True)\nfinal_test","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.626832Z","iopub.execute_input":"2022-08-11T10:02:56.627440Z","iopub.status.idle":"2022-08-11T10:02:56.657722Z","shell.execute_reply.started":"2022-08-11T10:02:56.627401Z","shell.execute_reply":"2022-08-11T10:02:56.656397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = final_test.pop(\"Survived\") # As we need y_test for validating our model","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.659426Z","iopub.execute_input":"2022-08-11T10:02:56.660762Z","iopub.status.idle":"2022-08-11T10:02:56.666319Z","shell.execute_reply.started":"2022-08-11T10:02:56.660712Z","shell.execute_reply":"2022-08-11T10:02:56.665342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.668076Z","iopub.execute_input":"2022-08-11T10:02:56.668859Z","iopub.status.idle":"2022-08-11T10:02:56.681218Z","shell.execute_reply.started":"2022-08-11T10:02:56.668807Z","shell.execute_reply":"2022-08-11T10:02:56.679827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_train","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.682410Z","iopub.execute_input":"2022-08-11T10:02:56.683351Z","iopub.status.idle":"2022-08-11T10:02:56.708727Z","shell.execute_reply.started":"2022-08-11T10:02:56.683315Z","shell.execute_reply":"2022-08-11T10:02:56.707550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_test","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.710441Z","iopub.execute_input":"2022-08-11T10:02:56.710869Z","iopub.status.idle":"2022-08-11T10:02:56.733344Z","shell.execute_reply.started":"2022-08-11T10:02:56.710831Z","shell.execute_reply":"2022-08-11T10:02:56.731551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Scalling**","metadata":{}},{"cell_type":"markdown","source":"**Fitting**","metadata":{}},{"cell_type":"code","source":"scaler= StandardScaler()\n\nscaler.fit(final_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.735239Z","iopub.execute_input":"2022-08-11T10:02:56.736117Z","iopub.status.idle":"2022-08-11T10:02:56.749752Z","shell.execute_reply.started":"2022-08-11T10:02:56.736065Z","shell.execute_reply":"2022-08-11T10:02:56.748506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Here, the magic happens, machine will create the model for us automaticly for us!**","metadata":{}},{"cell_type":"code","source":"Logistic_model = LogisticRegression(max_iter=4000)\n\nLogistic_model.fit(final_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.751438Z","iopub.execute_input":"2022-08-11T10:02:56.752134Z","iopub.status.idle":"2022-08-11T10:02:56.859315Z","shell.execute_reply.started":"2022-08-11T10:02:56.752097Z","shell.execute_reply":"2022-08-11T10:02:56.858122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred =  Logistic_model.predict(final_test)  #predicting the test","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.860942Z","iopub.execute_input":"2022-08-11T10:02:56.861318Z","iopub.status.idle":"2022-08-11T10:02:56.869530Z","shell.execute_reply.started":"2022-08-11T10:02:56.861283Z","shell.execute_reply":"2022-08-11T10:02:56.868656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_score(y_test, y_pred) # Testing the accuracy of our model by comparing y test and y predicted","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.870862Z","iopub.execute_input":"2022-08-11T10:02:56.871928Z","iopub.status.idle":"2022-08-11T10:02:56.881985Z","shell.execute_reply.started":"2022-08-11T10:02:56.871889Z","shell.execute_reply":"2022-08-11T10:02:56.881112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Lets go we have 94.9% accuracy!**","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame() #Saving the result\nsubmission['PassengerId'] = test_data['PassengerId']\nsubmission['Survived'] = y_pred\nsubmission.to_csv('Logistic_Reg_submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.883258Z","iopub.execute_input":"2022-08-11T10:02:56.883963Z","iopub.status.idle":"2022-08-11T10:02:56.899145Z","shell.execute_reply.started":"2022-08-11T10:02:56.883909Z","shell.execute_reply":"2022-08-11T10:02:56.897898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Confusion Matrix","metadata":{}},{"cell_type":"code","source":"plot_confusion_matrix(Logistic_model, final_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T10:02:56.900808Z","iopub.execute_input":"2022-08-11T10:02:56.901172Z","iopub.status.idle":"2022-08-11T10:02:57.135563Z","shell.execute_reply.started":"2022-08-11T10:02:56.901139Z","shell.execute_reply":"2022-08-11T10:02:57.134342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}