{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom matplotlib import style\nimport seaborn as sns\n\nfrom sklearn.metrics import accuracy_score, confusion_matrix, classification_report, plot_confusion_matrix, precision_score, recall_score\n\nfrom sklearn.preprocessing import StandardScaler\n\nfrom sklearn.linear_model import LogisticRegression\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T14:18:03.234948Z","iopub.execute_input":"2022-08-10T14:18:03.235715Z","iopub.status.idle":"2022-08-10T14:18:04.748179Z","shell.execute_reply.started":"2022-08-10T14:18:03.235621Z","shell.execute_reply":"2022-08-10T14:18:04.746547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#importing input data\ntrain_data = pd.read_csv('../input/titanic/train.csv')\n\ntest_data = pd.read_csv('../input/titanic/test.csv') \n\nval_data = pd.read_csv('../input/titanic/gender_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:04.750581Z","iopub.execute_input":"2022-08-10T14:18:04.751050Z","iopub.status.idle":"2022-08-10T14:18:04.793240Z","shell.execute_reply.started":"2022-08-10T14:18:04.751019Z","shell.execute_reply":"2022-08-10T14:18:04.792280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Let us see the train Data","metadata":{}},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:04.794704Z","iopub.execute_input":"2022-08-10T14:18:04.795492Z","iopub.status.idle":"2022-08-10T14:18:04.836204Z","shell.execute_reply.started":"2022-08-10T14:18:04.795384Z","shell.execute_reply":"2022-08-10T14:18:04.835090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Lets see the Test Data","metadata":{}},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:04.838857Z","iopub.execute_input":"2022-08-10T14:18:04.839471Z","iopub.status.idle":"2022-08-10T14:18:04.861772Z","shell.execute_reply.started":"2022-08-10T14:18:04.839436Z","shell.execute_reply":"2022-08-10T14:18:04.860470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**As the test data does not have any column survive we need to add it in the data set from the kaggle inputs**","metadata":{}},{"cell_type":"code","source":"test_data['Survived'] = val_data['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:04.863868Z","iopub.execute_input":"2022-08-10T14:18:04.864687Z","iopub.status.idle":"2022-08-10T14:18:04.876948Z","shell.execute_reply.started":"2022-08-10T14:18:04.864636Z","shell.execute_reply":"2022-08-10T14:18:04.875875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:04.878235Z","iopub.execute_input":"2022-08-10T14:18:04.878997Z","iopub.status.idle":"2022-08-10T14:18:04.911087Z","shell.execute_reply.started":"2022-08-10T14:18:04.878952Z","shell.execute_reply":"2022-08-10T14:18:04.909860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preparation","metadata":{}},{"cell_type":"markdown","source":"**Both the train data and test data have some null values or missing values, we need to deal with them before working with the model**","metadata":{}},{"cell_type":"code","source":"#DATA PREP\ntrain_data.isnull().sum()#finding the amount of missing values ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:04.912885Z","iopub.execute_input":"2022-08-10T14:18:04.914309Z","iopub.status.idle":"2022-08-10T14:18:04.926538Z","shell.execute_reply.started":"2022-08-10T14:18:04.914251Z","shell.execute_reply":"2022-08-10T14:18:04.924788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"100*(train_data.isnull().sum()/len(train_data)) #finding the percentage of missing values","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:04.928502Z","iopub.execute_input":"2022-08-10T14:18:04.929574Z","iopub.status.idle":"2022-08-10T14:18:04.942383Z","shell.execute_reply.started":"2022-08-10T14:18:04.929522Z","shell.execute_reply":"2022-08-10T14:18:04.941174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Lets make a function to find the amount of missing values percentage for each column to make our work easier**","metadata":{}},{"cell_type":"code","source":"def null_percent_func(df):\n    null_percent= 100*(df.isnull().sum()/len(df))\n    null_percent= null_percent[null_percent>0].sort_values()\n    return null_percent","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:04.944273Z","iopub.execute_input":"2022-08-10T14:18:04.944878Z","iopub.status.idle":"2022-08-10T14:18:04.952688Z","shell.execute_reply.started":"2022-08-10T14:18:04.944844Z","shell.execute_reply":"2022-08-10T14:18:04.949858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_percent= null_percent_func(train_data) #taking only the missing colums","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:04.956414Z","iopub.execute_input":"2022-08-10T14:18:04.957013Z","iopub.status.idle":"2022-08-10T14:18:04.968884Z","shell.execute_reply.started":"2022-08-10T14:18:04.956978Z","shell.execute_reply":"2022-08-10T14:18:04.967506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(null_percent)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:04.971104Z","iopub.execute_input":"2022-08-10T14:18:04.971644Z","iopub.status.idle":"2022-08-10T14:18:04.979903Z","shell.execute_reply.started":"2022-08-10T14:18:04.971595Z","shell.execute_reply":"2022-08-10T14:18:04.978608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data= train_data.dropna(axis=0, subset=['Embarked']) #dropped the whole row as the percent of its low, so dropping them ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:04.982194Z","iopub.execute_input":"2022-08-10T14:18:04.983060Z","iopub.status.idle":"2022-08-10T14:18:04.993660Z","shell.execute_reply.started":"2022-08-10T14:18:04.983010Z","shell.execute_reply":"2022-08-10T14:18:04.992556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(train_data[\"Embarked\"])\ntrain_data.iloc[60]","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:04.995876Z","iopub.execute_input":"2022-08-10T14:18:04.996347Z","iopub.status.idle":"2022-08-10T14:18:05.007440Z","shell.execute_reply.started":"2022-08-10T14:18:04.996300Z","shell.execute_reply":"2022-08-10T14:18:05.005727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_percent = null_percent_func(train_data) #taking only the missing colums","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.009058Z","iopub.execute_input":"2022-08-10T14:18:05.009853Z","iopub.status.idle":"2022-08-10T14:18:05.022798Z","shell.execute_reply.started":"2022-08-10T14:18:05.009804Z","shell.execute_reply":"2022-08-10T14:18:05.021632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(null_percent)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.024436Z","iopub.execute_input":"2022-08-10T14:18:05.024806Z","iopub.status.idle":"2022-08-10T14:18:05.033351Z","shell.execute_reply.started":"2022-08-10T14:18:05.024772Z","shell.execute_reply":"2022-08-10T14:18:05.032254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You can see that one of the null percentage from the output of the function is gone, it means we are progressing!","metadata":{}},{"cell_type":"markdown","source":"**As we can see Age column have 19% of missing data, so we can not delete the whole column, for this kind of case we can take the mean and take the mean values in the null places**","metadata":{}},{"cell_type":"code","source":"train_data['Age'] = train_data['Age'].fillna(train_data['Age'].median()) #taking the median values of age columns","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.035014Z","iopub.execute_input":"2022-08-10T14:18:05.035663Z","iopub.status.idle":"2022-08-10T14:18:05.047625Z","shell.execute_reply.started":"2022-08-10T14:18:05.035628Z","shell.execute_reply":"2022-08-10T14:18:05.046501Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_percent = null_percent_func(train_data) #taking only the missing colums\nprint(null_percent)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.049926Z","iopub.execute_input":"2022-08-10T14:18:05.050573Z","iopub.status.idle":"2022-08-10T14:18:05.061746Z","shell.execute_reply.started":"2022-08-10T14:18:05.050531Z","shell.execute_reply":"2022-08-10T14:18:05.060487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Only one obstacle is left!","metadata":{}},{"cell_type":"code","source":"train_data=train_data.drop(['Cabin'], axis=1) #too many missing values, so it will be good if we drop the whole column","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.063196Z","iopub.execute_input":"2022-08-10T14:18:05.064462Z","iopub.status.idle":"2022-08-10T14:18:05.074855Z","shell.execute_reply.started":"2022-08-10T14:18:05.064389Z","shell.execute_reply":"2022-08-10T14:18:05.073726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.drop(['Name','Ticket'], axis = 1, inplace=True) #As Name and ticket are useless","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.076275Z","iopub.execute_input":"2022-08-10T14:18:05.076642Z","iopub.status.idle":"2022-08-10T14:18:05.096758Z","shell.execute_reply.started":"2022-08-10T14:18:05.076610Z","shell.execute_reply":"2022-08-10T14:18:05.095659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Yes we are done with our train data part, now we need to do the same things for the test data part**","metadata":{}},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.098016Z","iopub.execute_input":"2022-08-10T14:18:05.098678Z","iopub.status.idle":"2022-08-10T14:18:05.124118Z","shell.execute_reply.started":"2022-08-10T14:18:05.098634Z","shell.execute_reply":"2022-08-10T14:18:05.122266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Test pre-processing part\n100*(test_data.isnull().sum()/len(test_data))","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.125646Z","iopub.execute_input":"2022-08-10T14:18:05.126758Z","iopub.status.idle":"2022-08-10T14:18:05.139450Z","shell.execute_reply.started":"2022-08-10T14:18:05.126705Z","shell.execute_reply":"2022-08-10T14:18:05.138076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here embarked is okay but Fare got some null values","metadata":{}},{"cell_type":"code","source":"null_percent = null_percent_func(test_data) #taking only the missing colums\nprint(null_percent)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.141181Z","iopub.execute_input":"2022-08-10T14:18:05.141698Z","iopub.status.idle":"2022-08-10T14:18:05.157949Z","shell.execute_reply.started":"2022-08-10T14:18:05.141650Z","shell.execute_reply":"2022-08-10T14:18:05.155471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['Fare'] = test_data['Fare'].fillna(test_data['Fare'].median()) #as for the test we can not drop any values","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.159341Z","iopub.execute_input":"2022-08-10T14:18:05.160305Z","iopub.status.idle":"2022-08-10T14:18:05.169586Z","shell.execute_reply.started":"2022-08-10T14:18:05.160245Z","shell.execute_reply":"2022-08-10T14:18:05.168145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_percent = null_percent_func(test_data) #taking only the missing colums\nprint(null_percent)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.171319Z","iopub.execute_input":"2022-08-10T14:18:05.171752Z","iopub.status.idle":"2022-08-10T14:18:05.188828Z","shell.execute_reply.started":"2022-08-10T14:18:05.171717Z","shell.execute_reply":"2022-08-10T14:18:05.187297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['Age'] =test_data['Age'].fillna(test_data['Age'].median())","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.190723Z","iopub.execute_input":"2022-08-10T14:18:05.191453Z","iopub.status.idle":"2022-08-10T14:18:05.199612Z","shell.execute_reply.started":"2022-08-10T14:18:05.191390Z","shell.execute_reply":"2022-08-10T14:18:05.198192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_percent = null_percent_func(test_data) #taking only the missing colums\nprint(null_percent)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.201087Z","iopub.execute_input":"2022-08-10T14:18:05.201875Z","iopub.status.idle":"2022-08-10T14:18:05.216147Z","shell.execute_reply.started":"2022-08-10T14:18:05.201828Z","shell.execute_reply":"2022-08-10T14:18:05.214889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data=test_data.drop(['Cabin'], axis=1) # as for the train we have done it, and we want to do the same for the test","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.217755Z","iopub.execute_input":"2022-08-10T14:18:05.218562Z","iopub.status.idle":"2022-08-10T14:18:05.225683Z","shell.execute_reply.started":"2022-08-10T14:18:05.218521Z","shell.execute_reply":"2022-08-10T14:18:05.224580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.drop(['Name','Ticket'], axis = 1, inplace=True) #As Name and ticket are useless\ntest_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.235720Z","iopub.execute_input":"2022-08-10T14:18:05.236490Z","iopub.status.idle":"2022-08-10T14:18:05.265111Z","shell.execute_reply.started":"2022-08-10T14:18:05.236445Z","shell.execute_reply":"2022-08-10T14:18:05.263807Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dummy variables","metadata":{}},{"cell_type":"markdown","source":"**As computer is not inteligent like us, we need to make the data more simpler for computer to understand.**","metadata":{}},{"cell_type":"markdown","source":"Here unique() function helps us to ditect the amount of unique values in a column, our plan is to target the columns which have 2 3 unique values only!","metadata":{}},{"cell_type":"code","source":"#we need to make this good for regression\n\n#So, gotta work with dummy variables\n#train_data[\"Pclass\"].unique()\ntrain_data[\"Embarked\"].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.266676Z","iopub.execute_input":"2022-08-10T14:18:05.267799Z","iopub.status.idle":"2022-08-10T14:18:05.275885Z","shell.execute_reply.started":"2022-08-10T14:18:05.267752Z","shell.execute_reply":"2022-08-10T14:18:05.274968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.277267Z","iopub.execute_input":"2022-08-10T14:18:05.277925Z","iopub.status.idle":"2022-08-10T14:18:05.301154Z","shell.execute_reply.started":"2022-08-10T14:18:05.277890Z","shell.execute_reply":"2022-08-10T14:18:05.299915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['Survived'] = train_data['Survived'].apply(str)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.302765Z","iopub.execute_input":"2022-08-10T14:18:05.303490Z","iopub.status.idle":"2022-08-10T14:18:05.310390Z","shell.execute_reply.started":"2022-08-10T14:18:05.303440Z","shell.execute_reply":"2022-08-10T14:18:05.309358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.311643Z","iopub.execute_input":"2022-08-10T14:18:05.311990Z","iopub.status.idle":"2022-08-10T14:18:05.329602Z","shell.execute_reply.started":"2022-08-10T14:18:05.311960Z","shell.execute_reply":"2022-08-10T14:18:05.328334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['Pclass'] = train_data['Pclass'].apply(str)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.331006Z","iopub.execute_input":"2022-08-10T14:18:05.331607Z","iopub.status.idle":"2022-08-10T14:18:05.338598Z","shell.execute_reply.started":"2022-08-10T14:18:05.331573Z","shell.execute_reply":"2022-08-10T14:18:05.337092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.340262Z","iopub.execute_input":"2022-08-10T14:18:05.340745Z","iopub.status.idle":"2022-08-10T14:18:05.366154Z","shell.execute_reply.started":"2022-08-10T14:18:05.340702Z","shell.execute_reply":"2022-08-10T14:18:05.364966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Now we will start working with the targeted columns, get_dummies() function will create 3 columns if a columns have 3 unique object. This is why we need to drop one of the columns from the three columns**","metadata":{}},{"cell_type":"code","source":"final_train = pd.get_dummies(train_data,columns=[\"Sex\"])\nfinal_train.drop('Sex_female', axis = 1, inplace=True)\nfinal_train\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.367718Z","iopub.execute_input":"2022-08-10T14:18:05.368056Z","iopub.status.idle":"2022-08-10T14:18:05.396682Z","shell.execute_reply.started":"2022-08-10T14:18:05.368026Z","shell.execute_reply":"2022-08-10T14:18:05.395766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_train = pd.get_dummies(final_train,columns=[\"Pclass\"])\nfinal_train.drop('Pclass_1', axis = 1, inplace=True)\nfinal_train","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.397904Z","iopub.execute_input":"2022-08-10T14:18:05.398922Z","iopub.status.idle":"2022-08-10T14:18:05.427816Z","shell.execute_reply.started":"2022-08-10T14:18:05.398887Z","shell.execute_reply":"2022-08-10T14:18:05.426781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_train = pd.get_dummies(final_train,columns=[\"Embarked\"])\nfinal_train.drop('Embarked_C', axis = 1, inplace=True)\nfinal_train","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.429092Z","iopub.execute_input":"2022-08-10T14:18:05.429694Z","iopub.status.idle":"2022-08-10T14:18:05.456912Z","shell.execute_reply.started":"2022-08-10T14:18:05.429658Z","shell.execute_reply":"2022-08-10T14:18:05.455745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**We need the suvived columns in another variable for creating our model**","metadata":{}},{"cell_type":"code","source":"k = final_train.pop(\"Survived\")","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.458273Z","iopub.execute_input":"2022-08-10T14:18:05.458852Z","iopub.status.idle":"2022-08-10T14:18:05.464143Z","shell.execute_reply.started":"2022-08-10T14:18:05.458819Z","shell.execute_reply":"2022-08-10T14:18:05.462923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = k\nprint(y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.465368Z","iopub.execute_input":"2022-08-10T14:18:05.466222Z","iopub.status.idle":"2022-08-10T14:18:05.479511Z","shell.execute_reply.started":"2022-08-10T14:18:05.466189Z","shell.execute_reply":"2022-08-10T14:18:05.478318Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.480995Z","iopub.execute_input":"2022-08-10T14:18:05.481588Z","iopub.status.idle":"2022-08-10T14:18:05.488791Z","shell.execute_reply.started":"2022-08-10T14:18:05.481555Z","shell.execute_reply":"2022-08-10T14:18:05.487817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TEST PART**\n","metadata":{}},{"cell_type":"code","source":"test_data","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.489909Z","iopub.execute_input":"2022-08-10T14:18:05.490944Z","iopub.status.idle":"2022-08-10T14:18:05.518166Z","shell.execute_reply.started":"2022-08-10T14:18:05.490911Z","shell.execute_reply":"2022-08-10T14:18:05.516984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['Survived'] = test_data['Survived'].apply(str)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.519682Z","iopub.execute_input":"2022-08-10T14:18:05.520817Z","iopub.status.idle":"2022-08-10T14:18:05.528924Z","shell.execute_reply.started":"2022-08-10T14:18:05.520772Z","shell.execute_reply":"2022-08-10T14:18:05.527549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['Pclass'] = test_data['Pclass'].apply(str)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.530977Z","iopub.execute_input":"2022-08-10T14:18:05.531704Z","iopub.status.idle":"2022-08-10T14:18:05.545300Z","shell.execute_reply.started":"2022-08-10T14:18:05.531649Z","shell.execute_reply":"2022-08-10T14:18:05.544377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_test = pd.get_dummies(test_data,columns=[\"Sex\"])\nfinal_test.drop('Sex_female', axis = 1, inplace=True)\nfinal_test","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.547159Z","iopub.execute_input":"2022-08-10T14:18:05.548032Z","iopub.status.idle":"2022-08-10T14:18:05.578753Z","shell.execute_reply.started":"2022-08-10T14:18:05.547987Z","shell.execute_reply":"2022-08-10T14:18:05.577481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_test = pd.get_dummies(final_test,columns=[\"Pclass\"])\nfinal_test.drop('Pclass_1', axis = 1, inplace=True)\nfinal_test","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.581596Z","iopub.execute_input":"2022-08-10T14:18:05.582351Z","iopub.status.idle":"2022-08-10T14:18:05.610489Z","shell.execute_reply.started":"2022-08-10T14:18:05.582305Z","shell.execute_reply":"2022-08-10T14:18:05.609230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_test = pd.get_dummies(final_test,columns=[\"Embarked\"])\nfinal_test.drop('Embarked_C', axis = 1, inplace=True)\nfinal_test","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.612194Z","iopub.execute_input":"2022-08-10T14:18:05.612963Z","iopub.status.idle":"2022-08-10T14:18:05.640239Z","shell.execute_reply.started":"2022-08-10T14:18:05.612916Z","shell.execute_reply":"2022-08-10T14:18:05.639170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = final_test.pop(\"Survived\") # As we need y_test for validating our model","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.642888Z","iopub.execute_input":"2022-08-10T14:18:05.643696Z","iopub.status.idle":"2022-08-10T14:18:05.650272Z","shell.execute_reply.started":"2022-08-10T14:18:05.643648Z","shell.execute_reply":"2022-08-10T14:18:05.648794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(y_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.651872Z","iopub.execute_input":"2022-08-10T14:18:05.653033Z","iopub.status.idle":"2022-08-10T14:18:05.666511Z","shell.execute_reply.started":"2022-08-10T14:18:05.652987Z","shell.execute_reply":"2022-08-10T14:18:05.665167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_train","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.668542Z","iopub.execute_input":"2022-08-10T14:18:05.669179Z","iopub.status.idle":"2022-08-10T14:18:05.694138Z","shell.execute_reply.started":"2022-08-10T14:18:05.669145Z","shell.execute_reply":"2022-08-10T14:18:05.692469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_test","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.695606Z","iopub.execute_input":"2022-08-10T14:18:05.695965Z","iopub.status.idle":"2022-08-10T14:18:05.720233Z","shell.execute_reply.started":"2022-08-10T14:18:05.695934Z","shell.execute_reply":"2022-08-10T14:18:05.718969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Scalling**","metadata":{}},{"cell_type":"markdown","source":"**Fitting**","metadata":{}},{"cell_type":"code","source":"scaler= StandardScaler()\n\nscaler.fit(final_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.722181Z","iopub.execute_input":"2022-08-10T14:18:05.723020Z","iopub.status.idle":"2022-08-10T14:18:05.736458Z","shell.execute_reply.started":"2022-08-10T14:18:05.722975Z","shell.execute_reply":"2022-08-10T14:18:05.735538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Here, the magic happens, machine will create the model for us automaticly for us!**","metadata":{}},{"cell_type":"code","source":"Logistic_model = LogisticRegression(max_iter=4000)\n\nLogistic_model.fit(final_train,y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.737855Z","iopub.execute_input":"2022-08-10T14:18:05.739058Z","iopub.status.idle":"2022-08-10T14:18:05.854560Z","shell.execute_reply.started":"2022-08-10T14:18:05.739021Z","shell.execute_reply":"2022-08-10T14:18:05.853673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred =  Logistic_model.predict(final_test)  #predicting the test","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.855828Z","iopub.execute_input":"2022-08-10T14:18:05.856542Z","iopub.status.idle":"2022-08-10T14:18:05.864150Z","shell.execute_reply.started":"2022-08-10T14:18:05.856508Z","shell.execute_reply":"2022-08-10T14:18:05.863130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_score(y_test, y_pred) # Testing the accuracy of our model by comparing y test and y predicted","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.865469Z","iopub.execute_input":"2022-08-10T14:18:05.866523Z","iopub.status.idle":"2022-08-10T14:18:05.878431Z","shell.execute_reply.started":"2022-08-10T14:18:05.866482Z","shell.execute_reply":"2022-08-10T14:18:05.877339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Lets go we have 94.9% accuracy!**","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame() #Saving the result\nsubmission['PassengerId'] = test_data['PassengerId']\nsubmission['Survived'] = y_pred\nsubmission.to_csv('Logistic_Reg_submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:05.879864Z","iopub.execute_input":"2022-08-10T14:18:05.881338Z","iopub.status.idle":"2022-08-10T14:18:05.892323Z","shell.execute_reply.started":"2022-08-10T14:18:05.881302Z","shell.execute_reply":"2022-08-10T14:18:05.890976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Confusion Matrix","metadata":{}},{"cell_type":"code","source":"plot_confusion_matrix(Logistic_model, final_test, y_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:19:07.193878Z","iopub.execute_input":"2022-08-10T14:19:07.194313Z","iopub.status.idle":"2022-08-10T14:19:07.469346Z","shell.execute_reply.started":"2022-08-10T14:19:07.194279Z","shell.execute_reply":"2022-08-10T14:19:07.468117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}