{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-09T10:02:48.164270Z","iopub.execute_input":"2022-07-09T10:02:48.164719Z","iopub.status.idle":"2022-07-09T10:02:48.173745Z","shell.execute_reply.started":"2022-07-09T10:02:48.164685Z","shell.execute_reply":"2022-07-09T10:02:48.172543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Import Necessary libraies","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.svm import SVC\nfrom sklearn.metrics import classification_report,plot_confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.243858Z","iopub.execute_input":"2022-07-09T10:02:48.245076Z","iopub.status.idle":"2022-07-09T10:02:48.251882Z","shell.execute_reply.started":"2022-07-09T10:02:48.245003Z","shell.execute_reply":"2022-07-09T10:02:48.250763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/titanic/train.csv')\ndf_test = pd.read_csv('/kaggle/input/titanic/test.csv')\nresult = pd.read_csv('/kaggle/input/titanic/gender_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.253692Z","iopub.execute_input":"2022-07-09T10:02:48.254021Z","iopub.status.idle":"2022-07-09T10:02:48.277947Z","shell.execute_reply.started":"2022-07-09T10:02:48.253992Z","shell.execute_reply":"2022-07-09T10:02:48.276952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.495376Z","iopub.execute_input":"2022-07-09T10:02:48.496046Z","iopub.status.idle":"2022-07-09T10:02:48.518999Z","shell.execute_reply.started":"2022-07-09T10:02:48.496000Z","shell.execute_reply":"2022-07-09T10:02:48.518151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cleaning df_train","metadata":{}},{"cell_type":"code","source":"df_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.520753Z","iopub.execute_input":"2022-07-09T10:02:48.521512Z","iopub.status.idle":"2022-07-09T10:02:48.540468Z","shell.execute_reply.started":"2022-07-09T10:02:48.521478Z","shell.execute_reply":"2022-07-09T10:02:48.539247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.describe().transpose()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.541705Z","iopub.execute_input":"2022-07-09T10:02:48.542013Z","iopub.status.idle":"2022-07-09T10:02:48.581337Z","shell.execute_reply.started":"2022-07-09T10:02:48.541985Z","shell.execute_reply":"2022-07-09T10:02:48.580251Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sum of null values in each column\ndf_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.584338Z","iopub.execute_input":"2022-07-09T10:02:48.584688Z","iopub.status.idle":"2022-07-09T10:02:48.593774Z","shell.execute_reply.started":"2022-07-09T10:02:48.584657Z","shell.execute_reply":"2022-07-09T10:02:48.592763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filling the missing Age column with the median of age in the dataset\ndf_train['Age'] = df_train['Age'].fillna(df_train['Age'].median())","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.595976Z","iopub.execute_input":"2022-07-09T10:02:48.596508Z","iopub.status.idle":"2022-07-09T10:02:48.604369Z","shell.execute_reply.started":"2022-07-09T10:02:48.596476Z","shell.execute_reply":"2022-07-09T10:02:48.603629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dropping the Cabin column\ndf_train = df_train.drop('Cabin',axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.605604Z","iopub.execute_input":"2022-07-09T10:02:48.606133Z","iopub.status.idle":"2022-07-09T10:02:48.616310Z","shell.execute_reply.started":"2022-07-09T10:02:48.606102Z","shell.execute_reply":"2022-07-09T10:02:48.615070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.617881Z","iopub.execute_input":"2022-07-09T10:02:48.618422Z","iopub.status.idle":"2022-07-09T10:02:48.630334Z","shell.execute_reply.started":"2022-07-09T10:02:48.618390Z","shell.execute_reply":"2022-07-09T10:02:48.629113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the two rows that has no embarked value\ndf_train[df_train['Embarked'].isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.631957Z","iopub.execute_input":"2022-07-09T10:02:48.632309Z","iopub.status.idle":"2022-07-09T10:02:48.649564Z","shell.execute_reply.started":"2022-07-09T10:02:48.632278Z","shell.execute_reply":"2022-07-09T10:02:48.648255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filling the Embarked column\ndf_train['Embarked'] = df_train['Embarked'].fillna('S')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.651500Z","iopub.execute_input":"2022-07-09T10:02:48.652297Z","iopub.status.idle":"2022-07-09T10:02:48.663266Z","shell.execute_reply.started":"2022-07-09T10:02:48.652229Z","shell.execute_reply":"2022-07-09T10:02:48.662399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.667323Z","iopub.execute_input":"2022-07-09T10:02:48.668163Z","iopub.status.idle":"2022-07-09T10:02:48.679173Z","shell.execute_reply.started":"2022-07-09T10:02:48.668128Z","shell.execute_reply":"2022-07-09T10:02:48.677889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Analyzing df_train","metadata":{}},{"cell_type":"code","source":"# the columns that are corroleted to the survived column\ndf_train.corr()['Survived'].sort_values()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.680945Z","iopub.execute_input":"2022-07-09T10:02:48.681661Z","iopub.status.idle":"2022-07-09T10:02:48.692338Z","shell.execute_reply.started":"2022-07-09T10:02:48.681607Z","shell.execute_reply":"2022-07-09T10:02:48.691307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.corr()['Survived'].sort_values().plot(kind='bar')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.743488Z","iopub.execute_input":"2022-07-09T10:02:48.744720Z","iopub.status.idle":"2022-07-09T10:02:48.949319Z","shell.execute_reply.started":"2022-07-09T10:02:48.744673Z","shell.execute_reply":"2022-07-09T10:02:48.948221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8,4),dpi=100)\nsns.scatterplot(data=df_train,x='Fare',y='Pclass',hue='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:48.951721Z","iopub.execute_input":"2022-07-09T10:02:48.952376Z","iopub.status.idle":"2022-07-09T10:02:49.277073Z","shell.execute_reply.started":"2022-07-09T10:02:48.952321Z","shell.execute_reply":"2022-07-09T10:02:49.275707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# percentage of people that survived\nsurvived = np.sum(df_train['Survived']==1)\nunsurvied = np.sum(df_train['Survived']==0)\n\nperc_surv = 100 * survived / len(df_train)\nperc_unsurv = 100 * unsurvied / len(df_train)\n\nprint(f'Percentage of People that survived: {perc_surv}')\nprint(f\"Percentage of People that didn't survive: {perc_unsurv}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:49.278839Z","iopub.execute_input":"2022-07-09T10:02:49.279164Z","iopub.status.idle":"2022-07-09T10:02:49.287304Z","shell.execute_reply.started":"2022-07-09T10:02:49.279128Z","shell.execute_reply":"2022-07-09T10:02:49.286580Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8,4),dpi=100)\nsns.countplot(data=df_train,x='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:49.288387Z","iopub.execute_input":"2022-07-09T10:02:49.289269Z","iopub.status.idle":"2022-07-09T10:02:49.456002Z","shell.execute_reply.started":"2022-07-09T10:02:49.289236Z","shell.execute_reply":"2022-07-09T10:02:49.455047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# percentage of survived male and female\nsurvived_male = len(df_train[(df_train['Sex']=='male') & (df_train['Survived']==1)])\nsurvived_female = len(df_train[(df_train['Sex']=='female') & (df_train['Survived']==1)])\n\nsurv_male_perc = 100 * survived_male / len(df_train['Sex']=='male')\nsurv_female_perc = 100 * survived_female / len(df_train['Sex']=='female')\n\nprint(f'Perentage of Survived Male: {surv_male_perc}')\nprint(f'Perentage of Survived Female: {surv_female_perc}')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:49.459725Z","iopub.execute_input":"2022-07-09T10:02:49.460496Z","iopub.status.idle":"2022-07-09T10:02:49.472435Z","shell.execute_reply.started":"2022-07-09T10:02:49.460449Z","shell.execute_reply":"2022-07-09T10:02:49.471253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8,4),dpi=100)\nsns.countplot(data=df_train,x='Sex',hue='Survived')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:49.474193Z","iopub.execute_input":"2022-07-09T10:02:49.474901Z","iopub.status.idle":"2022-07-09T10:02:49.661058Z","shell.execute_reply.started":"2022-07-09T10:02:49.474856Z","shell.execute_reply":"2022-07-09T10:02:49.659887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8,4),dpi=100)\nsns.scatterplot(data=df_train,x='Age',y='PassengerId',hue='Survived',palette='viridis')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:49.663338Z","iopub.execute_input":"2022-07-09T10:02:49.664478Z","iopub.status.idle":"2022-07-09T10:02:49.985277Z","shell.execute_reply.started":"2022-07-09T10:02:49.664431Z","shell.execute_reply":"2022-07-09T10:02:49.983772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.heatmap(df_train.corr())","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:49.986679Z","iopub.execute_input":"2022-07-09T10:02:49.986988Z","iopub.status.idle":"2022-07-09T10:02:50.275557Z","shell.execute_reply.started":"2022-07-09T10:02:49.986959Z","shell.execute_reply":"2022-07-09T10:02:50.274709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Cleaning and scaling the test_data(df_test)","metadata":{}},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.277231Z","iopub.execute_input":"2022-07-09T10:02:50.277927Z","iopub.status.idle":"2022-07-09T10:02:50.295170Z","shell.execute_reply.started":"2022-07-09T10:02:50.277888Z","shell.execute_reply":"2022-07-09T10:02:50.293884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.296609Z","iopub.execute_input":"2022-07-09T10:02:50.296969Z","iopub.status.idle":"2022-07-09T10:02:50.315543Z","shell.execute_reply.started":"2022-07-09T10:02:50.296934Z","shell.execute_reply":"2022-07-09T10:02:50.314578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dropping the columns that we won't need for testing our model\nfin_test = df_test.drop(['Name','Ticket','Cabin'],axis=1)\nfin_test","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.317197Z","iopub.execute_input":"2022-07-09T10:02:50.317893Z","iopub.status.idle":"2022-07-09T10:02:50.337778Z","shell.execute_reply.started":"2022-07-09T10:02:50.317857Z","shell.execute_reply":"2022-07-09T10:02:50.336955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filling the missing values in Age column\nfin_test['Age'] = fin_test['Age'].fillna(fin_test['Age'].median())\nfin_test","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.339015Z","iopub.execute_input":"2022-07-09T10:02:50.340359Z","iopub.status.idle":"2022-07-09T10:02:50.364648Z","shell.execute_reply.started":"2022-07-09T10:02:50.340296Z","shell.execute_reply":"2022-07-09T10:02:50.363788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fin_test.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.366053Z","iopub.execute_input":"2022-07-09T10:02:50.366721Z","iopub.status.idle":"2022-07-09T10:02:50.378657Z","shell.execute_reply.started":"2022-07-09T10:02:50.366673Z","shell.execute_reply":"2022-07-09T10:02:50.377687Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# finding the row that has no Embarked value\nfin_test[fin_test[\"Fare\"].isnull()]","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.384215Z","iopub.execute_input":"2022-07-09T10:02:50.384915Z","iopub.status.idle":"2022-07-09T10:02:50.400562Z","shell.execute_reply.started":"2022-07-09T10:02:50.384875Z","shell.execute_reply":"2022-07-09T10:02:50.399643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# filling the value with average Fare in the dataset\nfin_test['Fare'] = fin_test['Fare'].fillna(fin_test['Fare'].mean())","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.402091Z","iopub.execute_input":"2022-07-09T10:02:50.402411Z","iopub.status.idle":"2022-07-09T10:02:50.408691Z","shell.execute_reply.started":"2022-07-09T10:02:50.402383Z","shell.execute_reply":"2022-07-09T10:02:50.407315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fin_test.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.410289Z","iopub.execute_input":"2022-07-09T10:02:50.410678Z","iopub.status.idle":"2022-07-09T10:02:50.423511Z","shell.execute_reply.started":"2022-07-09T10:02:50.410645Z","shell.execute_reply":"2022-07-09T10:02:50.422192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting PassengerId as our index_column and creating dummy variables for the categroical columns\nfin_test = pd.get_dummies(fin_test,drop_first=True)\nfin_test = fin_test.set_index('PassengerId')\nfin_test","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.425576Z","iopub.execute_input":"2022-07-09T10:02:50.426516Z","iopub.status.idle":"2022-07-09T10:02:50.451051Z","shell.execute_reply.started":"2022-07-09T10:02:50.426469Z","shell.execute_reply":"2022-07-09T10:02:50.449840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scaling the data\nscaler = StandardScaler()\nscaled_df_test = scaler.fit_transform(fin_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.452892Z","iopub.execute_input":"2022-07-09T10:02:50.453433Z","iopub.status.idle":"2022-07-09T10:02:50.464224Z","shell.execute_reply.started":"2022-07-09T10:02:50.453386Z","shell.execute_reply":"2022-07-09T10:02:50.463157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparing our test_data(test_df) for building a Model","metadata":{}},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.465601Z","iopub.execute_input":"2022-07-09T10:02:50.466015Z","iopub.status.idle":"2022-07-09T10:02:50.483020Z","shell.execute_reply.started":"2022-07-09T10:02:50.465983Z","shell.execute_reply":"2022-07-09T10:02:50.481935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dropping the columns that we won't need and creating dummy variables for the categorical columns\nfinal_df = pd.get_dummies(df_train.drop(['Ticket','Name'],axis=1),drop_first=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.484180Z","iopub.execute_input":"2022-07-09T10:02:50.485154Z","iopub.status.idle":"2022-07-09T10:02:50.499423Z","shell.execute_reply.started":"2022-07-09T10:02:50.485104Z","shell.execute_reply":"2022-07-09T10:02:50.498393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# setting PassengerId as the index column\nfinal_df = final_df.set_index('PassengerId')\nfinal_df","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.501027Z","iopub.execute_input":"2022-07-09T10:02:50.501704Z","iopub.status.idle":"2022-07-09T10:02:50.522746Z","shell.execute_reply.started":"2022-07-09T10:02:50.501672Z","shell.execute_reply":"2022-07-09T10:02:50.521548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# splitting the data into x and y\nX = final_df.drop('Survived',axis=1)\ny = final_df['Survived']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.524203Z","iopub.execute_input":"2022-07-09T10:02:50.524855Z","iopub.status.idle":"2022-07-09T10:02:50.533153Z","shell.execute_reply.started":"2022-07-09T10:02:50.524803Z","shell.execute_reply":"2022-07-09T10:02:50.532322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# scaling the our train_data(X)\nscaled_X_train = scaler.fit_transform(X)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.534507Z","iopub.execute_input":"2022-07-09T10:02:50.535644Z","iopub.status.idle":"2022-07-09T10:02:50.547498Z","shell.execute_reply.started":"2022-07-09T10:02:50.535606Z","shell.execute_reply":"2022-07-09T10:02:50.546500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fitting to a Model - SVM\n- We are going to be using Support Vector Machine","metadata":{}},{"cell_type":"code","source":"model = SVC()\nmodel.fit(scaled_X_train,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.548907Z","iopub.execute_input":"2022-07-09T10:02:50.549938Z","iopub.status.idle":"2022-07-09T10:02:50.583325Z","shell.execute_reply.started":"2022-07-09T10:02:50.549900Z","shell.execute_reply":"2022-07-09T10:02:50.582429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# using our model to predict the test_data\npredictions = model.predict(scaled_df_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.584586Z","iopub.execute_input":"2022-07-09T10:02:50.585101Z","iopub.status.idle":"2022-07-09T10:02:50.599719Z","shell.execute_reply.started":"2022-07-09T10:02:50.585070Z","shell.execute_reply":"2022-07-09T10:02:50.598635Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.600877Z","iopub.execute_input":"2022-07-09T10:02:50.601192Z","iopub.status.idle":"2022-07-09T10:02:50.608382Z","shell.execute_reply.started":"2022-07-09T10:02:50.601164Z","shell.execute_reply":"2022-07-09T10:02:50.607602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Comparing our predictions to the result dataset","metadata":{}},{"cell_type":"code","source":"result","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.609549Z","iopub.execute_input":"2022-07-09T10:02:50.610227Z","iopub.status.idle":"2022-07-09T10:02:50.627311Z","shell.execute_reply.started":"2022-07-09T10:02:50.610194Z","shell.execute_reply":"2022-07-09T10:02:50.626479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# turning the result into a numpy array\nres = np.array(result['Survived'])\nres","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.628783Z","iopub.execute_input":"2022-07-09T10:02:50.629657Z","iopub.status.idle":"2022-07-09T10:02:50.641163Z","shell.execute_reply.started":"2022-07-09T10:02:50.629609Z","shell.execute_reply":"2022-07-09T10:02:50.640040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets compare it to our result\nprint(classification_report(res,predictions))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.642663Z","iopub.execute_input":"2022-07-09T10:02:50.643448Z","iopub.status.idle":"2022-07-09T10:02:50.659442Z","shell.execute_reply.started":"2022-07-09T10:02:50.643400Z","shell.execute_reply":"2022-07-09T10:02:50.658016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_confusion_matrix(model,scaled_df_test,res)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:02:50.660601Z","iopub.execute_input":"2022-07-09T10:02:50.660908Z","iopub.status.idle":"2022-07-09T10:02:50.882601Z","shell.execute_reply.started":"2022-07-09T10:02:50.660880Z","shell.execute_reply":"2022-07-09T10:02:50.881427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"my_pred = {\"PassengerId\": result['PassengerId'], \"Survived\": predictions}\noutput = pd.DataFrame(my_pred)\n\noutput.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:18:46.224722Z","iopub.execute_input":"2022-07-09T10:18:46.225125Z","iopub.status.idle":"2022-07-09T10:18:46.237991Z","shell.execute_reply.started":"2022-07-09T10:18:46.225095Z","shell.execute_reply":"2022-07-09T10:18:46.236763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output.to_csv(\"output_data.csv\", index= False)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:19:09.338282Z","iopub.execute_input":"2022-07-09T10:19:09.338716Z","iopub.status.idle":"2022-07-09T10:19:09.349120Z","shell.execute_reply.started":"2022-07-09T10:19:09.338683Z","shell.execute_reply":"2022-07-09T10:19:09.348187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}