{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"\n\n","metadata":{}},{"cell_type":"markdown","source":"# Classification with Logistic Regression using Numpy\nAuthor: jvachier <br>\nCreation date: July 2022 <br>\nPublication date: August 2022 <br>\n\nMy goal is to classify passengers who survived the sinking of the Titanic or not, using Logistic Regression with Numpy. The training set (train.csv) contains $891$ labelled passengers and the testing set (test.csv) contains $418$ non-labelled passengers. ","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport random\nfrom matplotlib import pyplot as plt\nimport seaborn as sns\nimport tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:39.572379Z","iopub.execute_input":"2022-08-03T16:43:39.572841Z","iopub.status.idle":"2022-08-03T16:43:39.579111Z","shell.execute_reply.started":"2022-08-03T16:43:39.572750Z","shell.execute_reply":"2022-08-03T16:43:39.577920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Functions ##","metadata":{}},{"cell_type":"code","source":"## Sigmoid ##\ndef Sigmoid(x):\n    return 1.0/(1.0 + np.exp(-x))\n## Cost Function ##\ndef cost_logistic(y_data,y_prediction):\n    return np.sum((-y_data * np.log(y_prediction) - (1.0 - y_data) * np.log(1.0 - y_prediction))/y_data.size)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:39.581886Z","iopub.execute_input":"2022-08-03T16:43:39.582718Z","iopub.status.idle":"2022-08-03T16:43:39.593137Z","shell.execute_reply.started":"2022-08-03T16:43:39.582672Z","shell.execute_reply":"2022-08-03T16:43:39.591946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Logistic Regression","metadata":{}},{"cell_type":"code","source":"def Logistic_Regression(y_train,features,iterations,learning):\n    iterations = iterations\n    learning   = learning\n    \n    bias   = 0.0\n    weight = np.zeros((3,1))\n    \n    \n    accuracy_history = []\n    Cost_history     = []\n    \n    for iter in range(iterations):\n        prediction = bias + features @ weight\n        y_prediction = Sigmoid(prediction) \n        \n        ## Backward propagation ##\n        weight = weight - learning * (features.T @ (y_prediction - y_train))\n        bias   = bias - learning * (np.sum((y_prediction - y_train))/len(y_prediction - y_train))\n        \n        ## Accurary ##\n        \n        proba = [1 if p > 0.5 else 0 for p in y_prediction]\n        proba = np.array(proba)\n        \n        metric = tf.keras.metrics.BinaryAccuracy(threshold = 0.5)\n        metric.update_state(proba,y_train)\n\n        \n        accuracy_history.append(metric.result().numpy())\n        Cost_history.append(cost_logistic(y_train,y_prediction))\n        \n    \n    return Cost_history, accuracy_history, weight, bias","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:39.594721Z","iopub.execute_input":"2022-08-03T16:43:39.596058Z","iopub.status.idle":"2022-08-03T16:43:39.607823Z","shell.execute_reply.started":"2022-08-03T16:43:39.596017Z","shell.execute_reply":"2022-08-03T16:43:39.606778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Importation","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"../input/titanic/train.csv\")\ndf_test = pd.read_csv(\"../input/titanic/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:39.609535Z","iopub.execute_input":"2022-08-03T16:43:39.610214Z","iopub.status.idle":"2022-08-03T16:43:39.636478Z","shell.execute_reply.started":"2022-08-03T16:43:39.610176Z","shell.execute_reply":"2022-08-03T16:43:39.635397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:39.639582Z","iopub.execute_input":"2022-08-03T16:43:39.640388Z","iopub.status.idle":"2022-08-03T16:43:39.662449Z","shell.execute_reply.started":"2022-08-03T16:43:39.640345Z","shell.execute_reply":"2022-08-03T16:43:39.661396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:39.663551Z","iopub.execute_input":"2022-08-03T16:43:39.663860Z","iopub.status.idle":"2022-08-03T16:43:39.672556Z","shell.execute_reply.started":"2022-08-03T16:43:39.663832Z","shell.execute_reply":"2022-08-03T16:43:39.671047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.columns","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:39.674535Z","iopub.execute_input":"2022-08-03T16:43:39.675222Z","iopub.status.idle":"2022-08-03T16:43:39.683157Z","shell.execute_reply.started":"2022-08-03T16:43:39.675179Z","shell.execute_reply":"2022-08-03T16:43:39.682004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Select useful columns: 'PassengerId', 'Survived', 'Pclass', 'Sex', 'Age'","metadata":{}},{"cell_type":"code","source":"df_train_selected = df_train[['PassengerId', 'Survived', 'Pclass', 'Sex', 'Age']]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:39.685013Z","iopub.execute_input":"2022-08-03T16:43:39.685783Z","iopub.status.idle":"2022-08-03T16:43:39.694311Z","shell.execute_reply.started":"2022-08-03T16:43:39.685740Z","shell.execute_reply":"2022-08-03T16:43:39.693076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_selected.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:39.696067Z","iopub.execute_input":"2022-08-03T16:43:39.696745Z","iopub.status.idle":"2022-08-03T16:43:39.709580Z","shell.execute_reply.started":"2022-08-03T16:43:39.696701Z","shell.execute_reply":"2022-08-03T16:43:39.708382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Replace the missing value in Age using a Gaussian distribution with the mean and variance of Age","metadata":{}},{"cell_type":"code","source":"f = lambda row: np.int(random.gauss(df_train_selected['Age'].mean(),np.sqrt(df_train_selected['Age'].std())))\nm = df_train_selected['Age'].isna()\ndf_train_selected['Age'].loc[m] = df_train_selected['Age'].loc[m].apply(f)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:39.825641Z","iopub.execute_input":"2022-08-03T16:43:39.826893Z","iopub.status.idle":"2022-08-03T16:43:39.896985Z","shell.execute_reply.started":"2022-08-03T16:43:39.826841Z","shell.execute_reply":"2022-08-03T16:43:39.895804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_selected.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:39.899689Z","iopub.execute_input":"2022-08-03T16:43:39.900490Z","iopub.status.idle":"2022-08-03T16:43:39.911971Z","shell.execute_reply.started":"2022-08-03T16:43:39.900443Z","shell.execute_reply":"2022-08-03T16:43:39.910806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(data=df_train_selected, x=\"Sex\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:39.913974Z","iopub.execute_input":"2022-08-03T16:43:39.914341Z","iopub.status.idle":"2022-08-03T16:43:40.100589Z","shell.execute_reply.started":"2022-08-03T16:43:39.914310Z","shell.execute_reply":"2022-08-03T16:43:40.099709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(data=df_train_selected, x=\"Age\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:40.102335Z","iopub.execute_input":"2022-08-03T16:43:40.103051Z","iopub.status.idle":"2022-08-03T16:43:40.366335Z","shell.execute_reply.started":"2022-08-03T16:43:40.103005Z","shell.execute_reply":"2022-08-03T16:43:40.365375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(data=df_train_selected, x=\"Pclass\", binwidth=0.5)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:40.368696Z","iopub.execute_input":"2022-08-03T16:43:40.369002Z","iopub.status.idle":"2022-08-03T16:43:40.586919Z","shell.execute_reply.started":"2022-08-03T16:43:40.368973Z","shell.execute_reply":"2022-08-03T16:43:40.585750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Replace Male by 1 and Female by 0","metadata":{}},{"cell_type":"code","source":"df_train_selected['Sex'].replace(['female','male'],[0,1],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:40.588459Z","iopub.execute_input":"2022-08-03T16:43:40.588813Z","iopub.status.idle":"2022-08-03T16:43:40.597882Z","shell.execute_reply.started":"2022-08-03T16:43:40.588780Z","shell.execute_reply":"2022-08-03T16:43:40.596563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_selected['Sex'].head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:40.601442Z","iopub.execute_input":"2022-08-03T16:43:40.602139Z","iopub.status.idle":"2022-08-03T16:43:40.609628Z","shell.execute_reply.started":"2022-08-03T16:43:40.602095Z","shell.execute_reply":"2022-08-03T16:43:40.608772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = df_train_selected.loc[0:801,'Survived'].to_numpy().reshape(-1,1)\ny_validation = df_train_selected.loc[802:891,'Survived'].to_numpy().reshape(-1,1)\nprint(y_train.shape, y_validation.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:40.610787Z","iopub.execute_input":"2022-08-03T16:43:40.611789Z","iopub.status.idle":"2022-08-03T16:43:40.621399Z","shell.execute_reply.started":"2022-08-03T16:43:40.611757Z","shell.execute_reply":"2022-08-03T16:43:40.620624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_train = df_train_selected.loc[0:801,['Pclass', 'Sex', 'Age']].to_numpy()\nfeatures_validation = df_train_selected.loc[802:891,['Pclass', 'Sex', 'Age']].to_numpy()\nprint(features_train.shape, features_validation.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:40.622812Z","iopub.execute_input":"2022-08-03T16:43:40.624544Z","iopub.status.idle":"2022-08-03T16:43:40.638975Z","shell.execute_reply.started":"2022-08-03T16:43:40.624498Z","shell.execute_reply":"2022-08-03T16:43:40.637952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training ","metadata":{}},{"cell_type":"code","source":"Cost_history, accuracy_history, weight, bias = Logistic_Regression(y_train,features_train,10000,1E-5)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:43:40.640415Z","iopub.execute_input":"2022-08-03T16:43:40.640954Z","iopub.status.idle":"2022-08-03T16:44:27.538233Z","shell.execute_reply.started":"2022-08-03T16:43:40.640914Z","shell.execute_reply":"2022-08-03T16:44:27.537137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(Cost_history)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:27.539770Z","iopub.execute_input":"2022-08-03T16:44:27.540224Z","iopub.status.idle":"2022-08-03T16:44:27.744721Z","shell.execute_reply.started":"2022-08-03T16:44:27.540181Z","shell.execute_reply":"2022-08-03T16:44:27.743616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(accuracy_history)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:27.747204Z","iopub.execute_input":"2022-08-03T16:44:27.748165Z","iopub.status.idle":"2022-08-03T16:44:27.880462Z","shell.execute_reply.started":"2022-08-03T16:44:27.748128Z","shell.execute_reply":"2022-08-03T16:44:27.879479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_train = bias + features_train @ weight\ny_prediction_train = Sigmoid(prediction_train)\nproba_train = [1 if p > 0.5 else 0 for p in y_prediction_train]\nproba_train = np.array(proba_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:27.884009Z","iopub.execute_input":"2022-08-03T16:44:27.884387Z","iopub.status.idle":"2022-08-03T16:44:27.893060Z","shell.execute_reply.started":"2022-08-03T16:44:27.884353Z","shell.execute_reply":"2022-08-03T16:44:27.892316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(proba_train,'.')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:27.898882Z","iopub.execute_input":"2022-08-03T16:44:27.899572Z","iopub.status.idle":"2022-08-03T16:44:28.032113Z","shell.execute_reply.started":"2022-08-03T16:44:27.899538Z","shell.execute_reply":"2022-08-03T16:44:28.031286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Validation","metadata":{}},{"cell_type":"code","source":"prediction_validation = bias + features_validation @ weight\ny_prediction_validation = Sigmoid(prediction_validation)\nproba_validation = [1 if p > 0.5 else 0 for p in y_prediction_validation]\nproba_validation = np.array(proba_validation)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:28.033886Z","iopub.execute_input":"2022-08-03T16:44:28.034458Z","iopub.status.idle":"2022-08-03T16:44:28.041205Z","shell.execute_reply.started":"2022-08-03T16:44:28.034412Z","shell.execute_reply":"2022-08-03T16:44:28.039970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(proba_validation,'.')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:28.043026Z","iopub.execute_input":"2022-08-03T16:44:28.043643Z","iopub.status.idle":"2022-08-03T16:44:28.223943Z","shell.execute_reply.started":"2022-08-03T16:44:28.043601Z","shell.execute_reply":"2022-08-03T16:44:28.222779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metric = tf.keras.metrics.BinaryAccuracy(threshold = 0.5)\nmetric.update_state(proba_validation,y_validation)\nprint(metric.result().numpy())","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:28.225307Z","iopub.execute_input":"2022-08-03T16:44:28.225731Z","iopub.status.idle":"2022-08-03T16:44:28.238748Z","shell.execute_reply.started":"2022-08-03T16:44:28.225699Z","shell.execute_reply":"2022-08-03T16:44:28.237167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Test","metadata":{}},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:28.240628Z","iopub.execute_input":"2022-08-03T16:44:28.241037Z","iopub.status.idle":"2022-08-03T16:44:28.258585Z","shell.execute_reply.started":"2022-08-03T16:44:28.241000Z","shell.execute_reply":"2022-08-03T16:44:28.257354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_selected = df_test[['PassengerId', 'Pclass', 'Sex', 'Age']]","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:28.259681Z","iopub.execute_input":"2022-08-03T16:44:28.260137Z","iopub.status.idle":"2022-08-03T16:44:28.268140Z","shell.execute_reply.started":"2022-08-03T16:44:28.260104Z","shell.execute_reply":"2022-08-03T16:44:28.267164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_selected.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:28.270094Z","iopub.execute_input":"2022-08-03T16:44:28.270496Z","iopub.status.idle":"2022-08-03T16:44:28.283518Z","shell.execute_reply.started":"2022-08-03T16:44:28.270456Z","shell.execute_reply":"2022-08-03T16:44:28.282654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f = lambda row: np.int(random.gauss(df_test_selected['Age'].mean(),np.sqrt(df_test_selected['Age'].std())))\nm = df_test_selected['Age'].isna()\ndf_test_selected['Age'].loc[m] = df_test_selected['Age'].loc[m].apply(f)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:28.284844Z","iopub.execute_input":"2022-08-03T16:44:28.285154Z","iopub.status.idle":"2022-08-03T16:44:28.319955Z","shell.execute_reply.started":"2022-08-03T16:44:28.285126Z","shell.execute_reply":"2022-08-03T16:44:28.318778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_selected.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:28.322003Z","iopub.execute_input":"2022-08-03T16:44:28.322391Z","iopub.status.idle":"2022-08-03T16:44:28.333049Z","shell.execute_reply.started":"2022-08-03T16:44:28.322359Z","shell.execute_reply":"2022-08-03T16:44:28.331773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(data=df_test_selected, x=\"Sex\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:28.334497Z","iopub.execute_input":"2022-08-03T16:44:28.335033Z","iopub.status.idle":"2022-08-03T16:44:28.519574Z","shell.execute_reply.started":"2022-08-03T16:44:28.335002Z","shell.execute_reply":"2022-08-03T16:44:28.518315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(data=df_test_selected, x=\"Age\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:28.520765Z","iopub.execute_input":"2022-08-03T16:44:28.521082Z","iopub.status.idle":"2022-08-03T16:44:28.773224Z","shell.execute_reply.started":"2022-08-03T16:44:28.521053Z","shell.execute_reply":"2022-08-03T16:44:28.772001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(data=df_test_selected, x=\"Pclass\", binwidth=0.5)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:28.776893Z","iopub.execute_input":"2022-08-03T16:44:28.777298Z","iopub.status.idle":"2022-08-03T16:44:28.999003Z","shell.execute_reply.started":"2022-08-03T16:44:28.777255Z","shell.execute_reply":"2022-08-03T16:44:28.997902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Replace Male by 1 and Female by 0","metadata":{}},{"cell_type":"code","source":"df_test_selected['Sex'].replace(['female','male'],[0,1],inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:29.000519Z","iopub.execute_input":"2022-08-03T16:44:29.000872Z","iopub.status.idle":"2022-08-03T16:44:29.007741Z","shell.execute_reply.started":"2022-08-03T16:44:29.000840Z","shell.execute_reply":"2022-08-03T16:44:29.006908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_selected['Sex'].head()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:29.008754Z","iopub.execute_input":"2022-08-03T16:44:29.009511Z","iopub.status.idle":"2022-08-03T16:44:29.022470Z","shell.execute_reply.started":"2022-08-03T16:44:29.009478Z","shell.execute_reply":"2022-08-03T16:44:29.021435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features_test = df_test_selected[['Pclass', 'Sex', 'Age']].to_numpy()\nprint(features_test.shape)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:29.024522Z","iopub.execute_input":"2022-08-03T16:44:29.025354Z","iopub.status.idle":"2022-08-03T16:44:29.035475Z","shell.execute_reply.started":"2022-08-03T16:44:29.025309Z","shell.execute_reply":"2022-08-03T16:44:29.034444Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction_test = bias + features_test @ weight\ny_prediction_test = Sigmoid(prediction_test)\nproba_test = [1 if p > 0.5 else 0 for p in y_prediction_test]\nproba_test = np.array(proba_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:29.036964Z","iopub.execute_input":"2022-08-03T16:44:29.037358Z","iopub.status.idle":"2022-08-03T16:44:29.046398Z","shell.execute_reply.started":"2022-08-03T16:44:29.037325Z","shell.execute_reply":"2022-08-03T16:44:29.045560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_prediction = pd.DataFrame(proba_test, columns = ['Survived'])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:29.048020Z","iopub.execute_input":"2022-08-03T16:44:29.048830Z","iopub.status.idle":"2022-08-03T16:44:29.063321Z","shell.execute_reply.started":"2022-08-03T16:44:29.048794Z","shell.execute_reply":"2022-08-03T16:44:29.061660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = pd.concat([df_test_selected['PassengerId'], df_prediction], axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:29.064947Z","iopub.execute_input":"2022-08-03T16:44:29.066076Z","iopub.status.idle":"2022-08-03T16:44:29.074804Z","shell.execute_reply.started":"2022-08-03T16:44:29.066023Z","shell.execute_reply":"2022-08-03T16:44:29.073415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results.to_csv('prediction_titanic.csv',index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T16:44:29.076341Z","iopub.execute_input":"2022-08-03T16:44:29.076681Z","iopub.status.idle":"2022-08-03T16:44:29.085884Z","shell.execute_reply.started":"2022-08-03T16:44:29.076651Z","shell.execute_reply":"2022-08-03T16:44:29.085025Z"},"trusted":true},"execution_count":null,"outputs":[]}]}