{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Initial assumptions\n\nAs soon as I read the problem, I knew I was going to use **Logistic Regression** as I find it the most appropriate algorithm from the ones I already know.\n\nSince the shape of the function is linear, I was sure that eventually *the model would underfit the data*. Let's see if my initial assumption was correct:","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport tensorflow as tf\n\nfrom sklearn.preprocessing import StandardScaler\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense\nfrom tensorflow.keras.activations import linear, relu, sigmoid\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T15:12:04.613678Z","iopub.execute_input":"2022-08-10T15:12:04.614184Z","iopub.status.idle":"2022-08-10T15:12:12.317299Z","shell.execute_reply.started":"2022-08-10T15:12:04.614067Z","shell.execute_reply":"2022-08-10T15:12:12.315398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/titanic/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:12:12.319412Z","iopub.execute_input":"2022-08-10T15:12:12.320368Z","iopub.status.idle":"2022-08-10T15:12:12.351659Z","shell.execute_reply.started":"2022-08-10T15:12:12.320326Z","shell.execute_reply":"2022-08-10T15:12:12.350549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Seperate training data into two different data sets","metadata":{}},{"cell_type":"code","source":"df_train = train_df.iloc[:712, :] # 80% of training data set\ndf_test = train_df.iloc[712:, :] # 20% of training data set","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:12:12.353397Z","iopub.execute_input":"2022-08-10T15:12:12.353761Z","iopub.status.idle":"2022-08-10T15:12:12.359808Z","shell.execute_reply.started":"2022-08-10T15:12:12.353729Z","shell.execute_reply":"2022-08-10T15:12:12.358480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"age_mean = df_train['Age'].mean()\nfeatures = ['Pclass', 'Sex', 'Age', 'Fare']\nscaler = StandardScaler()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:12:12.362210Z","iopub.execute_input":"2022-08-10T15:12:12.363361Z","iopub.status.idle":"2022-08-10T15:12:12.377264Z","shell.execute_reply.started":"2022-08-10T15:12:12.363325Z","shell.execute_reply":"2022-08-10T15:12:12.376184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare train partition","metadata":{}},{"cell_type":"code","source":"df_train['Age'] = df_train['Age'].fillna(age_mean)\ndf_train['Sex'] = df_train['Sex'].map({'female': 0, 'male': 1})\nX_train = scaler.fit_transform(df_train[features].values)\ny_train = df_train['Survived'].values","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:12:12.380700Z","iopub.execute_input":"2022-08-10T15:12:12.381592Z","iopub.status.idle":"2022-08-10T15:12:12.399817Z","shell.execute_reply.started":"2022-08-10T15:12:12.381555Z","shell.execute_reply":"2022-08-10T15:12:12.399048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prepare test partition","metadata":{}},{"cell_type":"code","source":"df_test['Age'] = df_test['Age'].fillna(age_mean)\ndf_test['Sex'] = df_test['Sex'].map({'female': 0, 'male': 1})\nX_test = scaler.transform(df_test[features].values)\ny_test = df_test['Survived'].values","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:12:12.401105Z","iopub.execute_input":"2022-08-10T15:12:12.401593Z","iopub.status.idle":"2022-08-10T15:12:12.411628Z","shell.execute_reply.started":"2022-08-10T15:12:12.401563Z","shell.execute_reply":"2022-08-10T15:12:12.410571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train model","metadata":{}},{"cell_type":"code","source":"model = Sequential(\n    [               \n        tf.keras.Input(shape=(4,)),\n        tf.keras.layers.Dense(1, activation='sigmoid')\n    ],\n    name = \"simple_model\" \n)\n\nmodel.compile(\n    loss=tf.keras.losses.BinaryCrossentropy(),\n    optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n)\n\nmodel.fit(\n    X_train, y_train,\n    epochs=100\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:12:12.412826Z","iopub.execute_input":"2022-08-10T15:12:12.413503Z","iopub.status.idle":"2022-08-10T15:12:16.036989Z","shell.execute_reply.started":"2022-08-10T15:12:12.413470Z","shell.execute_reply":"2022-08-10T15:12:16.035653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"*J train = 0.4631*","metadata":{}},{"cell_type":"markdown","source":"# Make a prediction on test partition","metadata":{}},{"cell_type":"code","source":"y_prediction = model.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:12:16.038560Z","iopub.execute_input":"2022-08-10T15:12:16.038932Z","iopub.status.idle":"2022-08-10T15:12:16.172301Z","shell.execute_reply.started":"2022-08-10T15:12:16.038901Z","shell.execute_reply":"2022-08-10T15:12:16.171015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"m = y_prediction.shape[0]\nfor i in range(m):\n    if y_prediction[i] >= 0.5:\n        y_prediction[i] = 1\n    else:\n        y_prediction[i] = 0","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:12:16.174048Z","iopub.execute_input":"2022-08-10T15:12:16.174671Z","iopub.status.idle":"2022-08-10T15:12:16.180422Z","shell.execute_reply.started":"2022-08-10T15:12:16.174638Z","shell.execute_reply":"2022-08-10T15:12:16.179531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_prediction = tf.reshape(y_prediction, 179).numpy().astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:16:03.512359Z","iopub.execute_input":"2022-08-10T15:16:03.512799Z","iopub.status.idle":"2022-08-10T15:16:03.519392Z","shell.execute_reply.started":"2022-08-10T15:16:03.512764Z","shell.execute_reply":"2022-08-10T15:16:03.517960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"J test = \", 1 - np.sum(y_prediction == y_test) / len(y_test))","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:16:08.464197Z","iopub.execute_input":"2022-08-10T15:16:08.464676Z","iopub.status.idle":"2022-08-10T15:16:08.471974Z","shell.execute_reply.started":"2022-08-10T15:16:08.464640Z","shell.execute_reply":"2022-08-10T15:16:08.470620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluate the problem data","metadata":{}},{"cell_type":"code","source":"problem_data = test_df\nproblem_data['Age'] = problem_data['Age'].fillna(age_mean)\nproblem_data['Sex'] = problem_data['Sex'].map({'female': 0, 'male': 1})\nX_problem = scaler.transform(problem_data[features].values)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:12:16.197220Z","iopub.execute_input":"2022-08-10T15:12:16.197727Z","iopub.status.idle":"2022-08-10T15:12:16.210556Z","shell.execute_reply.started":"2022-08-10T15:12:16.197695Z","shell.execute_reply":"2022-08-10T15:12:16.209617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"problem_solution = model.predict(X_problem)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:12:16.212137Z","iopub.execute_input":"2022-08-10T15:12:16.212881Z","iopub.status.idle":"2022-08-10T15:12:16.284126Z","shell.execute_reply.started":"2022-08-10T15:12:16.212823Z","shell.execute_reply":"2022-08-10T15:12:16.283218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n = problem_solution.shape[0]\nfor i in range(n):\n    if problem_solution[i] >= 0.5:\n        problem_solution[i] = 1\n    else:\n        problem_solution[i] = 0","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:12:16.285797Z","iopub.execute_input":"2022-08-10T15:12:16.286524Z","iopub.status.idle":"2022-08-10T15:12:16.294036Z","shell.execute_reply.started":"2022-08-10T15:12:16.286469Z","shell.execute_reply":"2022-08-10T15:12:16.293196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"problem_solution = tf.reshape(problem_solution, 418).numpy().astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:12:16.295260Z","iopub.execute_input":"2022-08-10T15:12:16.295762Z","iopub.status.idle":"2022-08-10T15:12:16.306405Z","shell.execute_reply.started":"2022-08-10T15:12:16.295731Z","shell.execute_reply":"2022-08-10T15:12:16.305193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'PassengerId': test_df.PassengerId, 'Survived': problem_solution})\noutput.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T15:12:16.308381Z","iopub.execute_input":"2022-08-10T15:12:16.308979Z","iopub.status.idle":"2022-08-10T15:12:16.322364Z","shell.execute_reply.started":"2022-08-10T15:12:16.308943Z","shell.execute_reply":"2022-08-10T15:12:16.321269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Remember my first assumption? \n\nLet me point out the errors again:\n\n* J train = 0.4631\n* J test = 0.1955\n  \nSo eventually my model **underfits** as initially thought. It is surprising, though, that even though the great train error, the model still manages to generalize pretty well.","metadata":{}}]}