{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-02T20:48:10.667712Z","iopub.execute_input":"2022-08-02T20:48:10.668836Z","iopub.status.idle":"2022-08-02T20:48:10.695983Z","shell.execute_reply.started":"2022-08-02T20:48:10.668699Z","shell.execute_reply":"2022-08-02T20:48:10.694759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:10.717538Z","iopub.execute_input":"2022-08-02T20:48:10.717918Z","iopub.status.idle":"2022-08-02T20:48:10.722924Z","shell.execute_reply.started":"2022-08-02T20:48:10.717883Z","shell.execute_reply":"2022-08-02T20:48:10.721670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Loading Data","metadata":{}},{"cell_type":"code","source":"train_data=pd.read_csv('../input/titanic/train.csv')\ntest_data=pd.read_csv('../input/titanic/test.csv')\nsubmission_data=pd.read_csv('../input/titanic/gender_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:10.772632Z","iopub.execute_input":"2022-08-02T20:48:10.773578Z","iopub.status.idle":"2022-08-02T20:48:10.808680Z","shell.execute_reply.started":"2022-08-02T20:48:10.773527Z","shell.execute_reply":"2022-08-02T20:48:10.807535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Table display","metadata":{}},{"cell_type":"code","source":"train_data.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:10.820031Z","iopub.execute_input":"2022-08-02T20:48:10.822260Z","iopub.status.idle":"2022-08-02T20:48:10.846015Z","shell.execute_reply.started":"2022-08-02T20:48:10.822222Z","shell.execute_reply":"2022-08-02T20:48:10.845000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:10.859013Z","iopub.execute_input":"2022-08-02T20:48:10.860312Z","iopub.status.idle":"2022-08-02T20:48:10.881637Z","shell.execute_reply.started":"2022-08-02T20:48:10.860269Z","shell.execute_reply":"2022-08-02T20:48:10.880737Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check Nan vaule count","metadata":{}},{"cell_type":"code","source":"train_data.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:10.901006Z","iopub.execute_input":"2022-08-02T20:48:10.901653Z","iopub.status.idle":"2022-08-02T20:48:10.911929Z","shell.execute_reply.started":"2022-08-02T20:48:10.901620Z","shell.execute_reply":"2022-08-02T20:48:10.910696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:10.940149Z","iopub.execute_input":"2022-08-02T20:48:10.940729Z","iopub.status.idle":"2022-08-02T20:48:10.948831Z","shell.execute_reply.started":"2022-08-02T20:48:10.940697Z","shell.execute_reply":"2022-08-02T20:48:10.947756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Define data cleaning function","metadata":{}},{"cell_type":"code","source":"def data_cleaning(df:'pd.DataFrame'):\n  df['Age']=df['Age'].fillna(df.Age.mean()) # fill na value with mean of the Age\n  df['Fare']=df['Fare'].fillna(df.Fare.mean())# fill na value with mean of the Fare\n  one_hot_sex = pd.get_dummies(df.Sex) # one hot category colume\n  one_hot_embarked = pd.get_dummies(df.Embarked) # one hot category colume\n  df.drop(columns=['PassengerId','Cabin','Ticket','Name','Sex','Embarked'],inplace=True) # Drop columes that is not suitable for training or is one hot labeled\n  df = pd.concat([df, one_hot_sex, one_hot_embarked], axis=1) # colume-wise concatenate with one hot label\n\n  return df","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:10.999246Z","iopub.execute_input":"2022-08-02T20:48:11.000122Z","iopub.status.idle":"2022-08-02T20:48:11.008799Z","shell.execute_reply.started":"2022-08-02T20:48:11.000059Z","shell.execute_reply":"2022-08-02T20:48:11.007685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check the result of Data cleaning","metadata":{}},{"cell_type":"code","source":"test_data=data_cleaning(test_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:11.038793Z","iopub.execute_input":"2022-08-02T20:48:11.039789Z","iopub.status.idle":"2022-08-02T20:48:11.050368Z","shell.execute_reply.started":"2022-08-02T20:48:11.039751Z","shell.execute_reply":"2022-08-02T20:48:11.049484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:11.076056Z","iopub.execute_input":"2022-08-02T20:48:11.076636Z","iopub.status.idle":"2022-08-02T20:48:11.089508Z","shell.execute_reply.started":"2022-08-02T20:48:11.076604Z","shell.execute_reply":"2022-08-02T20:48:11.088520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data=data_cleaning(train_data)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:11.141748Z","iopub.execute_input":"2022-08-02T20:48:11.142206Z","iopub.status.idle":"2022-08-02T20:48:11.153145Z","shell.execute_reply.started":"2022-08-02T20:48:11.142166Z","shell.execute_reply":"2022-08-02T20:48:11.152147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:11.177508Z","iopub.execute_input":"2022-08-02T20:48:11.178028Z","iopub.status.idle":"2022-08-02T20:48:11.192675Z","shell.execute_reply.started":"2022-08-02T20:48:11.177997Z","shell.execute_reply":"2022-08-02T20:48:11.191704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Split train and test set","metadata":{}},{"cell_type":"code","source":"train_data=train_data.to_numpy()\nX_train,Y_train=train_data[:,1::],train_data[:,0]\nX_test,Y_test=test_data.to_numpy(),submission_data['Survived'].to_numpy()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:11.210261Z","iopub.execute_input":"2022-08-02T20:48:11.210585Z","iopub.status.idle":"2022-08-02T20:48:11.217413Z","shell.execute_reply.started":"2022-08-02T20:48:11.210556Z","shell.execute_reply":"2022-08-02T20:48:11.216309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Construct deep learing network","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.layers import *\nfrom tensorflow.keras import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint,EarlyStopping","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:11.242638Z","iopub.execute_input":"2022-08-02T20:48:11.243522Z","iopub.status.idle":"2022-08-02T20:48:17.236205Z","shell.execute_reply.started":"2022-08-02T20:48:11.243485Z","shell.execute_reply":"2022-08-02T20:48:17.235238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"inputs=Input(shape=(X_train.shape[1],))\nx = Dense(128, activation=\"relu\")(inputs)\nx = Dense(64, activation=\"relu\")(x)\noutputs = Dense(1, activation=\"sigmoid\")(x)\n\nmodel=Model(inputs=inputs, outputs=outputs)\n\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:17.237919Z","iopub.execute_input":"2022-08-02T20:48:17.238516Z","iopub.status.idle":"2022-08-02T20:48:17.354487Z","shell.execute_reply.started":"2022-08-02T20:48:17.238483Z","shell.execute_reply":"2022-08-02T20:48:17.352426Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:17.355863Z","iopub.execute_input":"2022-08-02T20:48:17.356226Z","iopub.status.idle":"2022-08-02T20:48:17.362886Z","shell.execute_reply.started":"2022-08-02T20:48:17.356195Z","shell.execute_reply":"2022-08-02T20:48:17.361752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Training","metadata":{}},{"cell_type":"code","source":"save_path = '/checkpoint'\ncheckpoint = ModelCheckpoint(\n    filepath=save_path,\n    monitor='val_loss',\n    save_best_only=True) #define checkpoint to save the best model\ncallback = EarlyStopping(monitor='val_loss', patience=5) # define early stop for proventing over fitting\nhistory = model.fit(X_train, Y_train, epochs=250, verbose=0, batch_size=32, validation_split=0.2 ,callbacks=[checkpoint,callback])","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:17.366318Z","iopub.execute_input":"2022-08-02T20:48:17.367105Z","iopub.status.idle":"2022-08-02T20:48:27.598576Z","shell.execute_reply.started":"2022-08-02T20:48:17.367041Z","shell.execute_reply":"2022-08-02T20:48:27.597418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.ylabel('loss')\nplt.xlabel('Epochs')\nplt.legend(['training', 'validation'], loc='upper left')\nplt.show()\n\nplt.plot(history.history['accuracy'])\nplt.plot(history.history['val_accuracy'])\nplt.ylabel('accuracy')\nplt.xlabel('Epochs')\nplt.legend(['training', 'validation'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:27.600034Z","iopub.execute_input":"2022-08-02T20:48:27.600413Z","iopub.status.idle":"2022-08-02T20:48:28.012558Z","shell.execute_reply.started":"2022-08-02T20:48:27.600380Z","shell.execute_reply":"2022-08-02T20:48:28.011278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Evaluate the model","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.models import load_model \nmodel=load_model(save_path) # import the best model","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:28.014044Z","iopub.execute_input":"2022-08-02T20:48:28.014545Z","iopub.status.idle":"2022-08-02T20:48:28.266763Z","shell.execute_reply.started":"2022-08-02T20:48:28.014511Z","shell.execute_reply":"2022-08-02T20:48:28.265650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"score=model.evaluate(X_test,Y_test)\nprint('test_loss:{:.3f}'.format(score[0]))\nprint('test_acc:{:.3f}'.format(score[1]))","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:28.269054Z","iopub.execute_input":"2022-08-02T20:48:28.270134Z","iopub.status.idle":"2022-08-02T20:48:28.488693Z","shell.execute_reply.started":"2022-08-02T20:48:28.270068Z","shell.execute_reply":"2022-08-02T20:48:28.487591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = np.asarray(X_test).astype('float32')\nY_pred=[1 if pred >0.5 else 0 for pred in model.predict(x)]","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:28.490137Z","iopub.execute_input":"2022-08-02T20:48:28.490623Z","iopub.status.idle":"2022-08-02T20:48:28.638889Z","shell.execute_reply.started":"2022-08-02T20:48:28.490574Z","shell.execute_reply":"2022-08-02T20:48:28.637952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Plotting confusion_matrix","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix, ConfusionMatrixDisplay","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:28.640305Z","iopub.execute_input":"2022-08-02T20:48:28.640653Z","iopub.status.idle":"2022-08-02T20:48:28.953910Z","shell.execute_reply.started":"2022-08-02T20:48:28.640621Z","shell.execute_reply":"2022-08-02T20:48:28.952848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm = confusion_matrix(Y_test, Y_pred)\ndisp = ConfusionMatrixDisplay(confusion_matrix=cm)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:28.956825Z","iopub.execute_input":"2022-08-02T20:48:28.957195Z","iopub.status.idle":"2022-08-02T20:48:28.964232Z","shell.execute_reply.started":"2022-08-02T20:48:28.957162Z","shell.execute_reply":"2022-08-02T20:48:28.962935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"disp.plot()","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:28.965786Z","iopub.execute_input":"2022-08-02T20:48:28.966694Z","iopub.status.idle":"2022-08-02T20:48:29.181185Z","shell.execute_reply.started":"2022-08-02T20:48:28.966661Z","shell.execute_reply":"2022-08-02T20:48:29.180229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = {\n    'PassengerId': submission_data['PassengerId'],\n    'Survived': Y_pred\n}\nsubmission = pd.DataFrame(result)\nsubmission.to_csv('./submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-02T20:48:29.182651Z","iopub.execute_input":"2022-08-02T20:48:29.182982Z","iopub.status.idle":"2022-08-02T20:48:29.192528Z","shell.execute_reply.started":"2022-08-02T20:48:29.182952Z","shell.execute_reply":"2022-08-02T20:48:29.191601Z"},"trusted":true},"execution_count":null,"outputs":[]}]}