{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# =========================== IMPORT ==========================================\nimport pandas as pd\nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2022-08-14T12:56:45.528520Z","iopub.execute_input":"2022-08-14T12:56:45.528887Z","iopub.status.idle":"2022-08-14T12:56:45.533955Z","shell.execute_reply.started":"2022-08-14T12:56:45.528857Z","shell.execute_reply":"2022-08-14T12:56:45.533115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-08-14T12:56:45.540704Z","iopub.execute_input":"2022-08-14T12:56:45.541281Z","iopub.status.idle":"2022-08-14T12:56:45.565228Z","shell.execute_reply.started":"2022-08-14T12:56:45.541245Z","shell.execute_reply":"2022-08-14T12:56:45.563904Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =========================== VISUALISATION ===================================\ndata_train = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ndata_train.sample(5)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T12:56:45.572412Z","iopub.execute_input":"2022-08-14T12:56:45.573498Z","iopub.status.idle":"2022-08-14T12:56:45.598187Z","shell.execute_reply.started":"2022-08-14T12:56:45.573461Z","shell.execute_reply":"2022-08-14T12:56:45.597105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =============================================================================\n# =========================== BUILD DATA train/test ===========================\n# =============================================================================\n\n# =============== data_train\ncolumn_input = [\"Sex\",\"Age\",\"Embarked\",\"Pclass\"]\ncolumn_class = [\"Survived\"]\ncolumn_keep = column_input+column_class\ndata_train = data_train.dropna(subset=column_keep)\n\ndata_train_Input = data_train[column_input]\ndata_train_Input = pd.get_dummies(data_train_Input)\ndata_train_Class = data_train[column_class]\n\n# =============== data_test\ndata_test = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ndata_test_Input = pd.get_dummies(data_test[column_input])\ndata_test_Input = data_test_Input.fillna(data_test_Input.mean())","metadata":{"execution":{"iopub.status.busy":"2022-08-14T12:56:45.663551Z","iopub.execute_input":"2022-08-14T12:56:45.663953Z","iopub.status.idle":"2022-08-14T12:56:45.709001Z","shell.execute_reply.started":"2022-08-14T12:56:45.663918Z","shell.execute_reply":"2022-08-14T12:56:45.708075Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# =============================================================================\n# ================================= BUILD MODEL ===============================\n# =============================================================================\nmodel = RandomForestClassifier(n_estimators=100, max_depth=5, random_state=1)\nmodel.fit(data_train_Input, data_train_Class)\npredictions = model.predict(data_test_Input)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T12:56:45.711077Z","iopub.execute_input":"2022-08-14T12:56:45.711734Z","iopub.status.idle":"2022-08-14T12:56:45.905337Z","shell.execute_reply.started":"2022-08-14T12:56:45.711699Z","shell.execute_reply":"2022-08-14T12:56:45.904114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'PassengerId': data_test.PassengerId, 'Survived': predictions})\noutput.to_csv('my_submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T12:56:45.906853Z","iopub.execute_input":"2022-08-14T12:56:45.907274Z","iopub.status.idle":"2022-08-14T12:56:45.917103Z","shell.execute_reply.started":"2022-08-14T12:56:45.907239Z","shell.execute_reply":"2022-08-14T12:56:45.915946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}