{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-10T09:49:37.712097Z","iopub.execute_input":"2022-05-10T09:49:37.712452Z","iopub.status.idle":"2022-05-10T09:49:37.738925Z","shell.execute_reply.started":"2022-05-10T09:49:37.712353Z","shell.execute_reply":"2022-05-10T09:49:37.738214Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/titanic/train.csv')\ndf_test = pd.read_csv('/kaggle/input/titanic/test.csv') \ntest_ids = df_test[\"PassengerId\"]","metadata":{"execution":{"iopub.status.busy":"2022-05-10T09:50:34.572375Z","iopub.execute_input":"2022-05-10T09:50:34.572851Z","iopub.status.idle":"2022-05-10T09:50:34.602873Z","shell.execute_reply.started":"2022-05-10T09:50:34.572802Z","shell.execute_reply":"2022-05-10T09:50:34.602242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean(df):\n    df=df.drop([\"Ticket\", \"Cabin\", \"Name\", \"PassengerId\"],axis=1)\n    \n    columns = [\"SibSp\", \"Parch\", \"Fare\", \"Age\"]\n    for columns in columns:\n        df[columns].fillna(df[columns].median(), inplace=True)\n        \n    df.Embarked.fillna(\"U\", inplace = True)\n    return df\n\ndf=clean(df)\ndf_test = clean(df_test)","metadata":{"execution":{"iopub.status.busy":"2022-05-10T09:50:52.098811Z","iopub.execute_input":"2022-05-10T09:50:52.099315Z","iopub.status.idle":"2022-05-10T09:50:52.118793Z","shell.execute_reply.started":"2022-05-10T09:50:52.099256Z","shell.execute_reply":"2022-05-10T09:50:52.118147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\nle = preprocessing.LabelEncoder()\n\ncolumns = ['Sex', 'Embarked']\n\nfor columns in columns:\n    df[columns] = le.fit_transform(df[columns])\n    df_test[columns] = le.transform(df_test[columns])\n    print(le.classes_)\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-10T09:51:09.85663Z","iopub.execute_input":"2022-05-10T09:51:09.857034Z","iopub.status.idle":"2022-05-10T09:51:10.895747Z","shell.execute_reply.started":"2022-05-10T09:51:09.857004Z","shell.execute_reply":"2022-05-10T09:51:10.89482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\npredictors = ['Pclass','Sex', 'Age', 'SibSp', 'Parch', 'Fare', 'Embarked']\nresponse = 'Survived'\n\nX=df[predictors].values\ny=df[response]","metadata":{"execution":{"iopub.status.busy":"2022-05-10T09:51:39.295206Z","iopub.execute_input":"2022-05-10T09:51:39.295491Z","iopub.status.idle":"2022-05-10T09:51:39.541409Z","shell.execute_reply.started":"2022-05-10T09:51:39.295461Z","shell.execute_reply":"2022-05-10T09:51:39.540707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"RF_model=RandomForestClassifier(n_estimators=200, max_depth=4, oob_score= True, random_state=42)\nRF_model.fit(X,y)","metadata":{"execution":{"iopub.status.busy":"2022-05-10T10:33:48.215334Z","iopub.execute_input":"2022-05-10T10:33:48.215599Z","iopub.status.idle":"2022-05-10T10:33:48.584352Z","shell.execute_reply.started":"2022-05-10T10:33:48.215572Z","shell.execute_reply":"2022-05-10T10:33:48.583522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_prediction = RF_model.predict(df_test)\nsubmission_prediction","metadata":{"execution":{"iopub.status.busy":"2022-05-10T10:33:53.908757Z","iopub.execute_input":"2022-05-10T10:33:53.909376Z","iopub.status.idle":"2022-05-10T10:33:53.947227Z","shell.execute_reply.started":"2022-05-10T10:33:53.909336Z","shell.execute_reply":"2022-05-10T10:33:53.946275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe = pd.DataFrame({\"PassengerId\":test_ids.values,\n                         \"Survived\": submission_prediction})","metadata":{"execution":{"iopub.status.busy":"2022-05-10T10:33:57.382366Z","iopub.execute_input":"2022-05-10T10:33:57.382650Z","iopub.status.idle":"2022-05-10T10:33:57.387540Z","shell.execute_reply.started":"2022-05-10T10:33:57.382620Z","shell.execute_reply":"2022-05-10T10:33:57.386540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataframe.to_csv(\"submission.csv\", index = False)","metadata":{"execution":{"iopub.status.busy":"2022-05-10T10:34:02.765768Z","iopub.execute_input":"2022-05-10T10:34:02.766621Z","iopub.status.idle":"2022-05-10T10:34:02.771913Z","shell.execute_reply.started":"2022-05-10T10:34:02.766579Z","shell.execute_reply":"2022-05-10T10:34:02.771158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}