{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T03:02:55.907208Z","iopub.execute_input":"2022-08-13T03:02:55.907503Z","iopub.status.idle":"2022-08-13T03:02:55.918149Z","shell.execute_reply.started":"2022-08-13T03:02:55.907481Z","shell.execute_reply":"2022-08-13T03:02:55.917021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T03:02:55.940943Z","iopub.execute_input":"2022-08-13T03:02:55.941708Z","iopub.status.idle":"2022-08-13T03:02:55.961441Z","shell.execute_reply.started":"2022-08-13T03:02:55.941633Z","shell.execute_reply":"2022-08-13T03:02:55.960571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Separate target from predictors\ny = df[\"Survived\"]\nX = df.drop(\"Survived\",axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T03:02:55.962979Z","iopub.execute_input":"2022-08-13T03:02:55.963221Z","iopub.status.idle":"2022-08-13T03:02:55.967959Z","shell.execute_reply.started":"2022-08-13T03:02:55.963199Z","shell.execute_reply":"2022-08-13T03:02:55.967254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train_full,X_valid_full, y_train_full, y_valid_full = train_test_split(X,y,test_size = 0.2, random_state = 0) ","metadata":{"execution":{"iopub.status.busy":"2022-08-13T03:02:55.973207Z","iopub.execute_input":"2022-08-13T03:02:55.974256Z","iopub.status.idle":"2022-08-13T03:02:55.982163Z","shell.execute_reply.started":"2022-08-13T03:02:55.974208Z","shell.execute_reply":"2022-08-13T03:02:55.980963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_columns = [\"Sex\",\"Embarked\"]\nnumerical_columns = [\"Pclass\",\"Age\",\"SibSp\",\"Parch\",\"Fare\"]\n\n# list of columns used\ncolumn_needed = categorical_columns + numerical_columns\n\n# assign useful data\nX_train = X_train_full[column_needed].copy()\nX_valid = X_valid_full[column_needed].copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T03:02:55.983702Z","iopub.execute_input":"2022-08-13T03:02:55.984270Z","iopub.status.idle":"2022-08-13T03:02:55.995002Z","shell.execute_reply.started":"2022-08-13T03:02:55.984246Z","shell.execute_reply":"2022-08-13T03:02:55.994070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.impute import SimpleImputer\n# preprocessing for numerical data (impute missing value)\nmissing_value_transformer = SimpleImputer(strategy = \"most_frequent\")\n\nfrom sklearn.preprocessing import OneHotEncoder\n# preprocessing for categorical data (one hot encode sex and embarked)\nencoder = OneHotEncoder(handle_unknown=\"ignore\")\n\nfrom sklearn.compose import ColumnTransformer\n# bundle preprocessing for data\npreprocessor = ColumnTransformer(\n        transformers= [\n            (\"num\", missing_value_transformer, numerical_columns),\n            (\"cat\", encoder, categorical_columns)\n        ]\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T03:02:56.008593Z","iopub.execute_input":"2022-08-13T03:02:56.009612Z","iopub.status.idle":"2022-08-13T03:02:56.014934Z","shell.execute_reply.started":"2022-08-13T03:02:56.009569Z","shell.execute_reply":"2022-08-13T03:02:56.013947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.metrics import mean_absolute_error\ndef score(n_estimators):\n    model = RandomForestClassifier(random_state = 0, max_depth = 3, n_estimators = n_estimators)\n    my_pipeline = Pipeline(steps=[\n        (\"preprocessor\", preprocessor),\n        (\"model\", model)\n    ])\n    \n    my_pipeline.fit(X_train, y_train_full)\n    preds = my_pipeline.predict(X_valid)\n    score = mean_absolute_error(y_valid_full, preds)\n    return score\n","metadata":{"execution":{"iopub.status.busy":"2022-08-13T03:02:56.017430Z","iopub.execute_input":"2022-08-13T03:02:56.018045Z","iopub.status.idle":"2022-08-13T03:02:56.030020Z","shell.execute_reply.started":"2022-08-13T03:02:56.018012Z","shell.execute_reply":"2022-08-13T03:02:56.028775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_list = [n for n in range(50,600,50)]\nans = {}\nfor i in n_list:\n    ans[i] = score(i)\n    \n    \nans","metadata":{"execution":{"iopub.status.busy":"2022-08-13T03:02:56.031892Z","iopub.execute_input":"2022-08-13T03:02:56.032486Z","iopub.status.idle":"2022-08-13T03:03:00.394382Z","shell.execute_reply.started":"2022-08-13T03:02:56.032453Z","shell.execute_reply":"2022-08-13T03:03:00.393460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_set = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ntest_set = test_set[column_needed].copy()\ntest_set","metadata":{"execution":{"iopub.status.busy":"2022-08-13T03:03:00.396389Z","iopub.execute_input":"2022-08-13T03:03:00.397212Z","iopub.status.idle":"2022-08-13T03:03:00.420842Z","shell.execute_reply.started":"2022-08-13T03:03:00.397188Z","shell.execute_reply":"2022-08-13T03:03:00.419850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def submit(test_set = test_set):\n    model = RandomForestClassifier(random_state = 0, max_depth = 3)\n    my_pipeline = Pipeline(steps=[\n        (\"preprocessor\", preprocessor),\n        (\"model\", model)\n    ])\n    \n    my_pipeline.fit(X_train, y_train_full)\n    preds = my_pipeline.predict(test_set)\n    \n    sub = {\"PassengerId\":[x for x in range(892,1310)],\n                  \"Survived\": preds}\n\n    submission = pd.DataFrame(sub)\n    \n    submission.to_csv(\"submission.csv\",index = False)\n    print(\"success\")\n\n\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-13T03:03:00.422877Z","iopub.execute_input":"2022-08-13T03:03:00.423329Z","iopub.status.idle":"2022-08-13T03:03:00.429847Z","shell.execute_reply.started":"2022-08-13T03:03:00.423305Z","shell.execute_reply":"2022-08-13T03:03:00.428841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T03:03:00.430980Z","iopub.execute_input":"2022-08-13T03:03:00.431265Z","iopub.status.idle":"2022-08-13T03:03:00.588562Z","shell.execute_reply.started":"2022-08-13T03:03:00.431239Z","shell.execute_reply":"2022-08-13T03:03:00.587900Z"},"trusted":true},"execution_count":null,"outputs":[]}]}