{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd \n\nfrom sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import RandomForestRegressor\nfrom sklearn.metrics import mean_absolute_error\nfrom xgboost import XGBRegressor\n\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-07T21:38:25.860594Z","iopub.execute_input":"2022-08-07T21:38:25.862787Z","iopub.status.idle":"2022-08-07T21:38:26.793933Z","shell.execute_reply.started":"2022-08-07T21:38:25.861014Z","shell.execute_reply":"2022-08-07T21:38:26.792794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/titanic/train.csv\")\ntrain_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:38:29.858624Z","iopub.execute_input":"2022-08-07T21:38:29.859191Z","iopub.status.idle":"2022-08-07T21:38:29.901695Z","shell.execute_reply.started":"2022-08-07T21:38:29.859145Z","shell.execute_reply":"2022-08-07T21:38:29.900301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:38:31.606381Z","iopub.execute_input":"2022-08-07T21:38:31.607135Z","iopub.status.idle":"2022-08-07T21:38:31.618749Z","shell.execute_reply.started":"2022-08-07T21:38:31.607072Z","shell.execute_reply":"2022-08-07T21:38:31.617399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = pd.read_csv(\"/kaggle/input/titanic/test.csv\")\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:38:34.806434Z","iopub.execute_input":"2022-08-07T21:38:34.807193Z","iopub.status.idle":"2022-08-07T21:38:34.832751Z","shell.execute_reply.started":"2022-08-07T21:38:34.807139Z","shell.execute_reply":"2022-08-07T21:38:34.832065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:38:37.221196Z","iopub.execute_input":"2022-08-07T21:38:37.222349Z","iopub.status.idle":"2022-08-07T21:38:37.242273Z","shell.execute_reply.started":"2022-08-07T21:38:37.222303Z","shell.execute_reply":"2022-08-07T21:38:37.241402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Filling the missing values in Age with the medians of Sex and Pclass groups\ntrain_data['Age'] = train_data.groupby(['Sex', 'Pclass'])['Age'].apply(lambda x: x.fillna(x.median()))\ntest_data['Age'] = test_data.groupby(['Sex', 'Pclass'])['Age'].apply(lambda x: x.fillna(x.median()))","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:38:44.089043Z","iopub.execute_input":"2022-08-07T21:38:44.089571Z","iopub.status.idle":"2022-08-07T21:38:44.124310Z","shell.execute_reply.started":"2022-08-07T21:38:44.089532Z","shell.execute_reply":"2022-08-07T21:38:44.123103Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = train_data[\"Survived\"]\nX_train = train_data.loc[:, train_data.columns!='Survived'].copy()\n\n# Select categorical columns with relatively low cardinality (convenient but arbitrary)\ncategorical_cols = [cname for cname in X_train.columns if X_train[cname].nunique() < 10 and \n                        X_train[cname].dtype == \"object\"]\n\n# Select numerical columns\nnumerical_cols = [cname for cname in X_train.columns if X_train[cname].dtype in ['int64', 'float64']]\n\n# Keep selected columns only\nmy_cols = categorical_cols + numerical_cols\nX_train = X_train[my_cols].copy()\nX_test = test_data[my_cols].copy()\n\n# features = [\"Pclass\", \"Fare\", \"Sex\", \"SibSp\", \"Parch\"]\n# X_train = pd.get_dummies(train_data[features])\n# X_test = pd.get_dummies(test_data[features])\n\n\n# Preprocessing for numerical data\nnumerical_transformer = SimpleImputer(strategy='constant')\n\n# Preprocessing for categorical data\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')),\n    ('onehot', OneHotEncoder(handle_unknown='ignore'))\n])\n\n# Bundle preprocessing for numerical and categorical data\npreprocessor = ColumnTransformer(\n    transformers=[\n        ('num', numerical_transformer, numerical_cols),\n        ('cat', categorical_transformer, categorical_cols)\n    ])","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:38:51.592454Z","iopub.execute_input":"2022-08-07T21:38:51.592949Z","iopub.status.idle":"2022-08-07T21:38:51.612330Z","shell.execute_reply.started":"2022-08-07T21:38:51.592911Z","shell.execute_reply":"2022-08-07T21:38:51.611175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### RandomForestClassifier","metadata":{}},{"cell_type":"code","source":"model1 = RandomForestClassifier(n_estimators=100, max_depth=5, random_state=1)\n\npipeline1 = Pipeline(steps=[('preprocessor', preprocessor),\n                              ('model', model1)\n                             ])\n\npipeline1.fit(X_train, y_train)\npredictions1 = pipeline1.predict(X_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:39:01.159066Z","iopub.execute_input":"2022-08-07T21:39:01.159501Z","iopub.status.idle":"2022-08-07T21:39:01.386004Z","shell.execute_reply.started":"2022-08-07T21:39:01.159464Z","shell.execute_reply":"2022-08-07T21:39:01.385250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### RandomForestRegressor","metadata":{}},{"cell_type":"code","source":"#model2 = RandomForestRegressor(n_estimators=100, random_state=0)\n\n# Bundle preprocessing and modeling code in a pipeline\n#pipeline2 = Pipeline(steps=[('preprocessor', preprocessor),\n#                              ('model', model2)\n#                             ])\n\n# Preprocessing of training data, fit model \n#pipeline2.fit(X_train, y_train)\n\n# Preprocessing of validation data, get predictions\n#predictions2 = pipeline2.predict(X_test)\n\n# Evaluating the model\n# score = mean_absolute_error(y_test, predictions2)\n# print('MAE:', score)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### XGBoost","metadata":{}},{"cell_type":"code","source":"#model3 = XGBRegressor(n_estimators=1000, learning_rate=0.05, n_jobs=4)\n\n#pipeline3 = Pipeline(steps=[('preprocessor', preprocessor),\n#                              ('model', model3)\n#                             ])\n\n#pipeline3.fit(X_train, y_train)\n\n\n#predictions3 = pipeline3.predict(X_test)\n#print(\"Mean Absolute Error: \" + str(mean_absolute_error(predictions, y_test)))","metadata":{"_kg_hide-output":true,"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"output = pd.DataFrame({'PassengerId': test_data.PassengerId, 'Survived': predictions1})\noutput.to_csv('submission.csv', index=False)\nprint(\"Done!\")","metadata":{"execution":{"iopub.status.busy":"2022-08-07T21:39:17.788410Z","iopub.execute_input":"2022-08-07T21:39:17.789076Z","iopub.status.idle":"2022-08-07T21:39:17.798389Z","shell.execute_reply.started":"2022-08-07T21:39:17.789040Z","shell.execute_reply":"2022-08-07T21:39:17.797562Z"},"trusted":true},"execution_count":null,"outputs":[]}]}