{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T09:39:33.456813Z","iopub.execute_input":"2022-08-10T09:39:33.457207Z","iopub.status.idle":"2022-08-10T09:39:33.464784Z","shell.execute_reply.started":"2022-08-10T09:39:33.457174Z","shell.execute_reply":"2022-08-10T09:39:33.463922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data Dictionary:**\n\nVariable\tDefinition\tKey\nsurvival\tSurvival\t0 = No, 1 = Yes\npclass\tTicket class\t1 = 1st, 2 = 2nd, 3 = 3rd\nsex\tSex\t\nAge\tAge in years\t\nsibsp\t# of siblings / spouses aboard the Titanic\t\nparch\t# of parents / children aboard the Titanic\t\nticket\tTicket number\t\nfare\tPassenger fare\t\ncabin\tCabin number\t\nembarked\tPort of Embarkation\tC = Cherbourg, Q = Queenstown, S = Southampton","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\nfrom xgboost import XGBRegressor\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:40:41.035175Z","iopub.execute_input":"2022-08-10T09:40:41.035559Z","iopub.status.idle":"2022-08-10T09:40:41.040977Z","shell.execute_reply.started":"2022-08-10T09:40:41.035517Z","shell.execute_reply":"2022-08-10T09:40:41.040050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Examine the data**","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/titanic/train.csv', index_col='PassengerId')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:39:42.814202Z","iopub.execute_input":"2022-08-10T09:39:42.814570Z","iopub.status.idle":"2022-08-10T09:39:42.841485Z","shell.execute_reply.started":"2022-08-10T09:39:42.814540Z","shell.execute_reply":"2022-08-10T09:39:42.840558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:26:16.071081Z","iopub.execute_input":"2022-08-04T20:26:16.071995Z","iopub.status.idle":"2022-08-04T20:26:16.086316Z","shell.execute_reply.started":"2022-08-04T20:26:16.071960Z","shell.execute_reply":"2022-08-04T20:26:16.085096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:26:16.088914Z","iopub.execute_input":"2022-08-04T20:26:16.089322Z","iopub.status.idle":"2022-08-04T20:26:16.100846Z","shell.execute_reply.started":"2022-08-04T20:26:16.089288Z","shell.execute_reply":"2022-08-04T20:26:16.100011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Data preparation. Defining categorical and numerical variables**","metadata":{}},{"cell_type":"code","source":"\n# #drop some columns, creat reduced set:\nred_df = df.drop(columns =['Name','Ticket', 'Cabin'])\n\n# Separate target from predictors\ny = red_df['Survived']\nX = red_df.drop(['Survived'], axis =1)\n\n\n# Divide data into training and validation subsets\nX_train, X_valid, y_train, y_valid = train_test_split(X, y, train_size=0.8, test_size=0.2,\n                                                                random_state=0)\n\n#define numerical columns:\nnum_col = [nc for nc in X_train.columns if X_train[nc].dtype in ['int64', 'float64']]\n\n#define categorical columns:\ncat_col = [cc for cc in X_train.columns if X_train[cc].dtype == 'object']\n\n\n\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:40:51.924709Z","iopub.execute_input":"2022-08-10T09:40:51.925111Z","iopub.status.idle":"2022-08-10T09:40:51.940530Z","shell.execute_reply.started":"2022-08-10T09:40:51.925078Z","shell.execute_reply":"2022-08-10T09:40:51.939567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"red_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:26:16.118947Z","iopub.execute_input":"2022-08-04T20:26:16.119498Z","iopub.status.idle":"2022-08-04T20:26:16.144437Z","shell.execute_reply.started":"2022-08-04T20:26:16.119450Z","shell.execute_reply":"2022-08-04T20:26:16.143499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Preprocessing**","metadata":{}},{"cell_type":"code","source":"from sklearn.compose import ColumnTransformer\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import OneHotEncoder\nfrom sklearn.ensemble import RandomForestClassifier","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:41:00.331289Z","iopub.execute_input":"2022-08-10T09:41:00.331712Z","iopub.status.idle":"2022-08-10T09:41:00.337894Z","shell.execute_reply.started":"2022-08-10T09:41:00.331676Z","shell.execute_reply":"2022-08-10T09:41:00.336582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Preprocesing for numerical data\nnumerical_transformer = SimpleImputer(strategy='constant')\n\n#Prepocesing for categorical data\n\ncategorical_transformer = Pipeline(steps=[\n    ('imputer', SimpleImputer(strategy='most_frequent')), \n    ('OneHotEnc', OneHotEncoder(handle_unknown='ignore'))\n])\n\n#Bundle preprocessing\npreprocesor = ColumnTransformer(transformers=[\n    ('numtran', numerical_transformer, num_col), #my error:put df[num_col] intead num_col\n    ('cattran',categorical_transformer, cat_col)\n\n])\n\n# Define model\nmodel = RandomForestClassifier(n_estimators=1000, random_state=0)\n\n# Bundle preprocessing and modeling\n\nmy_pipeline = Pipeline(steps = [\n    ('preprocesor', preprocesor),\n    ('model', model)\n])\nmy_pipeline.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:44:50.756843Z","iopub.execute_input":"2022-08-10T09:44:50.757254Z","iopub.status.idle":"2022-08-10T09:44:52.659195Z","shell.execute_reply.started":"2022-08-10T09:44:50.757222Z","shell.execute_reply":"2022-08-10T09:44:52.658074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import mean_absolute_error\n\n# Preprocessing of validation data, get predictions\npreds = my_pipeline.predict(X_valid)\n\n# Evaluate the model\nscore = mean_absolute_error(y_valid, preds)\nprint('MAE:', score)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:44:55.461356Z","iopub.execute_input":"2022-08-10T09:44:55.461787Z","iopub.status.idle":"2022-08-10T09:44:55.627355Z","shell.execute_reply.started":"2022-08-10T09:44:55.461750Z","shell.execute_reply":"2022-08-10T09:44:55.626498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Test data**","metadata":{}},{"cell_type":"code","source":"test_data = pd.read_csv('../input/titanic/test.csv')\ntest_X = test_data\ntest_preds = my_pipeline.predict(test_X)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:45:03.147864Z","iopub.execute_input":"2022-08-10T09:45:03.148612Z","iopub.status.idle":"2022-08-10T09:45:03.363502Z","shell.execute_reply.started":"2022-08-10T09:45:03.148576Z","shell.execute_reply":"2022-08-10T09:45:03.362538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/titanic/test.csv')\ntest['Survived'] = 0\ntest.loc[test['Sex'] == 'female','Survived'] = 1\ndata_to_submit = pd.DataFrame({\n    'PassengerId':test['PassengerId'],\n    'Survived':test['Survived']\n})","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:45:05.514011Z","iopub.execute_input":"2022-08-10T09:45:05.514788Z","iopub.status.idle":"2022-08-10T09:45:05.528288Z","shell.execute_reply.started":"2022-08-10T09:45:05.514742Z","shell.execute_reply":"2022-08-10T09:45:05.527312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_to_submit.to_csv('submission.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T09:45:10.358818Z","iopub.execute_input":"2022-08-10T09:45:10.359226Z","iopub.status.idle":"2022-08-10T09:45:10.367306Z","shell.execute_reply.started":"2022-08-10T09:45:10.359194Z","shell.execute_reply":"2022-08-10T09:45:10.366395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Distribution**","metadata":{}},{"cell_type":"code","source":"columns = ['Pclass', 'Age', 'SibSp','Parch', 'Fare']\nfor col in red_df[columns]:\n    sns.displot(red_df[col], height = 4)","metadata":{"execution":{"iopub.status.busy":"2022-08-04T20:26:16.607915Z","iopub.execute_input":"2022-08-04T20:26:16.608321Z","iopub.status.idle":"2022-08-04T20:26:18.311578Z","shell.execute_reply.started":"2022-08-04T20:26:16.608285Z","shell.execute_reply":"2022-08-04T20:26:18.310438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}