{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-14T19:34:08.426013Z","iopub.execute_input":"2022-08-14T19:34:08.426534Z","iopub.status.idle":"2022-08-14T19:34:08.465424Z","shell.execute_reply.started":"2022-08-14T19:34:08.426429Z","shell.execute_reply":"2022-08-14T19:34:08.464073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This notebook is for Titanic - Machine Learning from Disaster\n\nGoal: predict who will survive disaster on Titanic.","metadata":{}},{"cell_type":"markdown","source":"1. Load & check data","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv('/kaggle/input/titanic/train.csv') #train data\ndata.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:36:50.692355Z","iopub.execute_input":"2022-08-14T19:36:50.692873Z","iopub.status.idle":"2022-08-14T19:36:50.720682Z","shell.execute_reply.started":"2022-08-14T19:36:50.692828Z","shell.execute_reply":"2022-08-14T19:36:50.719435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Checking above:\n\na) Pclass, SibSp, Parch, Fare are ready for use, no additional action needed.\n\nb) Sex needs to be change from male/female to 0/1.\n\nc) Age requires to aproximation of missing rows.\n\nd) Cabin has only ~1/4 data, ~3/4 missing -> drop.\n\ne) Embarked has 2 missing values, but need replace from C = Cherbourg, Q = Queenstown, S = Southampton to 0/1/2.\n\nf) Ticket and PassengerId will be dropped. \n\ng) I'll leave Name for now.","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:35:17.975006Z","iopub.execute_input":"2022-08-14T19:35:17.975559Z","iopub.status.idle":"2022-08-14T19:35:17.987401Z","shell.execute_reply.started":"2022-08-14T19:35:17.975511Z","shell.execute_reply":"2022-08-14T19:35:17.985132Z"}}},{"cell_type":"markdown","source":"- First data modification.","metadata":{}},{"cell_type":"code","source":"data.replace(to_replace={'female':0, 'male':1}, inplace=True) #Replace in Sex column string to number.\ndata.replace(to_replace={'C':0, 'Q':1, 'S':2}, inplace=True) #Replace in Embarked column string to number.\ndata.drop(['PassengerId', 'Ticket', 'Cabin'], axis=1, inplace=True) #Remove 3 columns\ndata.fillna(data.median(), inplace=True) #fill missing values with median ","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:36:52.912002Z","iopub.execute_input":"2022-08-14T19:36:52.912479Z","iopub.status.idle":"2022-08-14T19:36:52.931055Z","shell.execute_reply.started":"2022-08-14T19:36:52.912439Z","shell.execute_reply":"2022-08-14T19:36:52.929996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Check correlation:","metadata":{}},{"cell_type":"code","source":"(data.corr(method='kendall')) #numeric version","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:37:15.065173Z","iopub.execute_input":"2022-08-14T19:37:15.065653Z","iopub.status.idle":"2022-08-14T19:37:15.614801Z","shell.execute_reply.started":"2022-08-14T19:37:15.065611Z","shell.execute_reply":"2022-08-14T19:37:15.613569Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Above pearson shows:\n\nStrongest >10%:     Sex, Pclass, Fare, Embarked\n\nWeaknest  <10%:     SibSp, Age, Parch ","metadata":{}},{"cell_type":"markdown","source":"_________________\n2. Visualize data","metadata":{}},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:37:46.397274Z","iopub.execute_input":"2022-08-14T19:37:46.397705Z","iopub.status.idle":"2022-08-14T19:37:46.621839Z","shell.execute_reply.started":"2022-08-14T19:37:46.397670Z","shell.execute_reply":"2022-08-14T19:37:46.619878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"g = sns.PairGrid(data, diag_sharey=False)\ng.map_upper(sns.scatterplot)\ng.map_lower(sns.kdeplot)\ng.map_diag(sns.kdeplot, lw=2)\ng.add_legend()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:37:53.184059Z","iopub.execute_input":"2022-08-14T19:37:53.184513Z","iopub.status.idle":"2022-08-14T19:38:21.417770Z","shell.execute_reply.started":"2022-08-14T19:37:53.184477Z","shell.execute_reply":"2022-08-14T19:38:21.416769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Above show:\n1. More people died than survived and more men than women.\n2. Outliers are in Fare, SibSp, Age vs Survived.\n3. Avarage age is 20:40.\n4. Most people traveled solo or in pair.","metadata":{}},{"cell_type":"code","source":"sns.set_theme(style=\"whitegrid\", palette=\"pastel\")\n\ndf = data\n\ndf = df.loc[:, ['Survived', 'Pclass', 'Sex', 'Age', 'SibSp', 'Parch', 'Fare', 'Embarked']]\n\n\n# Compute a correlation matrix and convert to long-form\ncorr_mat = df.corr(method='kendall').stack().reset_index(name=\"correlation\")\n\n# Draw each cell as a scatter point with varying size and color\ng = sns.relplot(\n    data=corr_mat,\n    x=\"level_0\", y=\"level_1\", hue=\"correlation\", size=\"correlation\",\n    palette=\"vlag\", hue_norm=(-1, 1), edgecolor=\".5\",\n    height=5, sizes=(25, 100), size_norm=(-.2, .8),\n)\n\n# Tweak the figure to finalize\ng.set(xlabel=\"\", ylabel=\"\", aspect=\"equal\")\ng.despine(left=True, bottom=True)\ng.ax.margins(.02)\nfor label in g.ax.get_xticklabels():\n    label.set_rotation(90)\nfor artist in g.legend.legendHandles:\n    artist.set_edgecolor(\".7\")\n","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:38:21.419660Z","iopub.execute_input":"2022-08-14T19:38:21.420781Z","iopub.status.idle":"2022-08-14T19:38:22.102143Z","shell.execute_reply.started":"2022-08-14T19:38:21.420737Z","shell.execute_reply":"2022-08-14T19:38:22.100766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Visualize correlation - method 'kendall'.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(16,5))\nplt.subplot(1,3,1)\nsns.distplot(data['Fare'])\nplt.subplot(1,3,2)\nsns.distplot(data['SibSp'])\nplt.subplot(1,3,3)\nsns.distplot(data['Age'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:38:40.952111Z","iopub.execute_input":"2022-08-14T19:38:40.952545Z","iopub.status.idle":"2022-08-14T19:38:41.769290Z","shell.execute_reply.started":"2022-08-14T19:38:40.952511Z","shell.execute_reply":"2022-08-14T19:38:41.767836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Treatment an outliers.\n- Quantile(.99) rigth -> fare, sibsp.\n- Z-score   -> age.","metadata":{}},{"cell_type":"markdown","source":"- Second data modification.","metadata":{}},{"cell_type":"code","source":"from scipy import stats","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:39:12.798524Z","iopub.execute_input":"2022-08-14T19:39:12.799088Z","iopub.status.idle":"2022-08-14T19:39:12.805010Z","shell.execute_reply.started":"2022-08-14T19:39:12.799020Z","shell.execute_reply":"2022-08-14T19:39:12.804000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = data[(np.abs(stats.zscore(data['Age'])) < 3)]\ndata = data[(data['Fare'] < (data['Fare'].quantile(0.99)))]\ndata = data[(data['SibSp'] < (data['SibSp'].quantile(0.99)))]","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:39:20.430259Z","iopub.execute_input":"2022-08-14T19:39:20.430674Z","iopub.status.idle":"2022-08-14T19:39:20.448129Z","shell.execute_reply.started":"2022-08-14T19:39:20.430638Z","shell.execute_reply":"2022-08-14T19:39:20.446754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16,5))\nplt.subplot(1,3,1)\nsns.distplot(data['Fare'])\nplt.subplot(1,3,2)\nsns.distplot(data['SibSp'])\nplt.subplot(1,3,3)\nsns.distplot(data['Age'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:39:26.101860Z","iopub.execute_input":"2022-08-14T19:39:26.102284Z","iopub.status.idle":"2022-08-14T19:39:26.815375Z","shell.execute_reply.started":"2022-08-14T19:39:26.102248Z","shell.execute_reply":"2022-08-14T19:39:26.814120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Little improvement.","metadata":{}},{"cell_type":"markdown","source":"______\n3. Model ","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import f1_score\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import confusion_matrix","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:40:14.157380Z","iopub.execute_input":"2022-08-14T19:40:14.157778Z","iopub.status.idle":"2022-08-14T19:40:14.278641Z","shell.execute_reply.started":"2022-08-14T19:40:14.157744Z","shell.execute_reply":"2022-08-14T19:40:14.277357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nx_train = data[['Pclass', 'Sex', 'Age', 'SibSp', 'Parch', 'Fare', 'Embarked']].copy()\ny_train = data['Survived'].copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:40:16.685215Z","iopub.execute_input":"2022-08-14T19:40:16.685621Z","iopub.status.idle":"2022-08-14T19:40:16.693235Z","shell.execute_reply.started":"2022-08-14T19:40:16.685587Z","shell.execute_reply":"2022-08-14T19:40:16.692025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sc = StandardScaler()\nX_train = sc.fit_transform(x_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:40:25.023236Z","iopub.execute_input":"2022-08-14T19:40:25.023671Z","iopub.status.idle":"2022-08-14T19:40:25.034388Z","shell.execute_reply.started":"2022-08-14T19:40:25.023632Z","shell.execute_reply":"2022-08-14T19:40:25.033143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('/kaggle/input/titanic/test.csv')  #test data\nsubmit_0 = test.PassengerId","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:43:17.927322Z","iopub.execute_input":"2022-08-14T19:43:17.927874Z","iopub.status.idle":"2022-08-14T19:43:17.942610Z","shell.execute_reply.started":"2022-08-14T19:43:17.927823Z","shell.execute_reply":"2022-08-14T19:43:17.941296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test data preparation\ntest.replace(to_replace={'female':0, 'male':1}, inplace=True) #Replace in Sex column string to number.\ntest.replace(to_replace={'C':0, 'Q':1, 'S':2}, inplace=True) #Replace in Embarked column string to number.\ntest.drop(['PassengerId', 'Ticket', 'Cabin'], axis=1, inplace=True) #Remove 3 columns\ntest.fillna(data.median(), inplace=True) #fill missing values with median \nx_test= test[['Pclass', 'Sex', 'Age', 'SibSp', 'Parch', 'Fare', 'Embarked']].copy()\nX_test = sc.fit_transform(x_test)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:43:19.644761Z","iopub.execute_input":"2022-08-14T19:43:19.645982Z","iopub.status.idle":"2022-08-14T19:43:19.673013Z","shell.execute_reply.started":"2022-08-14T19:43:19.645930Z","shell.execute_reply":"2022-08-14T19:43:19.671892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.model_selection import RandomizedSearchCV","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:43:33.108524Z","iopub.execute_input":"2022-08-14T19:43:33.108907Z","iopub.status.idle":"2022-08-14T19:43:33.334728Z","shell.execute_reply.started":"2022-08-14T19:43:33.108876Z","shell.execute_reply":"2022-08-14T19:43:33.333230Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Number of trees in random forest\nn_estimators = np.linspace(100, 3000, int((3000-100)/200) + 1, dtype=int)\n# Number of features to consider at every split\nmax_features = ['auto', 'sqrt']\n# Maximum number of levels in tree\nmax_depth = [1, 5, 10, 20, 50, 75, 100, 150, 200]\n# Minimum number of samples required to split a node\n# min_samples_split = [int(x) for x in np.linspace(start = 2, stop = 10, num = 9)]\nmin_samples_split = [1, 2, 5, 10, 15, 20, 30]\n# Minimum number of samples required at each leaf node\nmin_samples_leaf = [1, 2, 3, 4]\n# Method of selecting samples for training each tree\nbootstrap = [True, False]\n# Criterion\ncriterion=['gini', 'entropy']\nrandom_grid = {'n_estimators': n_estimators,\n               #'max_features': max_features,\n               'max_depth': max_depth,\n               'min_samples_split': min_samples_split,\n               'min_samples_leaf': min_samples_leaf,\n               'bootstrap': bootstrap,\n               'criterion': criterion}","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:44:16.317781Z","iopub.execute_input":"2022-08-14T19:44:16.318303Z","iopub.status.idle":"2022-08-14T19:44:16.327077Z","shell.execute_reply.started":"2022-08-14T19:44:16.318260Z","shell.execute_reply":"2022-08-14T19:44:16.326110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rf_base = RandomForestClassifier()\nrf_random = RandomizedSearchCV(estimator = rf_base,\n                               param_distributions = random_grid,\n                               n_iter = 100, cv = 5,\n                               verbose=2,\n                               random_state=42, n_jobs = 12)\nrf_random.fit(X_train, y_train)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:44:24.978672Z","iopub.execute_input":"2022-08-14T19:44:24.979131Z","iopub.status.idle":"2022-08-14T19:52:18.554017Z","shell.execute_reply.started":"2022-08-14T19:44:24.979092Z","shell.execute_reply":"2022-08-14T19:52:18.552656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_model = RandomForestClassifier(**rf_random.best_params_)\nfinal_model.fit(X_train, y_train)\nfinal_pred = final_model.predict(X_test)\nfinal_check = final_model.predict(X_train)\nprint(confusion_matrix(final_check, y_train))\nprint(classification_report(final_check, y_train))","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:52:18.556201Z","iopub.execute_input":"2022-08-14T19:52:18.556589Z","iopub.status.idle":"2022-08-14T19:52:22.785895Z","shell.execute_reply.started":"2022-08-14T19:52:18.556548Z","shell.execute_reply":"2022-08-14T19:52:22.784536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.concat([submit_0, pd.DataFrame(final_pred, columns=['Survived'])], join='inner', axis=1)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-14T19:52:22.812730Z","iopub.execute_input":"2022-08-14T19:52:22.813397Z","iopub.status.idle":"2022-08-14T19:52:22.826529Z","shell.execute_reply.started":"2022-08-14T19:52:22.813358Z","shell.execute_reply":"2022-08-14T19:52:22.825504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}