{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"****Import packages****","metadata":{"id":"USpWIfvuizP5"}},{"cell_type":"code","source":"# Basic Libraries\nimport pandas as pd\nimport seaborn as sns\n\nimport numpy as np\nfrom numpy import mean\nfrom numpy import std\nfrom numpy import absolute\n\nimport matplotlib.pyplot as plt\nfrom matplotlib.colors import ListedColormap\nfrom scipy.stats.mstats import winsorize\nimport scipy.stats as ss\nimport math","metadata":{"id":"DIZaKU5zgfxn"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Scikit learn\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import MinMaxScaler\nfrom sklearn.linear_model import Perceptron\nfrom sklearn.naive_bayes import GaussianNB\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.svm import SVC\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import classification_report\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.metrics import log_loss\nfrom sklearn.metrics import balanced_accuracy_score\nfrom sklearn.metrics import f1_score\nfrom sklearn.model_selection import cross_val_score\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import KFold\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import RepeatedStratifiedKFold\nfrom sklearn.model_selection import learning_curve\nfrom sklearn.model_selection import GridSearchCV\nfrom sklearn.model_selection import RepeatedKFold\nfrom sklearn.metrics import mean_squared_error, r2_score, mean_absolute_error, mean_absolute_percentage_error\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\n\n# mblearn library\nfrom imblearn.over_sampling import SMOTE\nfrom imblearn.pipeline import Pipeline as imbpipeline\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.datasets import make_classification\nfrom sklearn.datasets import load_breast_cancer\n\n# LightGBM\n#!pip install lightgbm\nfrom lightgbm import LGBMClassifier\n\n# XGBoost\nfrom xgboost import XGBClassifier\n\n\n# scikitplot\n#!pip install scikit-plot\nimport scikitplot as skplt","metadata":{"id":"z6njqu3LixHE","outputId":"b8d92909-b8d2-4dc6-f1bb-04c5bb0efca9"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n#Read\ndf = pd.read_csv('/kaggle/input/spaceship-titanic/train.csv')\ntest = pd.read_csv('/kaggle/input/spaceship-titanic/test.csv')\n","metadata":{"id":"GflJ6dvVgjd0","outputId":"b738476b-5928-4afa-9a9d-c24a1e432eac"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Read \nfile_ = \"/content/drive/MyDrive/ML/wow/train.csv\"   \ndf = pd.read_csv(file_) \nfile_ = \"/content/drive/MyDrive/ML/wow/test.csv\" \ntest = pd.read_csv(file_) ","metadata":{"id":"9yIkDHgRglMA"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"id":"UT2kpPXbgndi","outputId":"bd08228d-70ed-47d3-e010-c0ac85d15e6e"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check the columns\ndf= df.drop(columns='Name')\ntest= test.drop(columns='Name')","metadata":{"id":"gEkI1bzKgmZ_"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"id":"rwpQ0q4YkEVg","outputId":"c74e77ab-2e64-4e25-ec5f-01759f819c36"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Check for duplicates\nprint('\\n Duplicates\\n',df.duplicated().sum())","metadata":{"id":"PzRvBx-V7dCW","outputId":"e94b950f-2f98-436c-9084-7ba126364eaa"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Convert string into object/float**","metadata":{"id":"Pj6oyWWd0TUs"}},{"cell_type":"code","source":"#split Cabin column into three columns\nnew = df[\"Cabin\"].str.split(\"/\", n = 2, expand = True)\ndf[\"Deck\"]=new[0]\ndf[\"Num\"]=new[1]\ndf[\"Side\"]=new[2]\n\nnew = test[\"Cabin\"].str.split(\"/\", n = 2, expand = True)\ntest[\"Deck\"]=new[0]\ntest[\"Num\"]=new[1]\ntest[\"Side\"]=new[2]","metadata":{"id":"5XN9iimnldjb"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#split PassengerId column into two columns\nnew = df[\"PassengerId\"].str.split(\"_\", n = 1, expand = True)\ndf[\"Groupid\"]=new[0]\ndf[\"Id\"]=new[1]\n\nnew = test[\"PassengerId\"].str.split(\"_\", n = 1, expand = True)\ntest[\"Groupid\"]=new[0]\ntest[\"Id\"]=new[1]","metadata":{"id":"sizDzkUfOb7K"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df=df.drop(columns=['PassengerId','Cabin'])\ntest=test.drop(columns=['PassengerId','Cabin'])","metadata":{"id":"zrBNQ0p7jx2r"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"id":"tAZB5SiOkyqH","outputId":"11049311-db8e-4b3c-e6e1-9eba9bebe892"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.shape","metadata":{"id":"0INz5rWa4K47","outputId":"9ce4e674-4281-4891-e4fa-02c604dc4a8e"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"id":"KTZyfRXXI4WG","outputId":"9e4ac479-2e1d-4498-b9cb-b81a4ef18269"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"id":"4C2hIo1KrmX8","outputId":"50d7625e-9ed7-43ac-9216-4a2f3923244d"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Descriptive Statistics**","metadata":{"id":"1qjB7vAC9WyB"}},{"cell_type":"code","source":"#Correlation Matrix\nprint(\"CORRELATION MATRIX\\n\",df.corr())\nprint(\"\\n\\n\")\n\n#Correlation Matrix as a Heatmap\nsns.set_style('darkgrid')\nplt.figure(figsize = (12,10))\ncmap = sns.diverging_palette(120, 10, l = 40, s = 99, sep = 20, center = 'light', as_cmap = True) \nsns.heatmap((df).corr(), vmin = -1, vmax = 1, annot = True, cmap = cmap, lw = .5, linecolor = 'white')\nplt.title(\"Spaceship Dataset Correlation Heatmap\")\nplt.show()","metadata":{"id":"KT4n3BnC9hxK","outputId":"ea38b48d-0b91-4cdf-ce46-b41d8bd46b49"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#take a look at the numeric variables\ndf.hist(figsize=(15,15))","metadata":{"id":"3-1Hmojz9Vfs","outputId":"8e5edaf3-64e3-4f25-b334-369a25eb082d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot Target variable with numeric variables\n\nfig, axes = plt.subplots(3, 2, figsize=(15,10))\nsns.boxplot(  y=\"Age\", x= \"Transported\", data=df,  orient='v' , ax=axes[0, 0])\nsns.boxplot(  y=\"RoomService\", x= \"Transported\", data=df,  orient='v' , ax=axes[0, 1])\nsns.boxplot(  y=\"FoodCourt\", x= \"Transported\", data=df,  orient='v' , ax=axes[1, 0])\nsns.boxplot(  y=\"ShoppingMall\", x= \"Transported\", data=df,  orient='v' , ax=axes[1, 1])\nsns.boxplot(  y=\"Spa\", x= \"Transported\", data=df,  orient='v' , ax=axes[2, 0])\nsns.boxplot(  y=\"VRDeck\", x= \"Transported\", data=df,  orient='v' , ax=axes[2, 1])\nplt.show()","metadata":{"id":"KC1B3-xR9xTa","outputId":"2c6f9fa9-2fc1-46e2-c006-da728debde85"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot Target variable with numeric variables\n# sns.set_theme(style=\"ticks\")\n# sns.pairplot(df.iloc[:,[3,5,6,7,8,9,10]], hue=\"Transported\")","metadata":{"id":"5DiPxGQg-LZ8"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.columns","metadata":{"id":"hZhtMBbWAf_q","outputId":"3df85a9e-6327-49aa-905e-a0e49ba0c8de"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plot Target variable with catagorical variables\n\nplt.figure(figsize = (12,6))\nplt.subplot(1,2,1)\nsns.countplot(x= 'HomePlanet',hue='Transported', data=df)\n\nplt.subplot(1,2,2)\nsns.countplot(x='CryoSleep',hue='Transported',data=df)\nplt.show()\n\nplt.figure(figsize = (12,6))\nplt.subplot(1,2,1)\nsns.countplot(x= 'Destination',hue='Transported',data=df)\n\nplt.subplot(1,2,2)\nsns.countplot(x= 'VIP',hue='Transported', data=df)\nplt.show()\n\nplt.figure(figsize = (12,6))\nplt.subplot(1,2,1)\nsns.countplot(x= 'Deck',hue='Transported', data=df)\n\nplt.subplot(1,2,2)\nsns.countplot(x= 'Side',hue='Transported', data=df)\nplt.show()","metadata":{"id":"9wvTuOidA84S","outputId":"3a395393-cc81-4244-eab9-26c9f10a3b4d"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**EDA**","metadata":{"id":"9tGD2gzXk5Ch"}},{"cell_type":"markdown","source":"**Missing values**","metadata":{"id":"hTrGsP3Er5TV"}},{"cell_type":"code","source":"# Check for missing values\nprint('\\n\\nMissing Values\\n',df.isnull().sum(axis=0))\n\n# Missing value heatmap\nplt.figure(figsize=(10,6))\nsns.heatmap(df.isna().transpose(),\n            cmap=\"YlGnBu\",\n            cbar_kws={'label': 'Missing Data Heatmap'})\nplt.show()","metadata":{"id":"Cb_XAhEdk4h5","outputId":"b8173582-be8b-4f66-c22d-482b6c12b9b9"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Handling missing values\n#For column HomePlanet, Destination, Deck, Num, Side, and Groupid, we delete the record with NA\ndf.dropna(subset=[\"HomePlanet\",\"Destination\",\"Deck\",\"Num\",\"Side\",\"Groupid\"],how='any',inplace=True)\n\n#For column VIP and CryoSleep, we set defualt value as False for NA\ndf[\"VIP\"]=df[\"VIP\"].fillna(False)\ndf[\"CryoSleep\"]=df[\"CryoSleep\"].fillna(False)\n\n#For column Age, RoomService, FoodCourt, ShoppingMall, Spa, VRDeck, we replace NA with median\ndf[\"Age\"]=df[\"Age\"].fillna(df[\"Age\"].median())\ndf[\"RoomService\"]=df[\"RoomService\"].fillna(df[\"RoomService\"].median())\ndf[\"FoodCourt\"]=df[\"FoodCourt\"].fillna(df[\"FoodCourt\"].median())\ndf[\"ShoppingMall\"]=df[\"ShoppingMall\"].fillna(df[\"ShoppingMall\"].median())\ndf[\"Spa\"]=df[\"Spa\"].fillna(df[\"Spa\"].median())\ndf[\"VRDeck\"]=df[\"VRDeck\"].fillna(df[\"VRDeck\"].median())\n","metadata":{"id":"SId3g9RSzUJa"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.isnull().sum(axis=0).sum())","metadata":{"id":"zCkjDcW71yI5","outputId":"39222e8b-10a7-4d43-e057-af101c2f2cb6"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Encodings**","metadata":{"id":"D2c8thkkuX4H"}},{"cell_type":"code","source":"#Features for dummy encoding:['CryoSleep','VIP','Transported']\ndf = pd.get_dummies(data=df, columns=['CryoSleep'],drop_first=True)\ndf = pd.get_dummies(data=df, columns=['VIP'],drop_first=True)\ndf = pd.get_dummies(data=df, columns=['Transported'],drop_first=True)\n\ntest = pd.get_dummies(data=test, columns=['CryoSleep'],drop_first=True)\ntest = pd.get_dummies(data=test, columns=['VIP'],drop_first=True)","metadata":{"id":"mfFVLUUX9gma"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Features for One-hot encoding:['HomePlanet','Destination','Deck','Side']\n\ndf = pd.get_dummies(data=df, columns=['HomePlanet'],drop_first=True)\ndf = pd.get_dummies(data=df, columns=['Destination'],drop_first=True)\ndf = pd.get_dummies(data=df, columns=['Deck'],drop_first=True)\ndf = pd.get_dummies(data=df, columns=['Side'],drop_first=True)\n\ntest = pd.get_dummies(data=test, columns=['HomePlanet'],drop_first=True)\ntest = pd.get_dummies(data=test, columns=['Destination'],drop_first=True)\ntest = pd.get_dummies(data=test, columns=['Deck'],drop_first=True)\ntest = pd.get_dummies(data=test, columns=['Side'],drop_first=True)\n\ndf.columns","metadata":{"id":"CSCkw0jVAhM3","outputId":"2d6b3f5e-3cea-41af-8902-f5335105428f"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.columns","metadata":{"id":"nba9O3F-5tPW","outputId":"1e0d8609-66cd-4b6e-8e6d-abc0f8e742a9"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.dtypes","metadata":{"id":"xvJmMyaXG6IF","outputId":"13f75ba3-7e3b-4227-88c0-a030f9e6d87d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.astype({\"Num\": int}, errors='raise') \ndf = df.astype({\"Groupid\": int}, errors='raise') \ndf = df.astype({\"Id\": int}, errors='raise') \n\ntest = test.astype({\"Num\": float}, errors='raise') \ntest = test.astype({\"Groupid\": int}, errors='raise') \ntest = test.astype({\"Id\": int}, errors='raise')","metadata":{"id":"tclnMgpfGt3I"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"id":"PZd3R1n3_khI","outputId":"fd173bfa-d46c-416a-8750-481d4e28a1f1"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Duplicates**","metadata":{"id":"HI7O7WSRtrCT"}},{"cell_type":"code","source":"# Check for duplicates\nprint('\\n Duplicates\\n',df.duplicated().sum())","metadata":{"id":"U1-Jtb73lTWR","outputId":"402edceb-468a-41f4-81a7-5243ae6e096f"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Predictors & Target Variable**","metadata":{"id":"7YW1Irls0HCz"}},{"cell_type":"code","source":"# Split x into numeric(x1), catagorical(x2) and y\nx1_train=df[['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']] \nx2_train=df.drop(['Transported_True','Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck'], axis=1)\ny_train=df['Transported_True']\n\nx1_test=test[['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']] \nx2_test=test.drop(['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck'], axis=1)","metadata":{"id":"jCLLBQQgCVQK"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Skewness**","metadata":{"id":"ZaUxJte7tw6e"}},{"cell_type":"code","source":"#join train and test data\nx1_all = pd.concat([x1_train, x1_test], axis=0)","metadata":{"id":"f-KpwtL2428q"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1_all.describe()","metadata":{"id":"WBnhVrki69gt","outputId":"76341d11-486e-482e-b39a-69c54355e7d1"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1_all.iloc[12401,:]","metadata":{"id":"lUXxjk4p7AhT","outputId":"60a1aeca-c1c8-46e8-9b8b-6ee049f7406d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1_all.iloc[8124,:]\n##train data start from 0 to 8124, test data start from 8125 to 12401","metadata":{"id":"kSYqoDEb6PMh","outputId":"cce995a7-8b14-4665-9af6-5d70ecccb65b"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Ensure the maximum number of columns are displayed in pandas\npd.set_option('display.max_columns', None)\n\nprint(\"\\n SKEWNESS\\n\",x1_all.skew())\nprint(\"\\n FISHER'S KURTOSIS\\n\",x1_all.kurt())","metadata":{"id":"HXHRV6NylYEa","outputId":"e0512390-4cf3-4a37-d7f3-eeaeac760064"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**log transform**","metadata":{"id":"vO7FXfIMKjQG"}},{"cell_type":"code","source":"columns=['RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']\n\nfor col in columns:\n  z=np.log(x1_all[col]+1)\n  fig, axes = plt.subplots(1, 2, figsize=(10, 5))\n  sns.histplot(x1_all[col],ax=axes[0], color=\"blue\", label=\"100% Equities\", kde=True, stat=\"density\", linewidth=0)\n  sns.histplot(z,ax=axes[1], color=\"red\", label=\"100% Equities\", kde=True, stat=\"density\", linewidth=0)\n  x1_all[col]\n  x1_all[col]=z","metadata":{"id":"0WWqXRmtFsSp","outputId":"4599c6bc-0ee5-42e5-8f88-08bff6ad9fac"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**skewness from class(we did not use this code)**","metadata":{"id":"pMvMbB-4JSG6"}},{"cell_type":"code","source":"# -*- coding: utf-8 -*-\n\ndef skew_autotransform(DF, include = None, exclude = None, plot = False, threshold = 1, exp = False):\n    \n    #Get list of column names that should be processed based on input parameters\n    if include is None and exclude is None:\n        colnames = DF.columns.values\n    elif include is not None:\n        colnames = include\n    elif exclude is not None:\n        colnames = [item for item in list(DF.columns.values) if item not in exclude]\n    else:\n        print('No columns to process!')\n    \n    #Helper function that checks if all values are positive\n    def make_positive(series):\n        minimum = np.amin(series)\n        #If minimum is negative, offset all values by a constant to move all values to positive teritory\n        if minimum <= 0:\n            series = series + abs(minimum) + 0.01\n        return series\n    \n    \n    #Go through desired columns in DataFrame\n    for col in colnames:\n        #Get column skewness\n        skew = DF[col].skew()\n        transformed = True\n        \n        if plot:\n            #Prep the plot of original data\n            sns.set_style(\"darkgrid\")\n            sns.set_palette(\"Blues_r\")\n            fig, axes = plt.subplots(1, 2, figsize=(10, 5))\n            #ax1 = sns.distplot(DF[col], ax=axes[0])\n            ax1 = sns.histplot(DF[col], ax=axes[0], color=\"blue\", label=\"100% Equities\", kde=True, stat=\"density\", linewidth=0)\n            ax1.set(xlabel='Original ' + str(col))\n        \n        #If skewness is larger than threshold and positively skewed; If yes, apply appropriate transformation\n        if abs(skew) > threshold and skew > 0:\n            skewType = 'positive'\n            #Make sure all values are positive\n            DF[col] = make_positive(DF[col])\n            \n            if exp:\n               #Apply log transformation \n               DF[col] = DF[col].apply(math.log)\n            else:\n                #Apply boxcox transformation\n                DF[col] = ss.boxcox(DF[col])[0]\n            skew_new = DF[col].skew()\n         \n        elif abs(skew) > threshold and skew < 0:\n            skewType = 'negative'\n            #Make sure all values are positive\n            DF[col] = make_positive(DF[col])\n            \n            if exp:\n               #Apply exp transformation \n               DF[col] = DF[col].pow(10)\n            else:\n                #Apply boxcox transformation\n                DF[col] = ss.boxcox(DF[col])[0]\n            skew_new = DF[col].skew()\n        \n        else:\n            #Flag if no transformation was performed\n            transformed = False\n            skew_new = skew\n        \n        #Compare before and after if plot is True\n        if plot:\n            print('\\n ------------------------------------------------------')     \n            if transformed:\n                print('\\n %r had %r skewness of %2.2f' %(col, skewType, skew))\n                print('\\n Transformation yielded skewness of %2.2f' %(skew_new))\n                sns.set_palette(\"Paired\")\n                #ax2 = sns.distplot(DF[col], ax=axes[1], color = 'r')\n                ax2 = sns.histplot(DF[col], ax=axes[1], color=\"red\", label=\"100% Equities\", kde=True, stat=\"density\", linewidth=0)\n                ax2.set(xlabel='Transformed ' + str(col))\n                plt.show()\n            else:\n                print('\\n NO TRANSFORMATION APPLIED FOR %r . Skewness = %2.2f' %(col, skew))\n                #ax2 = sns.distplot(DF[col], ax=axes[1])\n                ax2 = sns.histplot(DF[col], ax=axes[1], color=\"blue\", label=\"100% Equities\", kde=True, stat=\"density\", linewidth=0)\n                ax2.set(xlabel='NO TRANSFORM ' + str(col))\n                plt.show()\n                \n\n    return DF","metadata":{"id":"NftXEEC2BaoW"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Use code above (adapted from https://github.com/datamadness/Automatic-skewness-transformation-for-Pandas-DataFrame) to correct skewness\n# x1_all = skew_autotransform(x1_all.copy(deep=True), plot = True, exp = True, threshold = 1)","metadata":{"id":"rKvMt3V-BkgQ"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Outliers**","metadata":{"id":"LnqpOtbWt3zD"}},{"cell_type":"code","source":"#split data into train and test data\nx1_train=x1_all.iloc[0:8125,:]\nx1_test=x1_all.iloc[8125:,:]","metadata":{"id":"7en2NMsM_Osf"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1_test","metadata":{"id":"dKumrn2-AMAZ","outputId":"550d2279-7188-430e-a81d-b876de6707fa"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check outliers in quant varianbles \nfor i in x1_train.columns:\n  sns.boxplot(x1_train[i])\n  plt.show()","metadata":{"id":"ibg6NXnvE6hE","outputId":"0588d4b3-1c4c-4df7-db21-c39c03f9a37b"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1_train.info()","metadata":{"id":"dviC2T0h8xxE","outputId":"96f0cb87-0cc9-46bc-a3e9-e8cd52c8dc75"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Windsorize X and check the results\nprint(\"Before\", x1_train['Age'].describe())\nx_winsorized = x1_train.copy(deep=True)\nx_winsorized['Age'] = winsorize(x1_train['Age'], limits=(0.05, 0.05))\nx_winsorized['ShoppingMall'] = winsorize(x1_train['ShoppingMall'], limits=(0.05, 0.05))\nx_winsorized['VRDeck'] = winsorize(x1_train['VRDeck'], limits=(0.05, 0.05))\nprint(\"After\", x_winsorized.describe())\n\nax = sns.boxplot(data=x_winsorized['Age'], orient=\"h\", palette=\"Set2\")\nax = sns.boxplot(data=x_winsorized['ShoppingMall'], orient=\"h\", palette=\"Set2\")\nax = sns.boxplot(data=x_winsorized['VRDeck'], orient=\"h\", palette=\"Set2\")\nplt.show()","metadata":{"id":"9BUE-pYZfJev","outputId":"6df389d3-5357-43c0-ff9c-f0c105f90686"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Holdout Sample**","metadata":{"id":"sHZjarUOJvNs"}},{"cell_type":"code","source":"x_train=pd.concat([x1_train, x2_train], axis=1)","metadata":{"id":"LpHD2pCvJ417"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(x_train, y_train, test_size=0.3, random_state=54321)","metadata":{"id":"jKT1jEJVJx-z"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**SMOTE**","metadata":{"id":"_j8qmIjgt5aG"}},{"cell_type":"code","source":"X_train","metadata":{"id":"5jNLpdtyUUNc","outputId":"cb8ecb06-7c2a-4294-c91a-fd45b2dab457"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.mean()\n#The train data is little unbalanced","metadata":{"id":"LkV_Sn_TYF3v","outputId":"5a7d83ec-99b3-4443-941a-957f27a4006a"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SMOTE (oversampling)\n# Data is unbalanced (38.54% converted instances)\n\nimport imblearn\nprint(\"imblearn version: \", imblearn.__version__)\n\nfrom imblearn.over_sampling import SMOTE\n\nsm = SMOTE(random_state=12346)\nX_tr_SMOTE, y_tr_SMOTE = sm.fit_resample(X_train, y_train)\n\nprint(\"Shape before SMOTE: \", X_train.shape, y_train.shape, \"\\n\")\nprint(\"Shape after SMOTE: \", X_tr_SMOTE.shape, y_tr_SMOTE.shape, \"\\n\")","metadata":{"id":"3Azehy9wX-sX","outputId":"e0c4c920-0766-4762-edb1-6496a16624fd"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_tr_SMOTE.mean()","metadata":{"id":"4S3eWwbqkmar","outputId":"cedaeae2-8468-4925-ffcc-fd124a35b6ee"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_tr_SMOTE","metadata":{"id":"TLK2OCSGk5Pw","outputId":"853c734b-2f08-4da9-a5a8-3f6393d78d08"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_tr_SMOTE.columns","metadata":{"id":"rvkyv4XnU2zK","outputId":"b2cb5c21-ddb2-4ae7-9ce6-2161ee009bfd"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Scaling**","metadata":{"id":"CvKExqhtuzHy"}},{"cell_type":"code","source":"#split x into x1 and x2\nX1_train=X_tr_SMOTE[['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']] \nX2_train=X_tr_SMOTE.drop(['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck'], axis=1)\n\nX1_test=X_test[['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck']] \nX2_test=X_test.drop(['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck'], axis=1)","metadata":{"id":"nwDe45PAU7bT"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stdsc1 = StandardScaler()  \nX1_tr_std = stdsc1.fit_transform(X1_train)\nx1_te_std = stdsc1.fit_transform(x1_test)\nX1_te_std = stdsc1.fit_transform(X1_test)","metadata":{"id":"T4R41kziPapc"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X1_tr_std","metadata":{"id":"ZemCXaEoVY1w","outputId":"45c6f0a2-8827-4b0e-c94a-d77d36de3f06"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1_te_std","metadata":{"id":"bks1Pzh2BDwP","outputId":"0eed2e3c-5a77-4256-a163-5f81153bc500"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X1_tr_std=pd.DataFrame(X1_tr_std,columns=['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck'])\nx1_te_std=pd.DataFrame(x1_te_std,columns=['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck'])\nX1_te_std=pd.DataFrame(X1_te_std,columns=['Age', 'RoomService', 'FoodCourt', 'ShoppingMall', 'Spa', 'VRDeck'])","metadata":{"id":"3ax2kHPmmSvQ"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X1_tr_std","metadata":{"id":"pSi1K7AhkVsD","outputId":"84da6228-97fc-4cbd-ff61-5e90454ea5ab"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1_te_std","metadata":{"id":"L_eAaZc_BJqA","outputId":"f5612272-4841-45e7-9a96-1a49a84c0d97"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s = pd.Series(range(2438))\nX2_test.set_index([s],inplace=True)","metadata":{"id":"cFlQWdCcOPHB"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concatenate x1 and x2\nX_tr_std = pd.concat([X1_tr_std, X2_train], axis=1)\nx_te_std = pd.concat([x1_te_std, x2_test], axis=1)\nX_te_std = pd.concat([X1_te_std, X2_test], axis=1)","metadata":{"id":"NPN5JGxsjgaI"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Feature Selection with Random Forest**","metadata":{"id":"pW1wZohfVs3m"}},{"cell_type":"code","source":"# Feature Importance\nfrom matplotlib import pyplot                            # Import pyplot (to be able generate the barchart later in this snippet)\nrf = RandomForestClassifier()                         # Create an instance of a RandomForestClassifier\n# fit the model\nrf.fit(X_tr_std, y_tr_SMOTE)                  # Fit the RandomForest instance using the traiing data\n# get importance \nimportance = rf.feature_importances_                  # The RandomForestClassifier instance computes feature importance as a bonus. Store them imprtance values in importance'.\n# summarize feature importance\ncol_names=X_tr_std.columns\nplt.barh(col_names, rf.feature_importances_) \n\n#CryoSleep, Groupid, Num, VRDeck, Spa, ShoppingMall, Foodcourt, RoomService, Age","metadata":{"id":"aiSdco8QV6N8","outputId":"b7636c45-87bc-4159-a653-3ad44e6cb9ae"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Feature Selection with Recursive Feature Elimination (RFE)**","metadata":{"id":"rpBDySOPSpgm"}},{"cell_type":"code","source":"from sklearn.feature_selection import RFE\nrfe = RFE(estimator=RandomForestClassifier(), n_features_to_select=10)\n_ = rfe.fit(X_tr_std,y_tr_SMOTE)\nprint('Important Features\\n',X_tr_std.columns[rfe.support_])\nrf = RandomForestClassifier()\n_ = rf.fit(rfe.transform(X_tr_std), y_tr_SMOTE)\nprint(\"\\n Accuracy: \",rf.score(rfe.transform(X_tr_std), y_tr_SMOTE))","metadata":{"id":"cEsNg2CHZ7Ac","outputId":"06783e64-90f8-45dd-99d4-a7ce1243f0cb"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Pipeline**","metadata":{"id":"jxdUE5xGFID0"}},{"cell_type":"code","source":"names = [\"Perceptron\", \"Logistic Regression\", \"SVM (RBF kernel)\", \"Decision Tree\", \"Naive Bayes\", \"k Nearest Neighbors\", \"MLP\", \"Random Forest\", \"XG Boost\", \"Light GBM\"]\nclassifiers = [\n    Perceptron(random_state=1),    \n    LogisticRegression(),   \n    SVC(kernel=\"rbf\", C=1),\n    DecisionTreeClassifier(max_depth=5),\n    GaussianNB(),\n    KNeighborsClassifier(3),\n    MLPClassifier(hidden_layer_sizes=(50,50),alpha=1, max_iter=1000),\n    RandomForestClassifier(max_depth=5, n_estimators=10, max_features=1),\n    XGBClassifier(random_state=0, n_jobs=-1, learning_rate=0.1, n_estimators=100, max_depth=3),\n    LGBMClassifier(boosting_type='gbdt', objective='binary', num_leaves=50, learning_rate=0.1, bagging_fraction=0.9, feature_fraction=0.9, reg_lambda=0.2)]\n\n# Build each classifier using the unbalanced TRAINING data, show decision region and petrformance of the unbalanced TEST data \nno_folds = 5 # number of folds desired for cross validation\nkf = StratifiedKFold(n_splits=no_folds, shuffle=True, random_state=12345)\nfor name, clf in zip(names, classifiers):\n  print('CLASSIFIER: ',name,'\\n')\n  mean_accuracy = 0.0\n  mean_balanced_accuracy = 0.0\n  mean_auc = 0.0\n  for fold, (train_index, test_index) in enumerate(kf.split(X_tr_std,y_tr_SMOTE),1):  \n    clf.fit(X_tr_std, y_tr_SMOTE) \n    y_pred = clf.predict(X_te_std)\n    print(f'For fold {fold}:')\n    print(f'Accuracy: {clf.score(X_te_std, y_test)}')\n    print(f'Balanced Accuracy: {balanced_accuracy_score(y_test, y_pred)}')\n    print(f'AUC: {roc_auc_score(y_test, y_pred)}')\n    mean_accuracy = mean_accuracy + clf.score(X_te_std, y_test)\n    mean_balanced_accuracy = mean_balanced_accuracy + balanced_accuracy_score(y_test, y_pred)\n    mean_auc = mean_auc + roc_auc_score(y_test, y_pred)\n  mean_accuracy = mean_accuracy / no_folds\n\n  mean_balanced_accuracy = mean_balanced_accuracy / no_folds\n  mean_auc = mean_auc / no_folds\n  print('Average accuracy: %.3f Average balanced accuracy: %.3f Average AUC: %.3f\\n\\n' % (mean_accuracy, mean_balanced_accuracy, mean_auc))","metadata":{"id":"hxAnA3ZQFSAe","outputId":"d4c714e9-2d18-4dd6-ce10-c39d85326a0f"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#XGB and Light GBM are the best two models","metadata":{"id":"I1zAPU2_H7zO"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Learning Curve**","metadata":{"id":"0Bl5hC1AQtAy"}},{"cell_type":"code","source":"#create X and y for learning curve\nX=pd.concat([X_tr_std,X_te_std],axis=0)\ny=pd.concat([y_tr_SMOTE,y_test],axis=0)","metadata":{"id":"sMNCoVYUz9rv"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.svm import SVC\nfrom sklearn.datasets import load_digits\nfrom sklearn.model_selection import learning_curve\nfrom sklearn.model_selection import ShuffleSplit\n\ndef plot_learning_curve(\n    estimator,\n    title,\n    X,\n    y,\n    axes=None,\n    ylim=None,\n    cv=None,\n    n_jobs=None,\n    scoring = 'roc_auc_score',\n    train_sizes=np.linspace(0.1, 1.0, 5),\n):\n    if axes is None:\n        _, axes = plt.subplots(1, 3, figsize=(20, 5))\n\n    axes[0].set_title(title)\n    if ylim is not None:\n        axes[0].set_ylim(*ylim)\n    axes[0].set_xlabel(\"Training examples\")\n    axes[0].set_ylabel(\"Score\")\n\n    train_sizes, train_scores, test_scores, fit_times, _ = learning_curve(\n        estimator,\n        X,\n        y,\n        cv=cv,\n        n_jobs=n_jobs,\n        train_sizes=train_sizes,\n        return_times=True,\n    )\n    train_scores_mean = np.mean(train_scores, axis=1)\n    train_scores_std = np.std(train_scores, axis=1)\n    test_scores_mean = np.mean(test_scores, axis=1)\n    test_scores_std = np.std(test_scores, axis=1)\n    fit_times_mean = np.mean(fit_times, axis=1)\n    fit_times_std = np.std(fit_times, axis=1)\n    print('Scores:',test_scores_mean,test_scores_std)\n\n    # Plot learning curve\n    axes[0].grid()\n    axes[0].fill_between(\n        train_sizes,\n        train_scores_mean - train_scores_std,\n        train_scores_mean + train_scores_std,\n        alpha=0.1,\n        color=\"r\",\n    )\n    axes[0].fill_between(\n        train_sizes,\n        test_scores_mean - test_scores_std,\n        test_scores_mean + test_scores_std,\n        alpha=0.1,\n        color=\"g\",\n    )\n    axes[0].plot(\n        train_sizes, train_scores_mean, \"o-\", color=\"r\", label=\"Training score\"\n    )\n    axes[0].plot(\n        train_sizes, test_scores_mean, \"o-\", color=\"g\", label=\"Cross-validation score\"\n    )\n    axes[0].legend(loc=\"best\")\n\n    # Plot n_samples vs fit_times\n    axes[1].grid()\n    axes[1].plot(train_sizes, fit_times_mean, \"o-\")\n    axes[1].fill_between(\n        train_sizes,\n        fit_times_mean - fit_times_std,\n        fit_times_mean + fit_times_std,\n        alpha=0.1,\n    )\n    axes[1].set_xlabel(\"Training examples\")\n    axes[1].set_ylabel(\"fit_times\")\n    axes[1].set_title(\"Scalability of the model\")\n\n    # Plot fit_time vs score\n    fit_time_argsort = fit_times_mean.argsort()\n    fit_time_sorted = fit_times_mean[fit_time_argsort]\n    test_scores_mean_sorted = test_scores_mean[fit_time_argsort]\n    test_scores_std_sorted = test_scores_std[fit_time_argsort]\n    axes[2].grid()\n    axes[2].plot(fit_time_sorted, test_scores_mean_sorted, \"o-\")\n    axes[2].fill_between(\n        fit_time_sorted,\n        test_scores_mean_sorted - test_scores_std_sorted,\n        test_scores_mean_sorted + test_scores_std_sorted,\n        alpha=0.1,\n    )\n    axes[2].set_xlabel(\"fit_times\")\n    axes[2].set_ylabel(\"Score\")\n    axes[2].set_title(\"Performance of the model\")\n\n    return plt\n\n\nfig, axes = plt.subplots(3, 3, figsize=(20, 20))\n\ntitle = \"Learning Curves (Logistic Regression (Logit))\"\n# Cross validation with 50 iterations to get smoother mean test and train\n# score curves, each time with 20% data randomly selected as a validation set.\ncv = ShuffleSplit(n_splits=3, test_size=0.2, random_state=0)\n# estimator = kNeighborsClassifier()\nestimator = imbpipeline(steps = [['smote', SMOTE(random_state=11)],    \n                                ['classifier', LogisticRegression(random_state=11,\n                                                                  max_iter=1000)]])\nplot_learning_curve(\n    estimator, title, X, y, axes=axes[:, 0], ylim=(0.5, 1.01), cv=cv, n_jobs=4\n)\n\n\ntitle = \"Learning Curves (Decision Tree)\"\ncv = ShuffleSplit(n_splits=3, test_size=0.2, random_state=0)\nestimator = imbpipeline(steps = [['smote', SMOTE(random_state=11)],    \n                                ['classifier', DecisionTreeClassifier()]])\nplot_learning_curve(\n    estimator, title, X, y, axes=axes[:, 1], ylim=(0.5, 1.01), cv=cv, n_jobs=4\n)\n\n\ntitle = \"Learning Curves (Support Vector Machine)\"\ncv = ShuffleSplit(n_splits=3, test_size=0.2, random_state=0)\nestimator = imbpipeline(steps = [['smote', SMOTE(random_state=11)],    \n                                ['classifier', SVC()]])\nplot_learning_curve(\n    estimator, title, X, y, axes=axes[:, 2], ylim=(0.5, 1.01), cv=cv, n_jobs=4\n)\n\nplt.show()\n \nfig, axes = plt.subplots(3, 3, figsize=(20, 20))\n\ntitle = \"Learning Curves (XGBoost)\"\n# Cross validation with 50 iterations to get smoother mean test and train\n# score curves, each time with 20% data randomly selected as a validation set.\ncv = ShuffleSplit(n_splits=3, test_size=0.2, random_state=0)\nestimator = imbpipeline(steps = [['smote', SMOTE(random_state=11)],    \n                                ['classifier', XGBClassifier(random_state=11)]])\nplot_learning_curve(\n    estimator, title, X, y, axes=axes[:, 0], ylim=(0.5, 1.01), cv=cv, n_jobs=4\n)\n\n\ntitle = \"Learning Curves (Random Forest)\"\ncv = ShuffleSplit(n_splits=3, test_size=0.2, random_state=0)\n#estimator = RandomForestClassifier()\nestimator = imbpipeline(steps = [['smote', SMOTE(random_state=11)],    \n                                ['classifier', RandomForestClassifier()]])\nplot_learning_curve(\n    estimator, title, X, y, axes=axes[:, 1], ylim=(0.5, 1.01), cv=cv, n_jobs=4\n)\n\n\ntitle = \"Learning Curves (LightGBM)\"\ncv = ShuffleSplit(n_splits=3, test_size=0.2, random_state=0)\n# estimator = LGBMClassifier()\nestimator = imbpipeline(steps = [['smote', SMOTE(random_state=11)],    \n                                ['classifier', LGBMClassifier()]])\nplot_learning_curve(\n    estimator, title, X, y, axes=axes[:, 2], ylim=(0.5, 1.01), cv=cv, n_jobs=4\n)\n\nplt.show()","metadata":{"id":"aCrAe0YpQzwN","outputId":"7e7ad7b4-e18f-40fc-b15e-321d066e4784"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Model Tuning**","metadata":{"id":"smIDWxsOIHpL"}},{"cell_type":"markdown","source":"**Grid Search**","metadata":{"id":"wyWrRWQLILSL"}},{"cell_type":"code","source":"import xgboost as xgb\nimport lightgbm as lgb","metadata":{"id":"nad6CJXJPktO"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Report performance\ndef Performance(actuals,predictions):\n  print('Accuracy: %.2f ' % accuracy_score(actuals, predictions))\n  print('MSE: %.2f ' % mean_squared_error(actuals, predictions))\n  print('MAE: %.2f ' % mean_absolute_error(actuals,predictions)) \n  print('R^2: %.2f' % r2_score(actuals, predictions))","metadata":{"id":"SaTQJHGqVHMJ"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**XGB**","metadata":{"id":"eNa9qxhVUbFx"}},{"cell_type":"code","source":"X_tr_std.columns","metadata":{"id":"TEnUHd6KMV03","outputId":"10961db3-9229-41f8-8b1d-2ce174a980ff"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = XGBClassifier()\n# define the grid of values to search\ngrid = dict()\ngrid['learning_rate']= [0.05,0.06, 0.07,0.08]\ngrid['max_depth']= [5,6,7,8]\ngrid['n_estimators'] = [100,150,200,250]\ncv = RepeatedKFold(n_splits=4, n_repeats=2, random_state=192837465)\ngrid_search = GridSearchCV(estimator=model, param_grid=grid, n_jobs=-1, cv=cv, scoring='neg_mean_squared_error')\ngrid_result = grid_search.fit(X_tr_std, y_tr_SMOTE)\nprint(\"Best: %f using %s\" % (grid_result.best_score_, grid_result.best_params_))\nmeans = grid_result.cv_results_['mean_test_score']\nstds = grid_result.cv_results_['std_test_score']\nparams = grid_result.cv_results_['params']\nfor mean, stdev, param in zip(means, stds, params):\n    print(\"%f (%f) with: %r\" % (absolute(mean), stdev, param))","metadata":{"id":"phlGFx0CPA1i","outputId":"e8279cee-39f3-433d-c8fe-fc6d0a8d603e"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test=pd.DataFrame(y_test)","metadata":{"id":"T-HINesOV4Uy"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"XGB_model_tuned = XGBClassifier(learning_rate=0.05,max_depth=6,n_estimators=200)\nXGB_model_tuned.fit(X_tr_std, np.ravel(y_tr_SMOTE))\n# Make predictions \ny_pred = XGB_model_tuned.predict(X_te_std)\ny_pred = pd.DataFrame(y_pred)\n# Performance\nPerformance(y_test,y_pred)\n# Worst-case instance prediction\ny_pred = pd.DataFrame(y_pred) \nresults = pd.concat([y_pred, y_test.set_index(y_pred.index)], axis=1)\nresults.columns=['Pred','Act']\nresults['error'] = (results.Pred - results.Act)/results.Act\nprint('Maximum error: ',100*np.max(results.error),'%\\n\\n')","metadata":{"id":"cxQfCSZ1UM8r","outputId":"580c4aaf-97ed-45ac-9fe0-4401de57d605"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_score(y_test,y_pred)","metadata":{"id":"_tbSVwpqNoxE","outputId":"42d5a589-d244-40aa-87aa-d474109d3f9a"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Light GBM**","metadata":{"id":"bXMEBnM5WERb"}},{"cell_type":"code","source":"# Gridsearch the Light LGB hyperparameter space\nmodel = LGBMClassifier(boosting_type='gbdt', objective='binary')\n# define the grid of values to search\ngrid = dict()\ngrid['learning_rate']= [0.05,0.06, 0.07,0.08]\ngrid['num_leaves']= [10,15,20,25]\ngrid['reg_lambda'] = [0.08,0.1,0.12]\ncv = RepeatedKFold(n_splits=4, n_repeats=2, random_state=192837465)\ngrid_search = GridSearchCV(estimator=model, param_grid=grid, n_jobs=-1, cv=cv, scoring='neg_mean_squared_error')\ngrid_result = grid_search.fit(X_tr_std, y_tr_SMOTE)\nprint(\"Best: %f using %s\" % (grid_result.best_score_, grid_result.best_params_))\nmeans = grid_result.cv_results_['mean_test_score']\nstds = grid_result.cv_results_['std_test_score']\nparams = grid_result.cv_results_['params']\nfor mean, stdev, param in zip(means, stds, params):\n    print(\"%f (%f) with: %r\" % (absolute(mean), stdev, param))","metadata":{"id":"fDCiVoehWIeu","outputId":"df8bcd7f-0b29-44ae-ecd6-a868de6f0bbc"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LGB_model_tuned = LGBMClassifier(boosting_type='gbdt', objective='binary',learning_rate=0.06,num_leaves=15,reg_lambda=0.05)\nLGB_model_tuned.fit(X_tr_std, np.ravel(y_tr_SMOTE))\n# Make predictions \ny_pred = LGB_model_tuned.predict(X_te_std)\ny_pred = pd.DataFrame(y_pred)\n# Performance\nPerformance(y_test,y_pred)\n# Worst-case instance prediction\ny_pred = pd.DataFrame(y_pred) \nresults = pd.concat([y_pred, y_test.set_index(y_pred.index)], axis=1)\nresults.columns=['Pred','Act']\nresults['error'] = (results.Pred - results.Act)/results.Act\nprint('Maximum error: ',100*np.max(results.error),'%\\n\\n')","metadata":{"id":"KaG2pM8maPeh","outputId":"f3fa7831-02b8-41ea-b133-017509246fb8"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_score(y_test,y_pred)","metadata":{"id":"oOZlKKJ5Np1J","outputId":"26bedd33-c1d4-4eb5-9560-d9d91c7d5a27"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Prediction**","metadata":{"id":"H3HOddIb1oyr"}},{"cell_type":"markdown","source":"**With XBG**","metadata":{"id":"j3ufiJNwFD5X"}},{"cell_type":"code","source":"# processed test data is x_te_std with 4277 rows\nx_te_std","metadata":{"id":"VXv35q381zeU","outputId":"774671b6-7ea1-4ec8-b087-c4f357edbc12"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The final model we chose is XGBClassifier: \n# XGB_model_tuned = XGBClassifier(learning_rate=0.07,max_depth=5,n_estimators=200)\n# Make predictions \nprediction1 = XGB_model_tuned.predict(x_te_std)\nprediction1 = pd.DataFrame(prediction1,columns=['Transported'])\nprediction1['Transported'] = prediction1['Transported'].map({1: True, 0: False})\n\ndrive.mount('/content/drive')\nfile_ = \"/content/drive/MyDrive/ML/wow/test.csv\" \ntest = pd.read_csv(file_)\ntest=test['PassengerId']\nprediction1=pd.concat([test,prediction1],axis=1)\nprediction1.set_index(['PassengerId'],inplace=True)\nprediction1","metadata":{"id":"4Py831Xw2KX8","outputId":"6d82826f-36dd-48fc-ed06-c029e108ae1f"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**With Light GBM**","metadata":{"id":"7bxNAApzFHUy"}},{"cell_type":"code","source":"#LGB_model_tuned = LGBMClassifier(boosting_type='gbdt', objective='binary',learning_rate=0.055,num_leaves=15,reg_lambda=0.05)\n\n# Make predictions \nprediction2 = LGB_model_tuned.predict(x_te_std)\nprediction2 = pd.DataFrame(prediction2,columns=['Transported'])\nprediction2['Transported'] = prediction2['Transported'].map({1: True, 0: False})\n\nfile_ = \"/content/drive/MyDrive/ML/wow/test.csv\" \ntest = pd.read_csv(file_)\ntest=test['PassengerId']\nprediction2=pd.concat([test,prediction2],axis=1)\nprediction2.set_index(['PassengerId'],inplace=True)\nprediction2","metadata":{"id":"BncEHnyoFL2M","outputId":"4e85d14a-638d-47d1-cd90-6f814e269bfb"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TPOT**","metadata":{"id":"u2sNMsY2sIfT"}},{"cell_type":"code","source":"#!pip install tpot","metadata":{"id":"YuBKUrUqsBFK"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tree-based pipeline from TPOT: credits: http://automl.info/tpot/\n# import the AutoMLpackage after installing tpot.\nimport tpot\n# import other necessary packages.\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.model_selection import RepeatedKFold\nfrom tpot import TPOTClassifier","metadata":{"id":"Bd4Rnd5BsC9k"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let TPOT build a model!\ncv = RepeatedKFold(n_splits=10, n_repeats=3, random_state=1)\nclf = TPOTClassifier(generations=5, population_size=50, cv=cv, scoring='roc_auc', verbosity=3, random_state=1, n_jobs=-1)\nclf.fit(X_tr_std, y_tr_SMOTE)\n# evaluate best model\ny_pred = clf.predict(X_te_std)\nprint(f'AUC: {roc_auc_score(y_test, y_pred)}')\nprint(f'Accuracy: {clf.score(X_te_std, y_test)}')","metadata":{"id":"VVno77xrsEcq"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction3 = clf.predict(x_te_std)\nprediction3 = pd.DataFrame(prediction3,columns=['Transported'])\nprediction3['Transported'] = prediction3['Transported'].map({1: True, 0: False})\n\ndrive.mount('/content/drive')\nfile_ = \"/content/drive/MyDrive/ML/wow/test.csv\" \ntest = pd.read_csv(file_)\ntest=test['PassengerId']\nprediction3=pd.concat([test,prediction3],axis=1)\nprediction3.set_index(['PassengerId'],inplace=True)\nprediction3","metadata":{"id":"MIJ5tb3osG0T"},"execution_count":null,"outputs":[]}]}