{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-13T15:57:02.113960Z","iopub.execute_input":"2022-07-13T15:57:02.114833Z","iopub.status.idle":"2022-07-13T15:57:02.151729Z","shell.execute_reply.started":"2022-07-13T15:57:02.114712Z","shell.execute_reply":"2022-07-13T15:57:02.150903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_style(\"darkgrid\")\n\nfrom sklearn.impute import SimpleImputer\nfrom sklearn.preprocessing import StandardScaler, MinMaxScaler, LabelEncoder\n\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.pipeline import Pipeline\n\nfrom sklearn.dummy import DummyClassifier\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.svm import SVC\nimport lightgbm as lgbm \n\nfrom sklearn.metrics import confusion_matrix, f1_score, auc, accuracy_score","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:02.153388Z","iopub.execute_input":"2022-07-13T15:57:02.153991Z","iopub.status.idle":"2022-07-13T15:57:05.158155Z","shell.execute_reply.started":"2022-07-13T15:57:02.153958Z","shell.execute_reply":"2022-07-13T15:57:05.156829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load DataSet\ntrain_df =  pd.read_csv(\"/kaggle/input/spaceship-titanic/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/spaceship-titanic/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:05.159986Z","iopub.execute_input":"2022-07-13T15:57:05.160829Z","iopub.status.idle":"2022-07-13T15:57:05.263374Z","shell.execute_reply.started":"2022-07-13T15:57:05.160782Z","shell.execute_reply":"2022-07-13T15:57:05.261910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:05.266766Z","iopub.execute_input":"2022-07-13T15:57:05.267270Z","iopub.status.idle":"2022-07-13T15:57:05.299589Z","shell.execute_reply.started":"2022-07-13T15:57:05.267226Z","shell.execute_reply":"2022-07-13T15:57:05.298186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Overview of the data:\ndef overview_data(df):\n    print(\"---------\")\n    print(\"Shape: \", df.shape)\n    print(\"Number of missing values :\", df.isna().sum().sum())\n    if df.isna().sum().sum() > 0:\n        missing = df.columns[np.where(df.isna().sum(),True, False)]\n        print(\"Missing columns are :\", missing)\n        print(\"Count of missing columns: \", len(missing),\"/\", len(df.columns))\n    print(\"Number of duplicate values :\", df.duplicated().sum())\n    \noverview_data(train_df)\noverview_data(test_df)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:05.301672Z","iopub.execute_input":"2022-07-13T15:57:05.302157Z","iopub.status.idle":"2022-07-13T15:57:05.373148Z","shell.execute_reply.started":"2022-07-13T15:57:05.302114Z","shell.execute_reply":"2022-07-13T15:57:05.371799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find % missing data per columns\ndef missing_data_perc(df):\n    print(\"---------\")\n    missing = df.columns[np.where(df.isna().sum(),True, False)]\n    missing_count = df[missing].isna().sum() / df.shape[0] * 100\n    sns.barplot(y = missing, x = missing_count, order = missing_count.sort_values().index)\n    plt.show()\n    \n#     print(\"Missing columns are :\", missing)\n#     print(\"Count of missing columns: \", len(missing),\"/\", len(df.columns))\n\nmissing_data_perc(train_df)\nmissing_data_perc(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:05.375317Z","iopub.execute_input":"2022-07-13T15:57:05.376151Z","iopub.status.idle":"2022-07-13T15:57:06.060654Z","shell.execute_reply.started":"2022-07-13T15:57:05.376102Z","shell.execute_reply":"2022-07-13T15:57:06.059813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Statistical Overview of the data:\n\ndef stats_df (df):\n    print(\"---------\")\n    display(df.describe().T)\n    display(df.dtypes.value_counts())\n    display(df.info())\n    \n    \nstats_df(train_df)\nstats_df(test_df)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:06.061768Z","iopub.execute_input":"2022-07-13T15:57:06.062517Z","iopub.status.idle":"2022-07-13T15:57:06.198651Z","shell.execute_reply.started":"2022-07-13T15:57:06.062475Z","shell.execute_reply":"2022-07-13T15:57:06.197274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. Median of almost all cont variable is 0. Should check the % of data = 0\n2. Roomservice, FoodCourt... is the amount the passenger has billed at each of the Spaceship Titanic's  many luxury amenities. Need to remember this during imputation.","metadata":{}},{"cell_type":"code","source":"train_df.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:06.200364Z","iopub.execute_input":"2022-07-13T15:57:06.201213Z","iopub.status.idle":"2022-07-13T15:57:06.211392Z","shell.execute_reply.started":"2022-07-13T15:57:06.201168Z","shell.execute_reply":"2022-07-13T15:57:06.209927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"categorical_vars = [\"HomePlanet\", \"CryoSleep\", \"Cabin\", \"Destination\", \"VIP\", \"Name\"]\ncont_vars = [\"Age\",\"RoomService\", \"FoodCourt\", \"ShoppingMall\", \"VRDeck\",\"Spa\"]\nluxury_vars = [\"RoomService\", \"FoodCourt\", \"ShoppingMall\", \"VRDeck\",\"Spa\"]\n\n\nX = train_df.drop([\"PassengerId\"], axis = 1)\n\nlabel_enc = LabelEncoder()\nX[\"Transported\"] = label_enc.fit_transform(X[\"Transported\"])\n\ny = X[\"Transported\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:06.213446Z","iopub.execute_input":"2022-07-13T15:57:06.213899Z","iopub.status.idle":"2022-07-13T15:57:06.227020Z","shell.execute_reply.started":"2022-07-13T15:57:06.213854Z","shell.execute_reply":"2022-07-13T15:57:06.225610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess \"Cabin\" Deck/Num/Side \n# \nname_vars = ['Name_length_of_first_name','Name_length_of_last_name','Name_unique_chars']\n\ndef unique_chars(x):\n    if pd.isna(x):\n        return 0\n    return len(set(x))\n\ndef preprocess_X(df_X, train = True, drop = True):\n    \n    df_X[[\"Cabin_Deck\", \"Cabin_Num\", \"Cabin_Side\"]] = df_X[\"Cabin\"].str.split(\"/\", expand = True)\n    df_X[\"Name_length_of_first_name\"] = df_X[\"Name\"].str.split(\" \", expand = True)[0].str.len()\n    df_X[\"Name_length_of_last_name\"] = df_X[\"Name\"].str.split(\" \", expand = True)[1].str.len()\n    df_X[\"Name_unique_chars\"] = df_X[\"Name\"].apply(lambda x: unique_chars(x))\n    \n\n    \n    if drop:\n        df_X.drop([\"Cabin\", \"Name\"], axis = 1, inplace = True)\n    \n    if train:\n        categorical_vars.remove(\"Cabin\")\n        categorical_vars.extend([\"Cabin_Deck\", \"Cabin_Side\"])\n    \n        categorical_vars.remove(\"Name\")\n        cont_vars.append(\"Cabin_Num\")\n        cont_vars.extend(name_vars)\n    \n    \n        \n    df_X[\"Cabin_Num\"] = pd.to_numeric(df_X[\"Cabin_Num\"])\n    df_X[categorical_vars] = df_X[categorical_vars].astype('category')\n    \n    return df_X\n","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:06.232536Z","iopub.execute_input":"2022-07-13T15:57:06.233675Z","iopub.status.idle":"2022-07-13T15:57:06.244753Z","shell.execute_reply.started":"2022-07-13T15:57:06.233627Z","shell.execute_reply":"2022-07-13T15:57:06.243662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = preprocess_X(X, train  = True, drop = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:06.245926Z","iopub.execute_input":"2022-07-13T15:57:06.246859Z","iopub.status.idle":"2022-07-13T15:57:06.484970Z","shell.execute_reply.started":"2022-07-13T15:57:06.246822Z","shell.execute_reply":"2022-07-13T15:57:06.483916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:06.487335Z","iopub.execute_input":"2022-07-13T15:57:06.487725Z","iopub.status.idle":"2022-07-13T15:57:06.517421Z","shell.execute_reply.started":"2022-07-13T15:57:06.487691Z","shell.execute_reply":"2022-07-13T15:57:06.516161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X.dtypes","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:06.519196Z","iopub.execute_input":"2022-07-13T15:57:06.519919Z","iopub.status.idle":"2022-07-13T15:57:06.531207Z","shell.execute_reply.started":"2022-07-13T15:57:06.519869Z","shell.execute_reply":"2022-07-13T15:57:06.529970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## EDA\n\n1. Distribution of the target variables.\n2. Violin Plot for continous variables.\n3. Zero_Rate for luxury continous variables.\n4. Passenger that didn't avail any luxury.\n5. Count plot for categorical variables.\n6. KDE Plot for name based created variable\n7. ","metadata":{}},{"cell_type":"markdown","source":"#### Univariate:","metadata":{}},{"cell_type":"code","source":"print(y.value_counts(normalize=True))\nsns.countplot(x = X[\"Transported\"])\n\n# Distribution is levelled","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:06.533120Z","iopub.execute_input":"2022-07-13T15:57:06.533850Z","iopub.status.idle":"2022-07-13T15:57:06.701962Z","shell.execute_reply.started":"2022-07-13T15:57:06.533805Z","shell.execute_reply":"2022-07-13T15:57:06.700944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Continous variable:\n\nplot, axes = plt.subplots(nrows = 2, ncols = 5, figsize = (25,10))\n\nfor i in range(len(cont_vars)):\n    sns.violinplot(x = cont_vars[i], data = X,\n                   ax = axes[i//5][i%5]).set(title='Median = {}'.format(np.nanmedian(X[cont_vars[i]])))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:06.703692Z","iopub.execute_input":"2022-07-13T15:57:06.704257Z","iopub.status.idle":"2022-07-13T15:57:08.260836Z","shell.execute_reply.started":"2022-07-13T15:57:06.704223Z","shell.execute_reply":"2022-07-13T15:57:08.259662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Zero rate\n\nX_zero_rate = X[luxury_vars] == 0\nplot, axes = plt.subplots(nrows = 1, ncols = 5, figsize = (20,6))\n\nfor i in range(len(luxury_vars)):\n    cal = X_zero_rate[luxury_vars[i]].value_counts(normalize=True)\n    sns.countplot(x = X_zero_rate[luxury_vars[i]], ax = axes[i%5]).set(title='Zero_rate = {:.2f}'.format(cal.loc[True]))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:08.262126Z","iopub.execute_input":"2022-07-13T15:57:08.262560Z","iopub.status.idle":"2022-07-13T15:57:08.787510Z","shell.execute_reply.started":"2022-07-13T15:57:08.262516Z","shell.execute_reply":"2022-07-13T15:57:08.786625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All Luxury varible have nearly equal % of zero values","metadata":{}},{"cell_type":"code","source":"X_zero_rate[\"all_zero\"] = X_zero_rate.apply(lambda x: np.all(x), axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:08.788637Z","iopub.execute_input":"2022-07-13T15:57:08.789838Z","iopub.status.idle":"2022-07-13T15:57:09.145229Z","shell.execute_reply.started":"2022-07-13T15:57:08.789797Z","shell.execute_reply":"2022-07-13T15:57:09.143968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"zero_rate = X_zero_rate[\"all_zero\"].value_counts(normalize=True).loc[True]\nsns.countplot(x = X_zero_rate[\"all_zero\"], hue = y).set(title = \"Passenger that didn't avail a single luxury = {}\".format(zero_rate))\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:09.146432Z","iopub.execute_input":"2022-07-13T15:57:09.146735Z","iopub.status.idle":"2022-07-13T15:57:09.352267Z","shell.execute_reply.started":"2022-07-13T15:57:09.146709Z","shell.execute_reply":"2022-07-13T15:57:09.351052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X[\"any_luxury\"] = X_zero_rate[\"all_zero\"].astype(\"category\")\ncategorical_vars.append(\"any_luxury\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:09.353392Z","iopub.execute_input":"2022-07-13T15:57:09.353693Z","iopub.status.idle":"2022-07-13T15:57:09.360618Z","shell.execute_reply.started":"2022-07-13T15:57:09.353666Z","shell.execute_reply":"2022-07-13T15:57:09.359817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X[\"any_luxury\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:09.361834Z","iopub.execute_input":"2022-07-13T15:57:09.362979Z","iopub.status.idle":"2022-07-13T15:57:09.379218Z","shell.execute_reply.started":"2022-07-13T15:57:09.362944Z","shell.execute_reply":"2022-07-13T15:57:09.377978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Categorical variable\n\nplot, axes = plt.subplots(nrows = 3, ncols = 3, figsize = (15,10))\n\nfor i in range(len(categorical_vars)):\n    sns.countplot(x = categorical_vars[i], data = X, ax = axes[i//3][i%3])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:09.380806Z","iopub.execute_input":"2022-07-13T15:57:09.381927Z","iopub.status.idle":"2022-07-13T15:57:10.397955Z","shell.execute_reply.started":"2022-07-13T15:57:09.381882Z","shell.execute_reply":"2022-07-13T15:57:10.397122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1. CryoSleep, VIP, Cabin_side are boolean categorical variable. \n2. HomePlanet, Destination, Cabin_Deck are multi category categorical variable.","metadata":{}},{"cell_type":"markdown","source":"1. Cabin can be split up into three different variable : deck/num/side","metadata":{}},{"cell_type":"code","source":"plot, axes = plt.subplots(nrows = 1, ncols = 3, figsize = (15,4))\n\nfor i in range(len(name_vars)):\n    sns.kdeplot(x = name_vars[i], data = X, ax = axes[i%3])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:10.399216Z","iopub.execute_input":"2022-07-13T15:57:10.399731Z","iopub.status.idle":"2022-07-13T15:57:11.141202Z","shell.execute_reply.started":"2022-07-13T15:57:10.399700Z","shell.execute_reply":"2022-07-13T15:57:11.140087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Bivariate Analysis\n1. All univariate analysis wrt target\n2. Continous varaible with hue of categories\n3.","metadata":{}},{"cell_type":"code","source":"X.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:11.143097Z","iopub.execute_input":"2022-07-13T15:57:11.143899Z","iopub.status.idle":"2022-07-13T15:57:11.152093Z","shell.execute_reply.started":"2022-07-13T15:57:11.143846Z","shell.execute_reply":"2022-07-13T15:57:11.150787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Continous variable with target\n\nplot, axes = plt.subplots(nrows = 3, ncols = 4, figsize = (19,15))\n\nfor i in range(len(cont_vars)):\n    sns.violinplot(y = X[cont_vars[i]], x = y, hue = X[\"any_luxury\"],split = True, ax = axes[i//4][i%4])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:11.153830Z","iopub.execute_input":"2022-07-13T15:57:11.154254Z","iopub.status.idle":"2022-07-13T15:57:13.063900Z","shell.execute_reply.started":"2022-07-13T15:57:11.154216Z","shell.execute_reply":"2022-07-13T15:57:13.062536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Categorical variable with target\n\nplot, axes = plt.subplots(nrows = 3, ncols = 3, figsize = (15,10))\n\nfor i in range(len(categorical_vars)):\n    sns.countplot(x = categorical_vars[i], hue = \"Transported\", data = X, ax = axes[i//3][i%3])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:13.065519Z","iopub.execute_input":"2022-07-13T15:57:13.065867Z","iopub.status.idle":"2022-07-13T15:57:14.262879Z","shell.execute_reply.started":"2022-07-13T15:57:13.065834Z","shell.execute_reply":"2022-07-13T15:57:14.262021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot, axes = plt.subplots(nrows = 1, ncols = 3, figsize = (15,5))\n\nfor i in range(len(name_vars)):\n    sns.countplot(x = X[name_vars[i]], hue = y, ax = axes[i%3])","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:14.264169Z","iopub.execute_input":"2022-07-13T15:57:14.264690Z","iopub.status.idle":"2022-07-13T15:57:15.109013Z","shell.execute_reply.started":"2022-07-13T15:57:14.264658Z","shell.execute_reply":"2022-07-13T15:57:15.108117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.pairplot(X)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:57:15.110336Z","iopub.execute_input":"2022-07-13T15:57:15.110867Z","iopub.status.idle":"2022-07-13T15:58:21.212781Z","shell.execute_reply.started":"2022-07-13T15:57:15.110834Z","shell.execute_reply":"2022-07-13T15:58:21.211277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Feature Correlation","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(12,7))\nsns.heatmap(X.corr().round(2), fmt = \"g\", cmap = \"Greens\", annot = True)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:58:21.217736Z","iopub.execute_input":"2022-07-13T15:58:21.218145Z","iopub.status.idle":"2022-07-13T15:58:22.000944Z","shell.execute_reply.started":"2022-07-13T15:58:21.218111Z","shell.execute_reply":"2022-07-13T15:58:22.000065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = X.drop([\"Transported\"], axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:58:22.002201Z","iopub.execute_input":"2022-07-13T15:58:22.003007Z","iopub.status.idle":"2022-07-13T15:58:22.008219Z","shell.execute_reply.started":"2022-07-13T15:58:22.002975Z","shell.execute_reply":"2022-07-13T15:58:22.007456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocessing","metadata":{}},{"cell_type":"code","source":"imputer = SimpleImputer(strategy = \"mean\")\n\n\nX_std = X.copy(deep = True)\nX_std[cont_vars] = imputer.fit_transform(X_std[cont_vars])\n# s_scaler = StandardScaler()\n\n# X_std[cont_vars] = s_scaler.fit_transform(X[cont_vars])\nX_std= pd.get_dummies(X_std)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:58:22.009401Z","iopub.execute_input":"2022-07-13T15:58:22.009704Z","iopub.status.idle":"2022-07-13T15:58:22.041790Z","shell.execute_reply.started":"2022-07-13T15:58:22.009670Z","shell.execute_reply":"2022-07-13T15:58:22.040942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed = 42\nnp.seed = seed\n\n\nmodels = {\n    \"dummy\": DummyClassifier(),\n#     \"svc\": SVC(random_state=seed, C = 1.34, degree = 4),\n    \"lgbm\" : lgbm.LGBMClassifier(objective=\"binary\", n_estimators=5000,random_state=seed)\n    \n#     \"logistic\": LogisticRegression(random_state=seed, C = 1.5, max_iter=10000),\n#     \"decision_trees\": DecisionTreeClassifier(random_state=seed, ),\n#     \"random_forest\" : RandomForestClassifier(random_state=seed),\n#     \"knn\": KNeighborsClassifier(n_neighbors=5),\n#     \"MLP\" : MLPClassifier(random_state=seed, max_iter=10000, learning_rate_init=0.001, learning_rate=\"invscaling\")\n}","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:58:22.043287Z","iopub.execute_input":"2022-07-13T15:58:22.043889Z","iopub.status.idle":"2022-07-13T15:58:22.049285Z","shell.execute_reply.started":"2022-07-13T15:58:22.043855Z","shell.execute_reply":"2022-07-13T15:58:22.048236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modelling","metadata":{}},{"cell_type":"code","source":"NSPLITS = 7","metadata":{"execution":{"iopub.status.busy":"2022-07-13T15:58:22.050973Z","iopub.execute_input":"2022-07-13T15:58:22.051739Z","iopub.status.idle":"2022-07-13T15:58:22.067442Z","shell.execute_reply.started":"2022-07-13T15:58:22.051707Z","shell.execute_reply":"2022-07-13T15:58:22.066419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_model(X, y,):\n    print(X.columns, len(X.columns))\n    X = X.values\n    \n    best_model = None\n    max_score = 0.0\n    \n    kf = StratifiedKFold(n_splits=NSPLITS, shuffle=True, random_state=seed)\n    for name, model in models.items():\n        train_score = []\n        print(name)\n        for train_idx, test_idx in kf.split(X,y):\n            X_train, X_test = X[train_idx], X[test_idx]\n            y_train, y_test = y[train_idx], y[test_idx]\n            \n            if name == \"lgbm\":\n                eval_set = [(X_test, y_test)]\n                model.fit(X_train, y_train,\n                          eval_set = eval_set, early_stopping_rounds=150, verbose=0, )\n                \n            else:\n                model.fit(X_train, y_train)\n            \n            y_preds = model.predict(X_test)\n            train_score.append(accuracy_score(y_test, y_preds))\n        \n        print(train_score)\n        final_score = sum(train_score)/NSPLITS\n        \n        \n        if final_score > max_score:\n            best_model = model\n            max_score = final_score\n                    \n        print(\"[SCORE]\", final_score)\n    return best_model","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:09:20.054807Z","iopub.execute_input":"2022-07-13T16:09:20.056347Z","iopub.status.idle":"2022-07-13T16:09:20.069629Z","shell.execute_reply.started":"2022-07-13T16:09:20.056293Z","shell.execute_reply":"2022-07-13T16:09:20.068529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_model = run_model(X_std,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:09:20.824828Z","iopub.execute_input":"2022-07-13T16:09:20.825241Z","iopub.status.idle":"2022-07-13T16:09:23.489966Z","shell.execute_reply.started":"2022-07-13T16:09:20.825206Z","shell.execute_reply":"2022-07-13T16:09:23.489092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"LGBM is best.","metadata":{}},{"cell_type":"code","source":"y_preds = best_model.predict(X_std.values)\nprint(best_model.score(X_std.values, y))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:10:35.750496Z","iopub.execute_input":"2022-07-13T16:10:35.751293Z","iopub.status.idle":"2022-07-13T16:10:35.813828Z","shell.execute_reply.started":"2022-07-13T16:10:35.751249Z","shell.execute_reply":"2022-07-13T16:10:35.812942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x = y_preds)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:10:36.622317Z","iopub.execute_input":"2022-07-13T16:10:36.623339Z","iopub.status.idle":"2022-07-13T16:10:36.790298Z","shell.execute_reply.started":"2022-07-13T16:10:36.623300Z","shell.execute_reply":"2022-07-13T16:10:36.789178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Error Analysis:","metadata":{}},{"cell_type":"code","source":"mat = confusion_matrix(y, y_preds)\nsns.heatmap(mat, annot = True, fmt = \"g\", cmap = \"Greens\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:10:43.762733Z","iopub.execute_input":"2022-07-13T16:10:43.763121Z","iopub.status.idle":"2022-07-13T16:10:43.998882Z","shell.execute_reply.started":"2022-07-13T16:10:43.763091Z","shell.execute_reply":"2022-07-13T16:10:43.998058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feat_imp = pd.Series(best_model.feature_importances_, index=X_std.columns)\nfeat_imp.nlargest(30).plot(kind='barh', figsize=(8,10))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:10:45.021806Z","iopub.execute_input":"2022-07-13T16:10:45.022920Z","iopub.status.idle":"2022-07-13T16:10:45.435473Z","shell.execute_reply.started":"2022-07-13T16:10:45.022881Z","shell.execute_reply":"2022-07-13T16:10:45.434327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"HyperParamter Tuning","metadata":{}},{"cell_type":"code","source":"from optuna.integration import LightGBMPruningCallback\nimport optuna\nfrom sklearn.metrics import log_loss\n\nfrom warnings import simplefilter\nsimplefilter(\"ignore\", category=RuntimeWarning)\n\ndef objective(trial, X, y):\n    param_grid = {\n        #         \"device_type\": trial.suggest_categorical(\"device_type\", ['gpu']),\n                    \"n_estimators\": trial.suggest_categorical(\"n_estimators\", [10000]),\n                    \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.3),\n                    \"num_leaves\": trial.suggest_int(\"num_leaves\", 20, 3000, step=20),\n                    \"max_depth\": trial.suggest_int(\"max_depth\", 3, 12),\n                    \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 200, 10000, step=100),\n                    \"max_bin\": trial.suggest_int(\"max_bin\", 200, 300),\n                    \"lambda_l1\": trial.suggest_int(\"lambda_l1\", 0, 100, step=5),\n                    \"lambda_l2\": trial.suggest_int(\"lambda_l2\", 0, 100, step=5),\n                    \"min_gain_to_split\": trial.suggest_float(\"min_gain_to_split\", 0, 15),\n                    \"bagging_fraction\": trial.suggest_float(\n                        \"bagging_fraction\", 0.2, 0.95, step=0.1\n                    ),\n                    \"bagging_freq\": trial.suggest_categorical(\"bagging_freq\", [1]),\n                    \"feature_fraction\": trial.suggest_float(\n                        \"feature_fraction\", 0.2, 0.95, step=0.1\n                    ),\n    }\n           \n    cv = StratifiedKFold(n_splits=NSPLITS, shuffle=True, random_state=seed)\n\n    cv_scores = np.empty(NSPLITS)\n    for idx, (train_idx, test_idx) in enumerate(cv.split(X, y)):\n        X_train, X_test = X.iloc[train_idx], X.iloc[test_idx]\n        y_train, y_test = y[train_idx], y[test_idx]\n\n        model = lgbm.LGBMClassifier(objective=\"binary\", **param_grid)\n        model.fit( X_train, y_train, eval_set=[(X_test, y_test)], eval_metric=\"binary_logloss\", early_stopping_rounds=100, verbose= 0, \n                  callbacks=[LightGBMPruningCallback(trial, \"binary_logloss\")])\n        preds = model.predict_proba(X_test)\n        cv_scores[idx] = log_loss(y_test, preds)\n\n    return np.mean(cv_scores)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:14:10.561970Z","iopub.execute_input":"2022-07-13T16:14:10.563015Z","iopub.status.idle":"2022-07-13T16:14:10.576174Z","shell.execute_reply.started":"2022-07-13T16:14:10.562976Z","shell.execute_reply":"2022-07-13T16:14:10.574772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study = optuna.create_study(direction=\"minimize\", study_name=\"LGBM Classifier\")\nfunc = lambda trial: objective(trial, X, y)\nstudy.optimize(func, n_trials=20)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:14:21.672527Z","iopub.execute_input":"2022-07-13T16:14:21.672911Z","iopub.status.idle":"2022-07-13T16:14:37.368661Z","shell.execute_reply.started":"2022-07-13T16:14:21.672878Z","shell.execute_reply":"2022-07-13T16:14:37.367600Z"},"scrolled":true,"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"\\tBest value (rmse): {study.best_value:.5f}\")\nprint(f\"\\tBest params:\")\n\nfor key, value in study.best_params.items():\n    print(f\"\\t\\t{key}: {value}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:14:57.365615Z","iopub.execute_input":"2022-07-13T16:14:57.365980Z","iopub.status.idle":"2022-07-13T16:14:57.375278Z","shell.execute_reply.started":"2022-07-13T16:14:57.365951Z","shell.execute_reply":"2022-07-13T16:14:57.373728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study.best_params","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:15:05.380321Z","iopub.execute_input":"2022-07-13T16:15:05.380715Z","iopub.status.idle":"2022-07-13T16:15:05.391213Z","shell.execute_reply.started":"2022-07-13T16:15:05.380685Z","shell.execute_reply":"2022-07-13T16:15:05.389353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models[\"lgbm\"].set_params(**study.best_params)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:16:03.574215Z","iopub.execute_input":"2022-07-13T16:16:03.574586Z","iopub.status.idle":"2022-07-13T16:16:03.585549Z","shell.execute_reply.started":"2022-07-13T16:16:03.574557Z","shell.execute_reply":"2022-07-13T16:16:03.584362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models[\"lgbm\"].fit(X_std,y)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:18:20.131802Z","iopub.execute_input":"2022-07-13T16:18:20.132274Z","iopub.status.idle":"2022-07-13T16:18:25.144757Z","shell.execute_reply.started":"2022-07-13T16:18:20.132237Z","shell.execute_reply":"2022-07-13T16:18:25.143476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_preds = models[\"lgbm\"].predict(X_std.values)\nprint(models[\"lgbm\"].score(X_std.values, y))","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:19:08.105022Z","iopub.execute_input":"2022-07-13T16:19:08.106095Z","iopub.status.idle":"2022-07-13T16:19:08.162102Z","shell.execute_reply.started":"2022-07-13T16:19:08.106055Z","shell.execute_reply":"2022-07-13T16:19:08.161121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x = y_preds)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:19:56.282505Z","iopub.execute_input":"2022-07-13T16:19:56.282915Z","iopub.status.idle":"2022-07-13T16:19:56.454561Z","shell.execute_reply.started":"2022-07-13T16:19:56.282882Z","shell.execute_reply":"2022-07-13T16:19:56.453292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Submission:","metadata":{}},{"cell_type":"code","source":"X_test = test_df.copy(deep = True)\nX_test.drop([\"PassengerId\"], axis = 1, inplace = True)\nX_test[\"any_luxury\"] = (X_test[luxury_vars] == 0).apply(lambda x: np.all(x), axis = 1).astype(\"category\")\n\nX_test  = preprocess_X(X_test, train = False)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:20:04.080137Z","iopub.execute_input":"2022-07-13T16:20:04.080514Z","iopub.status.idle":"2022-07-13T16:20:04.315190Z","shell.execute_reply.started":"2022-07-13T16:20:04.080484Z","shell.execute_reply":"2022-07-13T16:20:04.313876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test[cont_vars] = imputer.transform(X_test[cont_vars])\n# # X_test[cont_vars] = s_scaler.transform(X_test[cont_vars])\n\nX_test_std = pd.get_dummies(X_test)\n# X_test_std = X_test_std.values","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:20:05.294907Z","iopub.execute_input":"2022-07-13T16:20:05.295282Z","iopub.status.idle":"2022-07-13T16:20:05.316446Z","shell.execute_reply.started":"2022-07-13T16:20:05.295253Z","shell.execute_reply":"2022-07-13T16:20:05.315256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test_std.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:20:06.654493Z","iopub.execute_input":"2022-07-13T16:20:06.654913Z","iopub.status.idle":"2022-07-13T16:20:06.662767Z","shell.execute_reply.started":"2022-07-13T16:20:06.654881Z","shell.execute_reply":"2022-07-13T16:20:06.661637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_preds = models[\"lgbm\"].predict(X_test_std)\ny_preds_label = label_enc.inverse_transform(y_preds)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:20:08.007803Z","iopub.execute_input":"2022-07-13T16:20:08.008635Z","iopub.status.idle":"2022-07-13T16:20:08.026874Z","shell.execute_reply.started":"2022-07-13T16:20:08.008596Z","shell.execute_reply":"2022-07-13T16:20:08.025963Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x = y_preds_label)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:20:10.194971Z","iopub.execute_input":"2022-07-13T16:20:10.195404Z","iopub.status.idle":"2022-07-13T16:20:10.368583Z","shell.execute_reply.started":"2022-07-13T16:20:10.195369Z","shell.execute_reply":"2022-07-13T16:20:10.367035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ids = test_df[\"PassengerId\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:20:14.774308Z","iopub.execute_input":"2022-07-13T16:20:14.774705Z","iopub.status.idle":"2022-07-13T16:20:14.779503Z","shell.execute_reply.started":"2022-07-13T16:20:14.774670Z","shell.execute_reply":"2022-07-13T16:20:14.778459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.DataFrame({\"PassengerId\" : ids, \"Transported\" : y_preds_label})","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:20:15.459282Z","iopub.execute_input":"2022-07-13T16:20:15.459985Z","iopub.status.idle":"2022-07-13T16:20:15.465245Z","shell.execute_reply.started":"2022-07-13T16:20:15.459948Z","shell.execute_reply":"2022-07-13T16:20:15.464454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv',index= False)","metadata":{"execution":{"iopub.status.busy":"2022-07-13T16:20:16.172482Z","iopub.execute_input":"2022-07-13T16:20:16.173182Z","iopub.status.idle":"2022-07-13T16:20:16.188172Z","shell.execute_reply.started":"2022-07-13T16:20:16.173145Z","shell.execute_reply":"2022-07-13T16:20:16.186646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}