{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#Libraries Required\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\nimport numpy as np  \nimport pandas as pd\nimport time\n\npd.set_option(\"display.max_rows\", 2000)\npd.set_option(\"display.max_columns\", 100)\nimport matplotlib.pyplot as plt\ncolor_pal = plt.rcParams['axes.prop_cycle'].by_key()['color']\nimport seaborn as sns\nimport plotly.express as px\nimport pickle \n\nimport datetime\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nfrom sklearn.preprocessing import LabelEncoder, StandardScaler\n\nfrom scipy.stats import shapiro, stats, probplot\nfrom sklearn.model_selection import GroupKFold, train_test_split, KFold, cross_validate\nfrom sklearn.metrics import roc_auc_score, f1_score, auc, roc_curve\nfrom hyperopt.pyll.base import scope\n\nfrom lightgbm import LGBMClassifier\nfrom xgboost import XGBClassifier\nfrom catboost import CatBoostClassifier\n\nfrom hyperopt import hp, fmin, tpe, STATUS_OK, Trials\nfrom hyperopt.fmin import fmin\nfrom hyperopt.pyll.stochastic import sample","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-09T06:34:52.864062Z","iopub.execute_input":"2022-08-09T06:34:52.864788Z","iopub.status.idle":"2022-08-09T06:34:56.434741Z","shell.execute_reply.started":"2022-08-09T06:34:52.864684Z","shell.execute_reply":"2022-08-09T06:34:56.433579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Table of Contents\n\n* [Competetion Description](#section-one)\n* [Sneak Peek into Data](#section-two)\n    - [Product](#subsection-one)\n    - [Bi-variate (TARGET) Analysis of Attribute](#subsection-two)\n    - [Measurement Columns Analysis](#subsection-three)\n    \n* [Ensembling with Hyperopt Search Algorithm](#section-three)\n* [Submission](#section-four)\n* [MORE to COME](#section-five)","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/train.csv')\ntest = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/test.csv')\nsubmission = pd.read_csv('/kaggle/input/tabular-playground-series-aug-2022/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-08-09T06:34:56.436508Z","iopub.execute_input":"2022-08-09T06:34:56.437587Z","iopub.status.idle":"2022-08-09T06:34:56.744505Z","shell.execute_reply.started":"2022-08-09T06:34:56.437546Z","shell.execute_reply":"2022-08-09T06:34:56.742975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-one\"></a>\n# Competetion Description\n\n1. For each product_code we are given a number of product attributes as well as a number of measurement values for each individual product, representing various lab testing methods. 2. Each product absorbs a certain amount of fluid (loading) to see whether or not it fails.\n\n3. Our task is to predict individual product failures of new codes with their individual lab test results.","metadata":{}},{"cell_type":"markdown","source":"<a id=\"section-two\"></a>\n# Sneak Peek into Data","metadata":{}},{"cell_type":"code","source":"## To create the DF summary\ndef summarytable(df):\n    print(f\"Dataset Shape: {df.shape}\")\n    summary = pd.DataFrame(df.dtypes,columns=['dtypes'])\n    summary = summary.reset_index()\n    summary['Name'] = summary['index']\n    summary = summary[['Name','dtypes']]\n    summary['Missing'] = df.isnull().sum().values    \n    summary['Missing_pct'] = df.isnull().mean().round(2).values\n    summary['Uniques'] = df.nunique().values\n    summary['First Value'] = df.loc[0].values\n    summary['Second Value'] = df.loc[1].values\n    summary['Third Value'] = df.loc[2].values\n    return summary\n\nprint(\"Displaying data ------>\")\ndisplay(summarytable(train))","metadata":{"execution":{"iopub.status.busy":"2022-08-09T06:34:56.746892Z","iopub.execute_input":"2022-08-09T06:34:56.747278Z","iopub.status.idle":"2022-08-09T06:34:56.851953Z","shell.execute_reply.started":"2022-08-09T06:34:56.747244Z","shell.execute_reply":"2022-08-09T06:34:56.850732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"👉 **Observations**\n\n- Only *measurement_X* variables have missing values, in the rangeof 1-9%.\n- There are only 2 unique values present for attribute_0 - material_5 & material_7.\n- There are only 3 unique values present for attribute_1 - 'material_8', 'material_5', 'material_6.","metadata":{}},{"cell_type":"code","source":"print(\"Displaying data ------>\")\ndisplay(summarytable(test))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T06:02:26.009616Z","iopub.execute_input":"2022-08-03T06:02:26.010499Z","iopub.status.idle":"2022-08-03T06:02:26.066867Z","shell.execute_reply.started":"2022-08-03T06:02:26.010444Z","shell.execute_reply":"2022-08-03T06:02:26.065650Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"product_code_diff = set(train['product_code'].unique()) - set(test['product_code'].unique())\nprint(f\"{product_code_diff} is not present in test dataset, while it's present in train dataset\")\n\nattribute_1_diff = set(train['attribute_1'].unique()) - set(test['attribute_1'].unique())\nprint(f\"{attribute_1_diff} is not present in test dataset, while it's present in train dataset\")","metadata":{"execution":{"iopub.status.busy":"2022-08-02T06:44:56.135473Z","iopub.execute_input":"2022-08-02T06:44:56.135884Z","iopub.status.idle":"2022-08-02T06:44:56.151682Z","shell.execute_reply.started":"2022-08-02T06:44:56.135852Z","shell.execute_reply":"2022-08-02T06:44:56.150753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"👉 Observations\n\n- From above, there are few features(product code and attribute_1) values which are not consistent across both train and test dataset.\n- We will see how to deal with them later.","metadata":{}},{"cell_type":"markdown","source":"<a id=\"subsection-one\"></a>\n## Product","metadata":{}},{"cell_type":"code","source":"train['product_code'].value_counts().plot(kind='barh', title='Product Count', backend='plotly')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T09:31:59.166061Z","iopub.execute_input":"2022-08-01T09:31:59.166824Z","iopub.status.idle":"2022-08-01T09:31:59.248476Z","shell.execute_reply.started":"2022-08-01T09:31:59.166780Z","shell.execute_reply":"2022-08-01T09:31:59.247612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"👉 **Observations**\n\nC is the most frequent available code, followed by E, B, D and A.","metadata":{}},{"cell_type":"code","source":"train.groupby('product_code')['loading'].mean().round(2).plot(kind='bar', title='Mean Loading by Product Code', backend='plotly')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T12:19:12.494163Z","iopub.execute_input":"2022-08-01T12:19:12.494934Z","iopub.status.idle":"2022-08-01T12:19:12.577657Z","shell.execute_reply.started":"2022-08-01T12:19:12.494882Z","shell.execute_reply":"2022-08-01T12:19:12.576560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.groupby('product_code')['loading'].median().round(2).plot(kind='bar', title='Median Loading by Product Code', backend='plotly')","metadata":{"execution":{"iopub.status.busy":"2022-08-01T12:21:42.451746Z","iopub.execute_input":"2022-08-01T12:21:42.452160Z","iopub.status.idle":"2022-08-01T12:21:42.534666Z","shell.execute_reply.started":"2022-08-01T12:21:42.452121Z","shell.execute_reply":"2022-08-01T12:21:42.533565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"👉 **Observations**\n\n1. **Mean** Loading of all the Product Codes are ~127\n2. **Median** Loading of all the Product Codes are ~122","metadata":{}},{"cell_type":"markdown","source":"<a id=\"subsection-two\"></a>\n## Bi-variate (TARGET) Analysis of Attribute","metadata":{}},{"cell_type":"code","source":"fig, axarr = plt.subplots(1, 4, figsize=(30, 8))\nplt.suptitle('Bi-variate (TARGET) Analysis of Attributes', fontsize=16)\nattributes = ['attribute_0', 'attribute_1','attribute_2', 'attribute_3']\n\nfor e, att in enumerate(attributes):\n    pd.crosstab(index=train[att],columns=train['failure'],normalize='index').round(2).plot.bar(ax=axarr[e], title=att)","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:58:26.067164Z","iopub.execute_input":"2022-08-01T10:58:26.067665Z","iopub.status.idle":"2022-08-01T10:58:27.768561Z","shell.execute_reply.started":"2022-08-01T10:58:26.067617Z","shell.execute_reply":"2022-08-01T10:58:27.767267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"👉 **Observations**\n\n1. **attribute_0** - **~80%** Failure Rate for both the materials.\n2. **attribute_1** - **~80%** Failure Rate for ALL the materials.\n3. **attribute_2** - **~80%** Failure Rate for ALL the materials, **7** has NO FAILURE DATA available.\n4. **attribute_3** - **~80%** Failure Rate for ALL the materials. **7** has NO FAILURE DATA available\n","metadata":{}},{"cell_type":"markdown","source":"<a id=\"subsection-three\"></a>\n# Measurement Columns Analysis","metadata":{}},{"cell_type":"code","source":"measure_cols = train.columns[train.columns.str.contains('measurement')].tolist()\n\ntrain[measure_cols].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-01T10:56:25.169425Z","iopub.execute_input":"2022-08-01T10:56:25.169899Z","iopub.status.idle":"2022-08-01T10:56:25.285050Z","shell.execute_reply.started":"2022-08-01T10:56:25.169862Z","shell.execute_reply":"2022-08-01T10:56:25.283748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corr_mat = train[measure_cols+['loading']].corr()\nmask = np.zeros_like(corr_mat, dtype=np.bool)\nmask[np.triu_indices_from(mask)]= True\n\nf, ax = plt.subplots(figsize=(30, 20))\n\nheatmap = sns.heatmap(corr_mat.round(2),\n                      mask = mask,\n                      square = True,\n                      linewidths = .5,\n                      cmap = 'coolwarm',\n                      vmin = -1,\n                      vmax = 1,\n                      annot = True,\n                      annot_kws = {'size': 12})\n\n#add the column names as labels\nax.set_yticklabels(corr_mat.columns, rotation = 0)\nax.set_xticklabels(corr_mat.columns)\n\nsns.set_style({'xtick.bottom': True}, {'ytick.left': True})","metadata":{"execution":{"iopub.status.busy":"2022-08-01T11:29:13.439789Z","iopub.execute_input":"2022-08-01T11:29:13.440201Z","iopub.status.idle":"2022-08-01T11:29:14.874133Z","shell.execute_reply.started":"2022-08-01T11:29:13.440166Z","shell.execute_reply":"2022-08-01T11:29:14.872757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"👉 **Observations**\n\n1. **measurement_17** is mildly correlated to measurement_5,6,7, and 8.\n2. Rest all variables have very less correlation amongst them.","metadata":{}},{"cell_type":"code","source":"plt.figure(figsize=(30, 30))\ncols = measure_cols+['loading']\n\nfor e, c in enumerate(cols, start=1):\n    plt.subplot(6,5,e)\n    p = sns.histplot(train, x=c, kde=True)\n    plt.xlabel(c)\n    plt.title(f\"{c}'s Distribution\")","metadata":{"execution":{"iopub.status.busy":"2022-08-01T12:24:10.466113Z","iopub.execute_input":"2022-08-01T12:24:10.466781Z","iopub.status.idle":"2022-08-01T12:24:19.867242Z","shell.execute_reply.started":"2022-08-01T12:24:10.466738Z","shell.execute_reply":"2022-08-01T12:24:19.865985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"👉 Observations\n\n**All continuous variables are are distributed normally, but measurement_0,1,2,17 and loading are INT variables are positively (rightly) skewed.**","metadata":{}},{"cell_type":"markdown","source":"<a id=\"section-three\"></a>\n# Ensembling with Hyperopt Search Algorithm","metadata":{}},{"cell_type":"markdown","source":"**Bayesian Optimization**\n\nCompared with GridSearch which is a brute-force approach, or RandomSearch which is purely random, the classical Bayesian Optimization combines randomness and posterior probability distribution in searching the optimal parameters by approximating the target function through Gaussian Process (i.e. random samples are drawn iteratively (Sequential Model-Based Optimization (SMBO)) and the function outputs between the samples are approximated by a confidence region). New samples will be drawn from the parameter space at the high mean and variance over the confidence region for exploration and exploitation.\n\n**Hyperopt**\n\nHyperopt is a python library for search spaces optimizing. Currently it offers two algorithms in optimization: 1. Random Search and 2. Tree of Parzen Estimators (TPE) which is a Bayesian approach which makes use of P(x|y) instead of P(y|x), based on approximating two different distributions separated by a threshold instead of one in calculating the Expected Improvement (see this). It used to accommodate Gaussian processes and regression tress but now these are no longer being implemented.","metadata":{}},{"cell_type":"markdown","source":"## Hyperopt Example\n\n**fmin()** is the main function in hyperopt for optimization. It accepts four basic arguments and output the optimized parameter set:\n\n**1. Objective Function** — fn\n\n**2. Search Space** — space\n\n**3. Search Algorithm** — algo\n\n**4. (Maximum) no. of evaluations** — max_evals\n\nWe may also pass a Trials object to the trials argument which keeps track of the whole process. In order to run with trails the output of the objective function has to be a dictionary including at least the keys 'loss' and 'status' which contain the result and the optimization status respectively. The interim values could be extracted by the following:\n\n1. trials.trials - a list of dictionaries contains all relevant information\n2. trials.results - a list of dictionaries collecting the function outputs\n3. trials.losses() - a list of losses (float for each 'ok' trial)\n4. trials.statuses() - a list of status strings\n5. trials.vals - a dictionary of sampled parameters","metadata":{}},{"cell_type":"markdown","source":"Few more things to demystify:\n\n**Search Algortihm**: either hyperopt.tpe.suggest or hyperopt.rand.suggest\n\n**Search Space**: hp.uniform('x', -1, 1) define a search space with label ‘x’ that will be sampled uniformly between -1 and 1. The stochastic expressions currently recognized by hyperopt’s optimization algorithms are:\n\n1. hp.choice(label, options): index of an option\n2. hp.randint(label, upper) : random integer within [0, upper)\n3. hp.uniform(label, low, high) : uniform value between low/high\n4. hp.quniform(label, low, high, q) :round(uniform(.)/q)*q (note that the value is giving a float instead of integer)\n5. hp.loguniform(label, low, high) : exp(uniform(low, high)/q)*q\n6. hp.qloguniform(label, low, high, q) :round(loguniform(.))\n7. hp.normal(label, mu, sigma) : sampling from normal distribution\n8. hp.qnormal(label, mu, sigma, q) :round(normal(nu, sigma)/q)*q\n9. hp.lognormal(label, mu, sigma) :exp(normal(mu, sigma)\n10. hp.qlognormal(label, mu, sigma, q) : round(exp(normal(.))/q)*q\n\nSee https://github.com/hyperopt/hyperopt/wiki/FMin for more detail.","metadata":{}},{"cell_type":"markdown","source":"## Feature Engineering","metadata":{}},{"cell_type":"code","source":"train_objects = [col for col in train.columns if train[col].dtype=='object']\ntest_objects = [col for col in test.columns if test[col].dtype=='object']\ntrain_objects == test_objects","metadata":{"execution":{"iopub.status.busy":"2022-08-09T06:39:24.540603Z","iopub.execute_input":"2022-08-09T06:39:24.541073Z","iopub.status.idle":"2022-08-09T06:39:24.553073Z","shell.execute_reply.started":"2022-08-09T06:39:24.541023Z","shell.execute_reply":"2022-08-09T06:39:24.551887Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_with_different_attribute = ['product_code', 'attribute_1']\n\ncols_df = pd.concat([train[cols_with_different_attribute], test[cols_with_different_attribute]], ignore_index=True)\n\nproduct_code_freq_encoding = dict((cols_df['product_code'].value_counts())/cols_df.shape[0])\nattribute_1_freq_encoding = dict((cols_df['attribute_1'].value_counts())/cols_df.shape[0])","metadata":{"execution":{"iopub.status.busy":"2022-08-09T06:39:25.806391Z","iopub.execute_input":"2022-08-09T06:39:25.807241Z","iopub.status.idle":"2022-08-09T06:39:25.826963Z","shell.execute_reply.started":"2022-08-09T06:39:25.807192Z","shell.execute_reply":"2022-08-09T06:39:25.825600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a label encoder object\nle = LabelEncoder()\n\n# Iterate through the columns\nfor col in train_objects:\n    if col == 'attribute_0':\n        # Train on the training data\n        le.fit(train[col])\n        # Transform both training and testing data\n        train[col] = le.transform(train[col])\n        test[col] = le.transform(test[col])\n        \n    elif col == 'attribute_1':\n        train[col] = train[col].map(attribute_1_freq_encoding)\n        test[col] = test[col].map(attribute_1_freq_encoding)\n    else:\n        train[col] = train[col].map(product_code_freq_encoding)\n        test[col] = test[col].map(product_code_freq_encoding)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T06:39:31.351264Z","iopub.execute_input":"2022-08-09T06:39:31.351658Z","iopub.status.idle":"2022-08-09T06:39:31.381727Z","shell.execute_reply.started":"2022-08-09T06:39:31.351626Z","shell.execute_reply":"2022-08-09T06:39:31.380801Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# ENSEMBLING","metadata":{}},{"cell_type":"code","source":"class EnsembleModel:\n    def __init__(self, params):\n        \"\"\"\n        LGB + XGB + CatBoost model\n        \"\"\"\n        self.lgb_params = params['lgb']\n        self.xgb_params = params['xgb']\n        self.cat_params = params['cat']\n\n        self.lgb_model = LGBMClassifier(**self.lgb_params)\n        self.xgb_model = XGBClassifier(**self.xgb_params)\n        self.cat_model = CatBoostClassifier(**self.cat_params)\n\n    def fit(self, x, y, *args, **kwargs):\n        return (self.lgb_model.fit(x, y, *args, **kwargs),\n                self.xgb_model.fit(x, y, *args, **kwargs),\n               self.cat_model.fit(x, y, *args, **kwargs))\n\n    def predict(self, x, weights=[1.0, 1.0, 1.0]):\n        \"\"\"\n        Generate model predictions\n        :param x: data\n        :param weights: weights on model prediction, first one is the weight on lgb model\n        :return: array with predictions\n        \"\"\"\n        return np.rint((weights[0] * self.lgb_model.predict(x) +\n                weights[1] * self.xgb_model.predict(x) +\n                weights[2] * self.cat_model.predict(x)) / 3)\n    \n    def predict_proba(self, x, weights=[1.0, 1.0, 1.0]):\n        \"\"\"\n        Generate model class label probability predictions\n        :param x: data\n        :param weights: weights on model prediction, first one is the weight on lgb model\n        :return: array with predictions\n        \"\"\"\n        return np.rint((weights[0] * self.lgb_model.predict_proba(x) +\n                weights[1] * self.xgb_model.predict_proba(x) +\n                weights[2] * self.cat_model.predict_proba(x)) / 3)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T06:39:56.128756Z","iopub.execute_input":"2022-08-09T06:39:56.129219Z","iopub.status.idle":"2022-08-09T06:39:56.141445Z","shell.execute_reply.started":"2022-08-09T06:39:56.129182Z","shell.execute_reply":"2022-08-09T06:39:56.140465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEED = 47\n\nestimators = [100, 500, 1000, 2000, 5000, 10000]\n\n#integer and string parameters, used with hp.choice()\nensemble_params = {\n    \"lgb\": {\n#         \"n_estimators\": hp.choice(\"n_estimators\", estimators),\n        'n_estimators': 2000,\n        \"num_leaves\": scope.int(hp.quniform(\"num_leaves\", 31, 200, 1)),\n        \"max_depth\": scope.int(hp.quniform(\"max_depth\", 10, 24, 1)),\n        'learning_rate': hp.uniform('learning_rate', 0.01, 0.3),\n        'min_split_gain': hp.uniform('min_split_gain', 0, 1.0),\n        'min_child_samples': scope.int(hp.quniform(\"min_child_samples\", 2, 700, 1)),\n        \"subsample\": hp.uniform(\"subsample\", 0.2, 1.0),\n        \"colsample_bytree\": hp.uniform(\"colsample_bytree\", 0.5, 1.0),\n        'reg_alpha': hp.uniform('reg_alpha', 1e-5, 1.0),\n        'reg_lambda': hp.uniform('reg_lambda', 0, 50),\n        'metric': 'auc',\n        'n_jobs': -1},\n    \n    'xgb': {\n#         \"n_estimators\": hp.choice(\"n_estimators\", estimators),\n        'n_estimators': 2000,\n        'max_depth': scope.int(hp.quniform('xgb.max_depth', 10, 24, 1)),\n        'learning_rate': hp.uniform('xgb.learning_rate', 0.01, 0.3),\n        'gamma': hp.uniform('xgb.gamma', 1, 10),\n        'min_child_weight': scope.int(hp.quniform('xgb.min_child_weight', 2, 700, 1)),\n        'colsample_bytree': hp.uniform('xgb.colsample_bytree', 0.5, 0.9),\n        'subsample': hp.uniform('xgb.subsample', 0.5, 1.0),\n        'reg_lambda': hp.uniform('xgb.reg_lambda', 0, 100),\n        'reg_alpha': hp.uniform('xgb.reg_alpha', 1e-5, 0.5),\n        'objective': 'binary:logistic',\n        'tree_method': 'hist',\n        'eval_metric': 'auc',\n        'n_jobs': -1},\n    \n    'cat': {\n#         \"n_estimators\": hp.choice(\"n_estimators\", estimators),\n        'n_estimators': 2000,\n        'depth': hp.quniform(\"cat.depth\", 2, 16, 1),\n        'learning_rate': hp.uniform('cat.learning_rate', 0.01, 0.3),\n        'l2_leaf_reg': hp.uniform('cat.l2_leaf_reg', 3, 8),\n        'max_bin' : hp.quniform('cat.max_bin', 1, 254, 1),\n        'min_data_in_leaf' : hp.quniform('cat.min_data_in_leaf', 2, 700, 1),\n        'random_strength' : hp.loguniform('cat.random_strength', np.log(0.005), np.log(5)),\n        'fold_len_multiplier' : hp.loguniform('cat.fold_len_multiplier', np.log(1.01), np.log(2.5)),\n        'eval_metric': 'AUC'}\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-09T06:56:17.671972Z","iopub.execute_input":"2022-08-09T06:56:17.672381Z","iopub.status.idle":"2022-08-09T06:56:17.688519Z","shell.execute_reply.started":"2022-08-09T06:56:17.672348Z","shell.execute_reply":"2022-08-09T06:56:17.687284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def ensemble_search(params):\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=SEED)\n\n    model = EnsembleModel(params)\n\n    evaluation = [(X_test, y_test)]\n    \n#     for param in ['n_estimators','max_depth','num_leaves']:\n#         params[param] = int(params[param])\n\n    model.fit(X_train, y_train,\n              eval_set=evaluation,\n              early_stopping_rounds=100, verbose=False)\n\n    val_preds = model.predict(X_test)\n    \n    fpr, tpr, thresholds = roc_curve(y_test, val_preds, pos_label = 1)\n    auc_score = auc(fpr, tpr)\n\n    neg_auc_score = -1 * auc_score\n\n    return {\"loss\": neg_auc_score, \"status\": STATUS_OK}","metadata":{"execution":{"iopub.status.busy":"2022-08-09T06:56:20.598182Z","iopub.execute_input":"2022-08-09T06:56:20.598969Z","iopub.status.idle":"2022-08-09T06:56:20.607208Z","shell.execute_reply.started":"2022-08-09T06:56:20.598922Z","shell.execute_reply":"2022-08-09T06:56:20.606099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# search for model\n\nX = train.drop(columns=[\"failure\", \"id\"])\ny = train[\"failure\"]\nX_test = test.drop(columns=[\"id\"])\n\ntrials = Trials()\n\nbest_hyperparams = fmin(fn=ensemble_search,\n                       space=ensemble_params,\n                       algo=tpe.suggest,\n                       max_evals=20,\n                       trials=trials)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:33:44.485408Z","iopub.execute_input":"2022-08-09T07:33:44.485836Z","iopub.status.idle":"2022-08-09T07:34:21.939574Z","shell.execute_reply.started":"2022-08-09T07:33:44.485800Z","shell.execute_reply":"2022-08-09T07:34:21.938344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"best_hyperparams","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:16:48.254158Z","iopub.execute_input":"2022-08-09T07:16:48.254634Z","iopub.status.idle":"2022-08-09T07:16:48.263952Z","shell.execute_reply.started":"2022-08-09T07:16:48.254596Z","shell.execute_reply":"2022-08-09T07:16:48.262864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Kfold Cross Validation","metadata":{}},{"cell_type":"code","source":"# ------------------------------------------------------------------------------\n# Parameters\n# ------------------------------------------------------------------------------\nN_FOLDS = 5\nN_ESTIMATORS = 2000\nBAGGING_SEED = 48\n\nsince = time.time()\ncolumns = train.columns\n\nensemble_params = {\n    \"lgb\" : {\n        'colsample_bytree': 0.7503702962871435,\n        'learning_rate': 0.26896203515278,\n        'max_depth': 22,\n        'min_child_samples': 6,\n        'min_split_gain': 0.7322107135299711,\n        'num_leaves': 96,\n        'reg_alpha': 0.7792383989266799,\n        'reg_lambda': 38.73634740770757,\n        'subsample': 0.7836363181648811,\n        'boosting_type': 'gbdt',\n        'objective': 'binary',\n        'metric': 'auc',\n        'feature_fraction_seed': SEED,\n        'bagging_seed': SEED,\n        'random_state': SEED,\n        'n_jobs': -1,\n        'n_estimators': 2000\n    },\n    'xgb': {\n        \n        'colsample_bytree': 0.7571166877350035,\n        'gamma': 9.79047372074,\n        'learning_rate': 0.09171284775555011,\n        'max_depth': 13,\n        'min_child_weight': 390,\n        'reg_alpha': 0.20355247877998708,\n        'reg_lambda': 35.96183423626109,\n        'subsample': 0.909926579389054,\n        'n_estimators': 2000,\n        'objective': 'binary:logistic',\n        'tree_method': 'hist',\n        'random_state': SEED,\n        'eval_metric': 'auc',\n        'n_jobs': -1\n    },\n    'cat': {\n        'depth': 10,\n        'fold_len_multiplier': 1.7404594174961332,\n        'l2_leaf_reg': 6.324968703135411,\n        'learning_rate': 0.07581783267402314,\n        'max_bin': 29,\n        'min_data_in_leaf': 263,\n        'random_strength': 0.12257435675081946,\n        'n_estimators': 2000,\n        'random_state': SEED,\n        'eval_metric': 'AUC',\n    }\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:28:11.604134Z","iopub.execute_input":"2022-08-09T07:28:11.604586Z","iopub.status.idle":"2022-08-09T07:28:11.615399Z","shell.execute_reply.started":"2022-08-09T07:28:11.604548Z","shell.execute_reply":"2022-08-09T07:28:11.614194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_best_iter = []\n\ndef hyperparameter_tuning_lgb(params):\n    \"\"\"Performs Hyperparameter Optimization on LGBMRegressor\"\"\"\n    \n    X = train.drop(columns=[\"failure\", \"id\"])\n    y = train[\"failure\"]\n\n    for param in ['n_estimators','max_depth','num_leaves']:\n        params[param] = int(params[param])\n        \n    model = LGBMClassifier(**ensemble_params['lgb'], importance_type='gain',random_state=SEED) #params\n    \n    group_kfold = GroupKFold(n_splits=5)\n    roc_auc_score_list = []\n    \n    for fold, (train_idx, valid_idx) in enumerate(group_kfold.split(X, y, train['product_code'])):\n        print(f\"======fold {fold}=======\")\n        X_train, X_valid = X.loc[train_idx], X.loc[valid_idx]\n        y_train, y_valid = y.loc[train_idx], y.loc[valid_idx]\n        \n        model.fit(X_train, y_train, \n                  eval_set=[(X_train, y_train), (X_valid,y_valid)], \n                  eval_names = ['train', 'valid'],\n                  eval_metric = 'auc',\n                  early_stopping_rounds = 100, \n                  verbose = 200)\n        \n        y_pred = model.predict_proba(X_valid)[:, 1]\n        score = roc_auc_score(y_valid, y_pred)\n        neg_roc_auc_score = -1 * score\n        roc_auc_score_list.append(score)\n        \n        model_best_iter.append(model.best_iteration_)\n        print(f\"ROC AUC score for fold {fold} is : {score}\")\n        \n    return {\"loss\": np.mean(neg_roc_auc_score), \"status\": STATUS_OK}","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:53:44.011948Z","iopub.execute_input":"2022-08-09T07:53:44.012397Z","iopub.status.idle":"2022-08-09T07:53:44.025082Z","shell.execute_reply.started":"2022-08-09T07:53:44.012362Z","shell.execute_reply":"2022-08-09T07:53:44.023958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SEED = 47\n\n# estimators = [100, 500, 1000, 2000, 5000, 10000]\n# lgb_space = {\n#         \"n_estimators\": hp.choice(\"n_estimators\", estimators),\n#         \"num_leaves\": scope.int(hp.quniform(\"num_leaves\", 31, 200, 1)),\n#         \"max_depth\": scope.int(hp.quniform(\"max_depth\", 10, 24, 1)),\n#         'learning_rate': hp.uniform('learning_rate', 0.01, 0.3),\n#         'min_split_gain': hp.uniform('min_split_gain', 0, 1.0),\n#         'min_child_samples': scope.int(hp.quniform(\"min_child_samples\", 2, 700, 1)),\n#         \"subsample\": hp.uniform(\"subsample\", 0.2, 1.0),\n#         \"colsample_bytree\": hp.uniform(\"colsample_bytree\", 0.5, 1.0),\n#         'reg_alpha': hp.uniform('reg_alpha', 1e-5, 1.0),\n#         'reg_lambda': hp.uniform('reg_lambda', 0, 50),\n#         'metric': 'auc',\n#         'n_jobs': -1\n# }\n\n# trials = Trials()\n\n# best_hyperparams = fmin(fn=hyperparameter_tuning_lgb,\n#                        space=lgb_space,\n#                        algo=tpe.suggest,\n#                        max_evals=20,\n#                        trials=trials)\n\n# best_hyperparams\n\n#  #copying the best params\n# import copy\n# params = copy.deepcopy(best_hyperparams)\n\n# params['n_estimators'] = estimators[best_hyperparams['n_estimators']] \n# params['max_depth'] = int(best_hyperparams['max_depth'])\n# params['num_leaves'] = int(best_hyperparams['num_leaves'])\n# params['min_child_samples'] = int(best_hyperparams['min_child_samples'])\n\n# # initialize the model and fit entire X and y data\n# print(\"Fitting the data...\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T11:16:40.181694Z","iopub.execute_input":"2022-08-03T11:16:40.182087Z","iopub.status.idle":"2022-08-03T11:17:52.673474Z","shell.execute_reply.started":"2022-08-03T11:16:40.182055Z","shell.execute_reply":"2022-08-03T11:17:52.672312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LGBM","metadata":{}},{"cell_type":"code","source":"lgbm_model = LGBMClassifier(**ensemble_params['lgb'])\nlgbm_model.fit(X,y)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:56:00.372977Z","iopub.execute_input":"2022-08-09T07:56:00.373406Z","iopub.status.idle":"2022-08-09T07:56:02.829887Z","shell.execute_reply.started":"2022-08-09T07:56:00.373373Z","shell.execute_reply":"2022-08-09T07:56:02.829021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test_lgb = lgbm_model.predict_proba(X_test)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-08-09T07:56:21.045277Z","iopub.execute_input":"2022-08-09T07:56:21.045709Z","iopub.status.idle":"2022-08-09T07:56:21.090069Z","shell.execute_reply.started":"2022-08-09T07:56:21.045675Z","shell.execute_reply":"2022-08-09T07:56:21.089016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## XGB","metadata":{}},{"cell_type":"code","source":"xgb_model = XGBClassifier(**ensemble_params['xgb'])\nxgb_model.fit(X,y)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T08:13:53.451638Z","iopub.execute_input":"2022-08-09T08:13:53.452148Z","iopub.status.idle":"2022-08-09T08:14:02.937757Z","shell.execute_reply.started":"2022-08-09T08:13:53.452106Z","shell.execute_reply":"2022-08-09T08:14:02.936519Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test_xgb = xgb_model.predict_proba(X_test)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-08-09T08:14:14.379220Z","iopub.execute_input":"2022-08-09T08:14:14.379667Z","iopub.status.idle":"2022-08-09T08:14:14.445988Z","shell.execute_reply.started":"2022-08-09T08:14:14.379634Z","shell.execute_reply":"2022-08-09T08:14:14.444857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CatBoost","metadata":{}},{"cell_type":"code","source":"cat_model = CatBoostClassifier(**ensemble_params['cat'])\ncat_model.fit(X,y)\n\ny_test_cat = cat_model.predict_proba(X_test)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-08-09T08:15:09.632458Z","iopub.execute_input":"2022-08-09T08:15:09.633549Z","iopub.status.idle":"2022-08-09T08:16:02.282021Z","shell.execute_reply.started":"2022-08-09T08:15:09.633503Z","shell.execute_reply":"2022-08-09T08:16:02.280840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a id=\"section-four\"></a>\n# Submission (LGBM+XGB+CatBoost)/3","metadata":{}},{"cell_type":"code","source":"y_test = np.vstack((y_test_lgb,y_test_xgb,y_test_cat))\nsubmission['failure'] = y_test.mean(axis=0)\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-09T08:33:56.804116Z","iopub.execute_input":"2022-08-09T08:33:56.804527Z","iopub.status.idle":"2022-08-09T08:33:56.866957Z","shell.execute_reply.started":"2022-08-09T08:33:56.804495Z","shell.execute_reply":"2022-08-09T08:33:56.865861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Please upvote if you like my work","metadata":{}}]}