{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"***This notebook is the introduction to Light Gradient Boosting and Optuna. I removed EDA, Transformations, and optimal paramaters. You can use this one as a quick cheatsheet and play around it by changing parameters to get better result. I will post more about LGBM, XGBoost and stacking methods.***\n\n**I got feather datas from https://www.kaggle.com/davidhammond.**\n\n**I will upload extended datasets as pickles which have 918 features**\n\n\n","metadata":{}},{"cell_type":"markdown","source":"*If you have any questions, do not hesitate to ask! Thanks!*","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\">\n$$\\Huge{\\color{blue}{\\textbf{💳 American Express - Default Prediction 💳}}}$$\n</div>","metadata":{"ExecuteTime":{"end_time":"2022-07-08T03:12:17.125702Z","start_time":"2022-07-08T03:12:17.118595Z"}}},{"cell_type":"markdown","source":"![](https://i.insider.com/5dc432453afd37558104a932?width=1200&format=jpeg)","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\">\n$\\Large{\\color{blue}{\\textbf{📌 Importing Modules and Functions:}}}$\n</div>","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport os\nimport dask.dataframe as dd\nimport pandas as pd\nfrom sklearn.tree import DecisionTreeClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score, f1_score, confusion_matrix, classification_report, mean_squared_error, precision_score, recall_score, precision_recall_curve, make_scorer, roc_curve, roc_auc_score\nfrom sklearn.model_selection import train_test_split, cross_validate, GridSearchCV, cross_val_score, StratifiedKFold, KFold\nimport lightgbm as lgb\nfrom sklearn import preprocessing\nimport gc\nimport gzip\npd.set_option(\"display.max_columns\", None)","metadata":{"ExecuteTime":{"end_time":"2022-07-14T20:14:15.309932Z","start_time":"2022-07-14T20:14:11.796768Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:34.806591Z","iopub.execute_input":"2022-07-15T05:33:34.807101Z","iopub.status.idle":"2022-07-15T05:33:34.818338Z","shell.execute_reply.started":"2022-07-15T05:33:34.807061Z","shell.execute_reply":"2022-07-15T05:33:34.816747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def pie_chart(df,col,title):\n    colors = [\"#0048ba\", '#485aa4']\n    fig, ax = plt.subplots(1,2,figsize=(16, 8))\n    fig.suptitle(title, size = 20)\n    labels = list(df[col].value_counts().index)\n    values = df[col].value_counts()\n    ax[0].pie(values,colors=colors,explode=(.05,0),startangle=60,labels=labels, autopct='%1.0f%%', pctdistance=0.6)\n    sns.countplot(x=col, data=df, hue=col,palette=colors, ax=ax[1])\n    ax[0].add_artist(plt.Circle((0,0),0.4,fc='white'))\n    plt.show()          ","metadata":{"ExecuteTime":{"end_time":"2022-07-12T21:57:53.739304Z","start_time":"2022-07-12T21:57:53.736080Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:34.824527Z","iopub.execute_input":"2022-07-15T05:33:34.824942Z","iopub.status.idle":"2022-07-15T05:33:34.835844Z","shell.execute_reply.started":"2022-07-15T05:33:34.824907Z","shell.execute_reply":"2022-07-15T05:33:34.834232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex(preds, train_data) -> float:\n    y_true=train_data.get_label()\n    y_pred = 1. / (1. + np.exp(-preds))\n\n    def amex_metric_mod(y_true, y_pred):\n\n        labels     = np.transpose(np.array([y_true, y_pred]))\n        labels     = labels[labels[:, 1].argsort()[::-1]]\n        weights    = np.where(labels[:,0]==0, 20, 1)\n        cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n        top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n        gini = [0,0]\n        for i in [1,0]:\n            labels         = np.transpose(np.array([y_true, y_pred]))\n            labels         = labels[labels[:, i].argsort()[::-1]]\n            weight         = np.where(labels[:,0]==0, 20, 1)\n            weight_random  = np.cumsum(weight / np.sum(weight))\n            total_pos      = np.sum(labels[:, 0] *  weight)\n            cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n            lorentz        = cum_pos_found / total_pos\n            gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n        return 0.5 * (gini[1]/gini[0] + top_four)\n    return 'amex', amex_metric_mod(y_true,y_pred), True","metadata":{"ExecuteTime":{"end_time":"2022-07-12T21:57:53.744621Z","start_time":"2022-07-12T21:57:53.740010Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:34.843515Z","iopub.execute_input":"2022-07-15T05:33:34.845131Z","iopub.status.idle":"2022-07-15T05:33:34.862498Z","shell.execute_reply.started":"2022-07-15T05:33:34.845067Z","shell.execute_reply":"2022-07-15T05:33:34.861273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\">\n$\\Large{\\color{blue}{\\textbf{📌 Exploratory Data Analysis:}}}$\n</div>","metadata":{}},{"cell_type":"code","source":"'''\nfor dirname, _, filenames in os.walk('/Users/jafar_bakhshaliyev/Desktop/Kaggle'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n'''        ","metadata":{"ExecuteTime":{"end_time":"2022-07-07T23:43:00.969913Z","start_time":"2022-07-07T23:43:00.912159Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:34.865019Z","iopub.execute_input":"2022-07-15T05:33:34.865674Z","iopub.status.idle":"2022-07-15T05:33:34.878141Z","shell.execute_reply.started":"2022-07-15T05:33:34.865621Z","shell.execute_reply":"2022-07-15T05:33:34.876816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\ntrain_data = pd.read_feather(\"/Users/jafar_bakhshaliyev/Desktop/Kaggle/train_data.ftr\")\n'''","metadata":{"ExecuteTime":{"end_time":"2022-07-12T21:04:37.058632Z","start_time":"2022-07-12T21:04:33.199714Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:34.880208Z","iopub.execute_input":"2022-07-15T05:33:34.880816Z","iopub.status.idle":"2022-07-15T05:33:34.892249Z","shell.execute_reply.started":"2022-07-15T05:33:34.880778Z","shell.execute_reply":"2022-07-15T05:33:34.890688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nfeatures = train_data.drop(['customer_ID', 'S_2'], axis=1).columns.to_list()\ncat_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nnum_features = [col for col in features if col not in cat_features]\n'''","metadata":{"ExecuteTime":{"end_time":"2022-07-12T21:04:44.408307Z","start_time":"2022-07-12T21:04:38.635796Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:34.895176Z","iopub.execute_input":"2022-07-15T05:33:34.896432Z","iopub.status.idle":"2022-07-15T05:33:34.906166Z","shell.execute_reply.started":"2022-07-15T05:33:34.896392Z","shell.execute_reply":"2022-07-15T05:33:34.905293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\ntrain_num_agg = train_data.groupby('customer_ID')[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\ntrain_num_agg.columns = ['_'.join(x) for x in train_num_agg.columns]\ntrain_cat_agg = train_data.groupby('customer_ID')[cat_features].agg(['count', 'last', 'nunique'])\ntrain_cat_agg.columns = ['_'.join(x) for x in train_cat_agg.columns]\ntrain_data = pd.concat([train_num_agg, train_cat_agg], axis=1)\ntrain_data.to_pickle(\"./train_extended.pkl\", compression=\"gzip\")\n'''","metadata":{"ExecuteTime":{"end_time":"2022-07-12T20:52:11.533603Z","start_time":"2022-07-12T20:49:08.456932Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:34.907936Z","iopub.execute_input":"2022-07-15T05:33:34.908426Z","iopub.status.idle":"2022-07-15T05:33:34.927013Z","shell.execute_reply.started":"2022-07-15T05:33:34.908393Z","shell.execute_reply":"2022-07-15T05:33:34.925384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\ntest_data = pd.read_pickle(\"./test_extended.pkl\", compression=\"gzip\")\n'''","metadata":{"ExecuteTime":{"end_time":"2022-07-12T21:49:04.329068Z","start_time":"2022-07-12T21:48:45.938957Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:34.929156Z","iopub.execute_input":"2022-07-15T05:33:34.929600Z","iopub.status.idle":"2022-07-15T05:33:34.939920Z","shell.execute_reply.started":"2022-07-15T05:33:34.929554Z","shell.execute_reply":"2022-07-15T05:33:34.938738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\ntest_data = pd.read_feather(\"/Users/jafar_bakhshaliyev/Desktop/Kaggle/test_data.ftr\")\n'''","metadata":{"ExecuteTime":{"end_time":"2022-07-12T21:05:04.833033Z","start_time":"2022-07-12T21:04:48.921797Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:34.941726Z","iopub.execute_input":"2022-07-15T05:33:34.942056Z","iopub.status.idle":"2022-07-15T05:33:34.952868Z","shell.execute_reply.started":"2022-07-15T05:33:34.942028Z","shell.execute_reply":"2022-07-15T05:33:34.951631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\ntest_num_agg = test_data.groupby('customer_ID')[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\ntest_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\ntest_cat_agg = test_data.groupby('customer_ID')[cat_features].agg(['count', 'last', 'nunique'])\ntest_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\ntest = pd.concat([test_num_agg, test_cat_agg], axis=1)\ntest.to_pickle(\"../data/test_agg.pkl\", compression=\"gzip\")\n'''","metadata":{"ExecuteTime":{"start_time":"2022-07-12T21:05:19.616Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:34.956172Z","iopub.execute_input":"2022-07-15T05:33:34.957141Z","iopub.status.idle":"2022-07-15T05:33:34.967767Z","shell.execute_reply.started":"2022-07-15T05:33:34.957033Z","shell.execute_reply":"2022-07-15T05:33:34.966167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nwith gzip.open(\"./train_extended.pkl\", 'rb') as train:\n    train_data = pickle.load(train)\n'''","metadata":{"execution":{"iopub.status.busy":"2022-07-15T05:33:34.969986Z","iopub.execute_input":"2022-07-15T05:33:34.970572Z","iopub.status.idle":"2022-07-15T05:33:34.979923Z","shell.execute_reply.started":"2022-07-15T05:33:34.970533Z","shell.execute_reply":"2022-07-15T05:33:34.978813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nwith gzip.open(\"./test_extended.pkl\", 'rb') as test:\n    test_data = pickle.load(test)\n'''","metadata":{"execution":{"iopub.status.busy":"2022-07-15T05:33:34.982747Z","iopub.execute_input":"2022-07-15T05:33:34.983100Z","iopub.status.idle":"2022-07-15T05:33:34.993315Z","shell.execute_reply.started":"2022-07-15T05:33:34.983069Z","shell.execute_reply":"2022-07-15T05:33:34.991861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train and Test Data have 918 features!","metadata":{}},{"cell_type":"code","source":"'''\ndelinquency_features = [col for col in train_data if col.startswith('D_')]\nspend_features = [col for col in train_data if col.startswith('S_')]\npayment_features = [col for col in train_data if col.startswith('P_')]\nbalance_features = [col for col in train_data if col.startswith('B_')]\nrisk_features = [col for col in train_data if col.startswith('R_')]\n'''","metadata":{"ExecuteTime":{"end_time":"2022-07-12T21:49:27.190187Z","start_time":"2022-07-12T21:49:27.172778Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:34.995151Z","iopub.execute_input":"2022-07-15T05:33:34.995980Z","iopub.status.idle":"2022-07-15T05:33:35.008307Z","shell.execute_reply.started":"2022-07-15T05:33:34.995921Z","shell.execute_reply":"2022-07-15T05:33:35.006992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nprint(\"# of delinquency features:\",len(delinquency_features))\nprint(\"# of spend features:\",len(spend_features))\nprint(\"# of payment features:\",len(payment_features))\nprint(\"# of balance features:\",len(balance_features))\nprint(\"# of risk features:\",len(risk_features))\n'''","metadata":{"ExecuteTime":{"end_time":"2022-07-12T21:49:28.099683Z","start_time":"2022-07-12T21:49:28.089628Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:35.010744Z","iopub.execute_input":"2022-07-15T05:33:35.011422Z","iopub.status.idle":"2022-07-15T05:33:35.020979Z","shell.execute_reply.started":"2022-07-15T05:33:35.011384Z","shell.execute_reply":"2022-07-15T05:33:35.019820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\ntrain_labels = pd.read_csv('/Users/jafar_bakhshaliyev/Desktop/Kaggle/train_labels.csv').set_index('customer_ID', drop=True)\ntrain_labels\n'''","metadata":{"ExecuteTime":{"end_time":"2022-07-12T21:49:37.746961Z","start_time":"2022-07-12T21:49:37.270251Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:35.025929Z","iopub.execute_input":"2022-07-15T05:33:35.026456Z","iopub.status.idle":"2022-07-15T05:33:35.037683Z","shell.execute_reply.started":"2022-07-15T05:33:35.026422Z","shell.execute_reply":"2022-07-15T05:33:35.036749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\ntrain_data = pd.merge(train_data, train_labels, left_index=True, right_index=True)\ntrain_data\n'''","metadata":{"ExecuteTime":{"end_time":"2022-07-12T21:49:44.606447Z","start_time":"2022-07-12T21:49:40.908992Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:35.048949Z","iopub.execute_input":"2022-07-15T05:33:35.050210Z","iopub.status.idle":"2022-07-15T05:33:35.057514Z","shell.execute_reply.started":"2022-07-15T05:33:35.050165Z","shell.execute_reply":"2022-07-15T05:33:35.056364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\npie_chart(train_data,'target','Target') \n'''","metadata":{"ExecuteTime":{"end_time":"2022-07-12T21:50:11.888099Z","start_time":"2022-07-12T21:50:11.440671Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:35.063376Z","iopub.execute_input":"2022-07-15T05:33:35.063753Z","iopub.status.idle":"2022-07-15T05:33:35.071755Z","shell.execute_reply.started":"2022-07-15T05:33:35.063723Z","shell.execute_reply":"2022-07-15T05:33:35.070482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\npd.set_option('display.float_format', lambda x: '%.5f' % x)\ntrain_data.describe()\n'''","metadata":{"ExecuteTime":{"end_time":"2022-07-12T16:55:47.388573Z","start_time":"2022-07-12T16:55:40.191871Z"},"scrolled":true,"execution":{"iopub.status.busy":"2022-07-15T05:33:35.078619Z","iopub.execute_input":"2022-07-15T05:33:35.079548Z","iopub.status.idle":"2022-07-15T05:33:35.087002Z","shell.execute_reply.started":"2022-07-15T05:33:35.079513Z","shell.execute_reply":"2022-07-15T05:33:35.085684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nx = train_data.drop(['target'],axis = 1)\ny = train_data['target']\nx_test = test_data\n'''","metadata":{"ExecuteTime":{"end_time":"2022-07-12T21:53:39.381538Z","start_time":"2022-07-12T21:53:37.345629Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:35.092509Z","iopub.execute_input":"2022-07-15T05:33:35.093149Z","iopub.status.idle":"2022-07-15T05:33:35.101908Z","shell.execute_reply.started":"2022-07-15T05:33:35.093098Z","shell.execute_reply":"2022-07-15T05:33:35.100515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\">\n$\\Large{\\color{blue}{\\textbf{📌 Hyperarameter Tuning by Optuna:}}}$\n</div>","metadata":{}},{"cell_type":"code","source":"import optuna","metadata":{"ExecuteTime":{"end_time":"2022-07-15T02:36:06.729600Z","start_time":"2022-07-15T02:36:06.724288Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:35.104172Z","iopub.execute_input":"2022-07-15T05:33:35.104822Z","iopub.status.idle":"2022-07-15T05:33:35.113362Z","shell.execute_reply.started":"2022-07-15T05:33:35.104776Z","shell.execute_reply":"2022-07-15T05:33:35.112313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nlgb_train = lgb.Dataset(X_train, label=y_train,free_raw_data=False)\nlgb_test = lgb.Dataset(X_val,label=y_val,free_raw_data=False)\n\ndef objective(trial):\n                        \n    param = {\n        \"objective\": \"binary\",\n        \"metric\": \"binary_logloss\",\n        \"verbosity\": -1,\n        \"boosting_type\": \"gbdt\", \n        \"seed\": 42,\n        'lambda_l1': trial.suggest_loguniform('lambda_l1', 1e-8, 50),\n        'lambda_l2': trial.suggest_loguniform('lambda_l2', 1e-8, 50),\n        'max_depth': trial.suggest_int('max_depth', 1, 30),\n        'num_leaves': trial.suggest_int('num_leaves', 2, 400),\n        'feature_fraction': trial.suggest_uniform('feature_fraction', 0.1, 1.0),\n        'bagging_fraction': trial.suggest_uniform('bagging_fraction', 0.1, 1.0),\n        'min_split_gain': trial.suggest_loguniform('min_split_gain', 1e-8, 10.0),\n        'bagging_freq': trial.suggest_int('bagging_freq', 0, 70),\n        'min_child_samples': trial.suggest_int('min_child_samples', 1, 250),\n        'min_data_in_leaf': trial.suggest_int('min_data_in_leaf', 1, 400),\n        'learning_rate': trial.suggest_uniform('learning_rate', 0.001, 0.1),\n        'feature_pre_filter': False,\n        'seed': 1979\n    }\n   \n    lgbcv = lgb.train(param,\n                      lgb_train,\n                      valid_sets=[lgb_train, lgb_test],\n                      feval = amex,\n                      verbose_eval=100,                   \n                      early_stopping_rounds = 50,                   \n                      num_boost_round=1500                    \n                  )\n    \n      \n   \n    cv_score = dict(lgbcv.best_score)\n    cv_score = cv_score['valid_1']['amex']\n    \n    return cv_score\n\n\noptuna.logging.set_verbosity(optuna.logging.WARNING) \nstudy = optuna.create_study(direction='maximize')  \nstudy.optimize(objective, n_trials = 150) \n'''","metadata":{"ExecuteTime":{"end_time":"2022-07-12T12:42:56.415859Z","start_time":"2022-07-12T04:20:12.713885Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:35.123498Z","iopub.execute_input":"2022-07-15T05:33:35.124935Z","iopub.status.idle":"2022-07-15T05:33:35.134816Z","shell.execute_reply.started":"2022-07-15T05:33:35.124875Z","shell.execute_reply":"2022-07-15T05:33:35.133741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import joblib","metadata":{"execution":{"iopub.status.busy":"2022-07-15T05:33:35.136727Z","iopub.execute_input":"2022-07-15T05:33:35.137334Z","iopub.status.idle":"2022-07-15T05:33:35.151121Z","shell.execute_reply.started":"2022-07-15T05:33:35.137294Z","shell.execute_reply":"2022-07-15T05:33:35.150157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study = joblib.load('../input/optuna-study-pickle/study.pkl')","metadata":{"ExecuteTime":{"end_time":"2022-07-15T02:35:28.004761Z","start_time":"2022-07-15T02:35:27.338989Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:35.153783Z","iopub.execute_input":"2022-07-15T05:33:35.154875Z","iopub.status.idle":"2022-07-15T05:33:35.237687Z","shell.execute_reply.started":"2022-07-15T05:33:35.154811Z","shell.execute_reply":"2022-07-15T05:33:35.236221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of finished trials: {}\".format(len(study.trials)))\n\nprint(\"Best trial:\")\ntrial = study.best_trial\n\nprint(\"  Value: {}\".format(trial.value))\n\nprint(\"  Params: \")\nfor key, value in trial.params.items():\n    print(\"    {}: {}\".format(key, value))","metadata":{"ExecuteTime":{"end_time":"2022-07-15T02:36:11.370749Z","start_time":"2022-07-15T02:36:11.330030Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:35.241610Z","iopub.execute_input":"2022-07-15T05:33:35.242390Z","iopub.status.idle":"2022-07-15T05:33:35.298376Z","shell.execute_reply.started":"2022-07-15T05:33:35.242341Z","shell.execute_reply":"2022-07-15T05:33:35.297071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_optimization_history(study)","metadata":{"ExecuteTime":{"end_time":"2022-07-15T02:36:14.134447Z","start_time":"2022-07-15T02:36:13.922022Z"},"scrolled":true,"execution":{"iopub.status.busy":"2022-07-15T05:33:35.300620Z","iopub.execute_input":"2022-07-15T05:33:35.300972Z","iopub.status.idle":"2022-07-15T05:33:35.366905Z","shell.execute_reply.started":"2022-07-15T05:33:35.300941Z","shell.execute_reply":"2022-07-15T05:33:35.365707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_slice(study)","metadata":{"ExecuteTime":{"end_time":"2022-07-15T02:36:22.848857Z","start_time":"2022-07-15T02:36:22.468253Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:35.368756Z","iopub.execute_input":"2022-07-15T05:33:35.369508Z","iopub.status.idle":"2022-07-15T05:33:35.720066Z","shell.execute_reply.started":"2022-07-15T05:33:35.369452Z","shell.execute_reply":"2022-07-15T05:33:35.718973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optuna.visualization.plot_param_importances(study)","metadata":{"ExecuteTime":{"end_time":"2022-07-15T02:36:28.793692Z","start_time":"2022-07-15T02:36:25.742848Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:35.721867Z","iopub.execute_input":"2022-07-15T05:33:35.722566Z","iopub.status.idle":"2022-07-15T05:33:42.927582Z","shell.execute_reply.started":"2022-07-15T05:33:35.722519Z","shell.execute_reply":"2022-07-15T05:33:42.926218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\">\n$\\Large{\\color{blue}{\\textbf{📌 Model Implementation:}}}$\n</div>","metadata":{}},{"cell_type":"code","source":"def lgbm_kfold(x, x_test, num_folds):\n\n    folds = StratifiedKFold(n_splits = num_folds, shuffle=True, random_state=326)\n    valid_pred = np.zeros(x.shape[0])\n    test_pred = np.zeros(x_test.shape[0])\n    \n    for n_fold, (train_idx, valid_idx) in enumerate(folds.split(x, y)):\n        train_x, train_y = train_df.iloc[train_idx], y.iloc[train_idx]\n        valid_x, valid_y = train_df.iloc[valid_idx], y.iloc[valid_idx]\n\n        lgb_train = lgb.Dataset(train_x,label=train_y,free_raw_data=False)\n        lgb_test = lgb.Dataset(valid_x,label=valid_y,free_raw_data=False)\n    \n\n        params = {\n                        'task': 'train',\n                        'boosting': 'dart',\n                        'objective': 'binary',\n                        'metric': ['binary_logloss','auc'],\n                        'learning_rate': 0.07435597175485592,\n                        'max_bin': 130,\n                        'max_depth': 26,\n                        'num_leaves': 47,\n                        'min_child_samples': 96,\n                        'lambda_l2': 1.106581197456048e-06,\n                        'lambda_l1': 10.593730984653938,\n                        'feature_fraction': 0.519246662147126,\n                        'bagging_fraction': 0.9418098609025605,\n                        'min_split_gain': 2.661672940371817e-07,\n                        'bagging_freq': 21,\n                        'min_data_in_leaf': 82,\n                        'verbose': -1,\n                        'seed':int(2**n_fold),\n                        'bagging_seed':int(2**n_fold),\n                        'drop_seed':int(2**n_fold)\n                        }\n        \n        model_lgbm = lgb.train(\n                        params,\n                        lgb_train,\n                        feval = amex,\n                        valid_sets=[lgb_train, lgb_test],\n                        valid_names=['train', 'test'],\n                        num_boost_round=3500,early_stopping_rounds = 100,\n                        verbose_eval=100,\n                        )\n\n        valid_pred[valid_idx] = model_lgbm.predict(valid_x, num_iteration=model_lgbm.best_iteration)\n        test_pred += model_lgbm.predict(x_test, num_iteration=model_lgbm.best_iteration) / folds.n_splits\n        \n        print('Fold %2d roc_auc_score : %.6f' % (n_fold + 1, roc_auc_score(valid_y, valid_pred[valid_idx])))\n        del reg, train_x, train_y, valid_x, valid_y\n        gc.collect()\n        \n    return test_pred\n        ","metadata":{"ExecuteTime":{"end_time":"2022-07-12T13:28:27.986439Z","start_time":"2022-07-12T13:28:27.959243Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:42.930106Z","iopub.execute_input":"2022-07-15T05:33:42.930564Z","iopub.status.idle":"2022-07-15T05:33:42.947695Z","shell.execute_reply.started":"2022-07-15T05:33:42.930528Z","shell.execute_reply":"2022-07-15T05:33:42.946372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-block alert-info\">\n$\\Large{\\color{blue}{\\textbf{📌 Submission:}}}$\n</div>","metadata":{}},{"cell_type":"code","source":"'''\nsubmission = lgbm_kfold(x, x_test, num_folds = 5) \n'''","metadata":{"ExecuteTime":{"start_time":"2022-07-12T13:28:38.811Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:42.949845Z","iopub.execute_input":"2022-07-15T05:33:42.950202Z","iopub.status.idle":"2022-07-15T05:33:42.970294Z","shell.execute_reply.started":"2022-07-15T05:33:42.950173Z","shell.execute_reply":"2022-07-15T05:33:42.968297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nsubm = pd.read_csv('/Users/jafar_bakhshaliyev/Desktop/Kaggle/sample_submission.csv')\nsubm[\"prediction\"] = pred.values\nsubm.to_csv(\"submission.csv\", index=False)\n'''","metadata":{"ExecuteTime":{"end_time":"2022-07-12T01:37:30.985343Z","start_time":"2022-07-12T01:37:30.153361Z"},"execution":{"iopub.status.busy":"2022-07-15T05:33:42.974193Z","iopub.execute_input":"2022-07-15T05:33:42.975183Z","iopub.status.idle":"2022-07-15T05:33:42.986045Z","shell.execute_reply.started":"2022-07-15T05:33:42.975130Z","shell.execute_reply":"2022-07-15T05:33:42.984849Z"},"trusted":true},"execution_count":null,"outputs":[]}]}