{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This competition has a metric where 50% depends on the percentage of positive cases that are in the top 4%.\n\nThe idea of the proposed loss function is **to penalize both the Gradient and the Hessian of the positive cases that are further away from the top positions**. The intention is that the training of the LGB model will put more effort to improve the prediction of these instances.\n","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\n\ndef weighted_logloss(preds, dtrain):\n    global MULT_NO4PERC, MAX_WEIGHTS\n    eps = 1e-16\n    labels = dtrain.get_label()\n    preds = 1.0 / (1.0 + np.exp(-preds))\n    \n    # top 4%\n    labels_mat = np.transpose(np.array([np.arange(len(labels)), labels, preds]))\n    pos_ord = labels_mat[:, 2].argsort()[::-1]\n    labels_mat = labels_mat[pos_ord]\n    weights_4perc    = np.where(labels_mat[:,1]==0, 20, 1)\n    top4   = np.cumsum(weights_4perc) <= int(0.04 * np.sum(weights_4perc))\n    top4   = top4[labels_mat[:, 0].argsort()]\n\n    weights = 1+np.exp(-MULT_NO4PERC*np.linspace(MAX_WEIGHTS-1,0,len(top4)))[labels_mat[:, 0].argsort()]\n    weights[top4 & (labels==1.0)] = 1.0 # Set to one weights of positive labels in top 4%\n    weights[(labels==0.0)] = 1.0 # Set to one weights of negative labels\n\n    grad = (preds - labels) * weights\n    hess = np.maximum(preds * (1.0 - preds) * weights , eps)\n    return grad, hess","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:26:09.626915Z","iopub.execute_input":"2022-08-13T16:26:09.628811Z","iopub.status.idle":"2022-08-13T16:26:09.641925Z","shell.execute_reply.started":"2022-08-13T16:26:09.628742Z","shell.execute_reply":"2022-08-13T16:26:09.640624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In this case, an increasing exponential curve like the one in the figure is used to calculate the weights. Thus, the weights increase as the distance of the positive instances to the 4% zone increases. The challenge is **to determine the best values for this curve (MULT_NO4PERC and MAX_WEIGHTS)**.","metadata":{}},{"cell_type":"code","source":"MULT_NO4PERC = 5\nMAX_WEIGHTS = 2.0\nlen_preds = 458913//5  # OOF preds\ntop4_position = len_preds*0.04\nplt.plot(1+np.exp(-MULT_NO4PERC*np.linspace(MAX_WEIGHTS-1.0,0,len_preds)))\nplt.vlines(x=top4_position, ymin=1, ymax=MAX_WEIGHTS, color='r')\nplt.text(0, 1.5, 'top 4%', color='r', rotation=90)\nplt.title('Weights')\nplt.savefig('/kaggle/working/weigths_curve.png')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T16:28:03.465186Z","iopub.execute_input":"2022-08-13T16:28:03.465533Z","iopub.status.idle":"2022-08-13T16:28:03.654684Z","shell.execute_reply.started":"2022-08-13T16:28:03.465508Z","shell.execute_reply":"2022-08-13T16:28:03.653177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The weights of the negative cases are set to 1.0, and the **positive cases that are within the 4% zone are also set to 1.0.**\n\nThis loss function can be easily adapted to other algorithms such as: XGB, CATBoost, ANNs, ...\n\nFollowing this idea, other powerfull custom loss functions can be created.\n\n**All the code of this notebook, except the new loss function, is an adaptation of the excellent notebook by Martin Kovacevic Buvinic\n(@ragnar123). All credit goes to him!!!!: [Amex LGBM Dart CV 0.7977](https://www.kaggle.com/code/ragnar123/amex-lgbm-dart-cv-0-7977)**","metadata":{}},{"cell_type":"code","source":"# ====================================================\n# Library\n# ====================================================\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')\nimport scipy as sp\nimport numpy as np\nimport pandas as pd\npd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\npd.set_option('display.width', 1000)\nfrom tqdm.auto import tqdm\nimport itertools","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:49:35.289517Z","iopub.execute_input":"2022-08-13T13:49:35.290082Z","iopub.status.idle":"2022-08-13T13:49:35.406818Z","shell.execute_reply.started":"2022-08-13T13:49:35.290055Z","shell.execute_reply":"2022-08-13T13:49:35.405576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================================\n# Get the difference\n# ====================================================\ndef get_difference(data, num_features):\n    df1 = []\n    customer_ids = []\n    for customer_id, df in tqdm(data.groupby(['customer_ID'])):\n        # Get the differences\n        diff_df1 = df[num_features].diff(1).iloc[[-1]].values.astype(np.float32)\n        # Append to lists\n        df1.append(diff_df1)\n        customer_ids.append(customer_id)\n    # Concatenate\n    df1 = np.concatenate(df1, axis = 0)\n    # Transform to dataframe\n    df1 = pd.DataFrame(df1, columns = [col + '_diff1' for col in df[num_features].columns])\n    # Add customer id\n    df1['customer_ID'] = customer_ids\n    return df1\n\n# ====================================================\n# Read & preprocess data and save it to disk\n# ====================================================\ndef read_preprocess_data():\n    train = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/train.parquet')\n    features = train.drop(['customer_ID', 'S_2'], axis = 1).columns.to_list()\n    cat_features = [\n        \"B_30\",\n        \"B_38\",\n        \"D_114\",\n        \"D_116\",\n        \"D_117\",\n        \"D_120\",\n        \"D_126\",\n        \"D_63\",\n        \"D_64\",\n        \"D_66\",\n        \"D_68\",\n    ]\n    num_features = [col for col in features if col not in cat_features]\n    print('Starting training feature engineer...')\n    train_num_agg = train.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    train_num_agg.columns = ['_'.join(x) for x in train_num_agg.columns]\n    train_num_agg.reset_index(inplace = True)\n    train_cat_agg = train.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    train_cat_agg.columns = ['_'.join(x) for x in train_cat_agg.columns]\n    train_cat_agg.reset_index(inplace = True)\n    train_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\n    # Transform float64 columns to float32\n    cols = list(train_num_agg.dtypes[train_num_agg.dtypes == 'float64'].index)\n    for col in tqdm(cols):\n        train_num_agg[col] = train_num_agg[col].astype(np.float32)\n    # Transform int64 columns to int32\n    cols = list(train_cat_agg.dtypes[train_cat_agg.dtypes == 'int64'].index)\n    for col in tqdm(cols):\n        train_cat_agg[col] = train_cat_agg[col].astype(np.int32)\n    # Get the difference\n    train_diff = get_difference(train, num_features)\n    train = train_num_agg.merge(train_cat_agg, how = 'inner', on = 'customer_ID').merge(train_diff, how = 'inner', on = 'customer_ID').merge(train_labels, how = 'inner', on = 'customer_ID')\n    del train_num_agg, train_cat_agg, train_diff\n    gc.collect()\n    test = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/test.parquet')\n    print('Starting test feature engineer...')\n    test_num_agg = test.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n    test_num_agg.reset_index(inplace = True)\n    test_cat_agg = test.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n    test_cat_agg.reset_index(inplace = True)\n    # Transform float64 columns to float32\n    cols = list(test_num_agg.dtypes[test_num_agg.dtypes == 'float64'].index)\n    for col in tqdm(cols):\n        test_num_agg[col] = test_num_agg[col].astype(np.float32)\n    # Transform int64 columns to int32\n    cols = list(test_cat_agg.dtypes[test_cat_agg.dtypes == 'int64'].index)\n    for col in tqdm(cols):\n        test_cat_agg[col] = test_cat_agg[col].astype(np.int32)\n    # Get the difference\n    test_diff = get_difference(test, num_features)\n    test = test_num_agg.merge(test_cat_agg, how = 'inner', on = 'customer_ID').merge(test_diff, how = 'inner', on = 'customer_ID')\n    del test_num_agg, test_cat_agg, test_diff\n    gc.collect()\n    # Save files to disk\n    train.to_parquet('/content/drive/MyDrive/Amex/train_fe.parquet')\n    test.to_parquet('/content/drive/MyDrive/Amex/test_fe.parquet')\n\n# Read & Preprocess Data\n# read_preprocess_data()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:49:36.147527Z","iopub.execute_input":"2022-08-13T13:49:36.148274Z","iopub.status.idle":"2022-08-13T13:49:36.166473Z","shell.execute_reply.started":"2022-08-13T13:49:36.148246Z","shell.execute_reply":"2022-08-13T13:49:36.165661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training & Inference","metadata":{}},{"cell_type":"code","source":"# ====================================================\n# Library\n# ====================================================\nimport os\nimport gc\nimport warnings\nwarnings.filterwarnings('ignore')\nimport random\nimport scipy as sp\nimport numpy as np\nimport pandas as pd\nimport joblib\nimport itertools\npd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\npd.set_option('display.width', 1000)\nfrom tqdm.auto import tqdm\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport lightgbm as lgb\nfrom itertools import combinations\n\n# ====================================================\n# Configurations\n# ====================================================\nclass CFG:\n    input_dir = '../input/amex-defaut-predict-competition-fe/' #'/content/data/'\n    seed = 42\n    n_folds = 5\n    target = 'target'\n    boosting_type = 'dart'\n    metric = 'binary_logloss'\n\n# ====================================================\n# Seed everything\n# ====================================================\ndef seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n\n# ====================================================\n# Read data\n# ====================================================\ndef read_data():\n    train = pd.read_parquet(CFG.input_dir + 'train_fe.parquet')\n    test = pd.read_parquet(CFG.input_dir + 'test_fe.parquet')\n    return train, test\n\n# ====================================================\n# Amex metric\n# ====================================================\ndef amex_metric(y_true, y_pred):\n    labels = np.transpose(np.array([y_true, y_pred]))\n    labels = labels[labels[:, 1].argsort()[::-1]]\n    weights = np.where(labels[:,0]==0, 20, 1)\n    cut_vals = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n    gini = [0,0]\n    for i in [1,0]:\n        labels = np.transpose(np.array([y_true, y_pred]))\n        labels = labels[labels[:, i].argsort()[::-1]]\n        weight = np.where(labels[:,0]==0, 20, 1)\n        weight_random = np.cumsum(weight / np.sum(weight))\n        total_pos = np.sum(labels[:, 0] *  weight)\n        cum_pos_found = np.cumsum(labels[:, 0] * weight)\n        lorentz = cum_pos_found / total_pos\n        gini[i] = np.sum((lorentz - weight_random) * weight)\n    return 0.5 * (gini[1]/gini[0] + top_four)\n\n# ====================================================\n# LGBM amex metric\n# ====================================================\ndef lgb_amex_metric(y_pred, y_true):\n    y_true = y_true.get_label()\n    return 'amex_metric', amex_metric(y_true, y_pred), True\n\ndef save_model():\n    def callback(env):\n        global max_score\n        global num_fold_iter\n        global CUSTOM_LOSS\n        iteration = env.iteration\n        \n#         print(env.evaluation_result_list)\n        if CUSTOM_LOSS==None:\n            score = env.evaluation_result_list[3][2]\n        else:\n            score = env.evaluation_result_list[1][2]\n#         if iteration % 100 == 0:\n#             print('iteration {}, score= {:.05f}'.format(iteration,score))\n        if score > max_score:\n#             print(f'Model saved in iter={iteration} score={score}')\n            max_score = score\n            env.model.save_model(f'/kaggle/working/model_fold{num_fold_iter}.lgb')\n    callback.order = 0\n    return callback\n\n\n\n# Global parameters\nmax_score = 0.0\nnum_fold_iter = 0\n\n# ====================================================\n# Train & Evaluate\n# ====================================================\ndef train_and_evaluate(train, test):\n    global NUM_BOOST_ROUND\n    global max_score\n    global num_fold_iter\n    global CUSTOM_LOSS\n            \n    # Get feature list\n    features = [col for col in train.columns if col not in ['customer_ID', CFG.target]]\n    params = {\n        'objective': 'binary',\n#         'metric': CFG.metric,\n        'boosting': CFG.boosting_type,\n        'seed': CFG.seed,\n        'num_leaves': 100,\n        'learning_rate': 0.01,\n        'feature_fraction': 0.20,\n        'bagging_freq': 10,\n        'bagging_fraction': 0.50,\n        'n_jobs': -1,\n        'lambda_l2': 2,\n        'min_data_in_leaf': 40,\n        'verbose': -1,\n        }\n    # Create a numpy array to store test predictions\n    test_predictions = np.zeros(len(test))\n    # Create a numpy array to store out of folds predictions\n    oof_predictions = np.zeros(len(train))\n    kfold = StratifiedKFold(n_splits = CFG.n_folds, shuffle = True, random_state = CFG.seed)\n    all_eval_dict = []\n    for fold, (trn_ind, val_ind) in enumerate(kfold.split(train, train[CFG.target])):\n        num_fold_iter = fold\n        max_score = 0.0\n        \n#         print(' ')\n#         print('-'*50)\n#         print(f'Training fold {fold} with {len(features)} features...')\n        x_train, x_val = train[features].iloc[trn_ind], train[features].iloc[val_ind]\n        y_train, y_val = train[CFG.target].iloc[trn_ind], train[CFG.target].iloc[val_ind]\n        lgb_train = lgb.Dataset(x_train, y_train, categorical_feature = cat_features)\n        lgb_valid = lgb.Dataset(x_val, y_val, categorical_feature = cat_features)\n        \n        eval_dict = dict()\n        model = lgb.train(\n            params = params,\n            train_set = lgb_train,\n            num_boost_round = NUM_BOOST_ROUND, #10500,\n            valid_sets = [lgb_train, lgb_valid],\n#             early_stopping_rounds = 1500,\n            verbose_eval = False,\n            feval = lgb_amex_metric,\n            fobj = CUSTOM_LOSS,\n            callbacks=[save_model(),\n                     lgb.log_evaluation(500), \n                     lgb.record_evaluation(eval_dict)]\n            )\n        \n        all_eval_dict.append(eval_dict)\n        model = lgb.Booster(model_file=f'/kaggle/working/model_fold{num_fold_iter}.lgb')\n        # Predict validation\n        val_pred = model.predict(x_val)\n        # Add to out of folds array\n        oof_predictions[val_ind] = val_pred\n        # Predict the test set\n        test_pred = model.predict(test[features])\n        test_predictions += test_pred / CFG.n_folds\n        # Compute fold metric\n        score = amex_metric(y_val, val_pred)\n        print(f'Our fold {fold} CV score is {score}')\n        del x_train, x_val, y_train, y_val, lgb_train, lgb_valid\n        gc.collect()\n        \n    # Compute out of folds metric\n    score = amex_metric(train[CFG.target], oof_predictions)\n    print(f'Our out of folds CV score is {score}')\n    # Create a dataframe to store out of folds predictions\n    oof_df = pd.DataFrame({'customer_ID': train['customer_ID'], 'target': train[CFG.target], 'prediction': oof_predictions})\n    oof_df.to_csv(f'/kaggle/working/cooof_lgbm_{CFG.boosting_type}_baseline_{CFG.n_folds}fold_seed{CFG.seed}.csv', index = False)\n    # Create a dataframe to store test prediction\n    test_df = pd.DataFrame({'customer_ID': test['customer_ID'], 'prediction': test_predictions})\n#     test_df.to_csv(f'/kaggle/working/test_lgbm_{CFG.boosting_type}_baseline_{CFG.n_folds}fold_seed{CFG.seed}.csv', index = False)\n    test_df.to_csv(f'/kaggle/working/submission.csv', index = False)\n    return score, all_eval_dict","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:49:37.505300Z","iopub.execute_input":"2022-08-13T13:49:37.506890Z","iopub.status.idle":"2022-08-13T13:49:39.013254Z","shell.execute_reply.started":"2022-08-13T13:49:37.506806Z","shell.execute_reply":"2022-08-13T13:49:39.012239Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load data and preprocess\nseed_everything(CFG.seed)\ntrain, test = read_data()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:49:39.015027Z","iopub.execute_input":"2022-08-13T13:49:39.015343Z","iopub.status.idle":"2022-08-13T13:50:17.589332Z","shell.execute_reply.started":"2022-08-13T13:49:39.015319Z","shell.execute_reply":"2022-08-13T13:50:17.588663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use only 50k training rows (remove to use the complete training and testing data)\nseed_everything(CFG.seed)\ntrain = train.sample(50000).reset_index()\ntest = test.sample(100).reset_index()\nALL_DATA = False # Not save submission\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:50:17.590624Z","iopub.execute_input":"2022-08-13T13:50:17.591066Z","iopub.status.idle":"2022-08-13T13:50:18.440471Z","shell.execute_reply.started":"2022-08-13T13:50:17.591041Z","shell.execute_reply":"2022-08-13T13:50:18.438053Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Label encode categorical features\ncat_features = [\n    \"B_30\",\n    \"B_38\",\n    \"D_114\",\n    \"D_116\",\n    \"D_117\",\n    \"D_120\",\n    \"D_126\",\n    \"D_63\",\n    \"D_64\",\n    \"D_66\",\n    \"D_68\"\n]\ncat_features = [f\"{cf}_last\" for cf in cat_features]\nfor cat_col in cat_features:\n    encoder = LabelEncoder()\n    train[cat_col] = encoder.fit_transform(train[cat_col])\n    test[cat_col] = encoder.transform(test[cat_col])\n# Round last float features to 2 decimal place\nnum_cols = list(train.dtypes[(train.dtypes == 'float32') | (train.dtypes == 'float64')].index)\nnum_cols = [col for col in num_cols if 'last' in col]\nfor col in num_cols:\n    train[col + '_round2'] = train[col].round(2)\n    test[col + '_round2'] = test[col].round(2)\n# Get the difference between last and mean\nnum_cols = [col for col in train.columns if 'last' in col]\nnum_cols = [col[:-5] for col in num_cols if 'round' not in col]\nfor col in num_cols:\n    try:\n        train[f'{col}_last_mean_diff'] = train[f'{col}_last'] - train[f'{col}_mean']\n        test[f'{col}_last_mean_diff'] = test[f'{col}_last'] - test[f'{col}_mean']\n    except:\n        pass\n# Transform float64 and float32 to float16\nnum_cols = list(train.dtypes[(train.dtypes == 'float32') | (train.dtypes == 'float64')].index)\nfor col in tqdm(num_cols):\n    train[col] = train[col].astype(np.float16)\n    test[col] = test[col].astype(np.float16)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:50:18.444545Z","iopub.execute_input":"2022-08-13T13:50:18.445082Z","iopub.status.idle":"2022-08-13T13:50:34.748432Z","shell.execute_reply.started":"2022-08-13T13:50:18.445040Z","shell.execute_reply":"2022-08-13T13:50:34.746016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Original","metadata":{}},{"cell_type":"code","source":"# Global Vars\nNUM_BOOST_ROUND = 1000 # Change to 10500!!!!!\nCUSTOM_LOSS = None\nMULT_NO4PERC = 0.0","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:50:34.752516Z","iopub.execute_input":"2022-08-13T13:50:34.753144Z","iopub.status.idle":"2022-08-13T13:50:34.760229Z","shell.execute_reply.started":"2022-08-13T13:50:34.753088Z","shell.execute_reply":"2022-08-13T13:50:34.758999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Original loss (logloss)\nCUSTOM_LOSS=None\nscore, all_eval_dict = train_and_evaluate(train, test)\nprint(score)\n\nif ALL_DATA:\n    sub = pd.read_csv(f'/kaggle/working/submission.csv')\n    sub.to_csv(f'submission_orig_score{score}.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T13:50:34.761520Z","iopub.execute_input":"2022-08-13T13:50:34.761880Z","iopub.status.idle":"2022-08-13T14:07:29.440403Z","shell.execute_reply.started":"2022-08-13T13:50:34.761855Z","shell.execute_reply":"2022-08-13T14:07:29.438721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Using LGBM Custom Loss","metadata":{}},{"cell_type":"code","source":"# #Search the best MULT_NO4PERC\n# res = []\n# MAX_WEIGHTS = 2.0\n# for MULT_NO4PERC in np.arange(0.0,10.0,0.5):\n#     if MULT_NO4PERC==0.0:\n#         CUSTOM_LOSS=None\n#     else:\n#         CUSTOM_LOSS=weighted_logloss\n        \n#     score, all_eval_dict = train_and_evaluate(train, test)\n#     res.append(dict(MULT_NO4PERC=MULT_NO4PERC, MAX_WEIGHTS=MAX_WEIGHTS, SCORE=score))\n#     display(pd.DataFrame(res))","metadata":{"execution":{"iopub.status.busy":"2022-08-13T14:07:29.442941Z","iopub.execute_input":"2022-08-13T14:07:29.443740Z","iopub.status.idle":"2022-08-13T15:22:42.510591Z","shell.execute_reply.started":"2022-08-13T14:07:29.443700Z","shell.execute_reply":"2022-08-13T15:22:42.507691Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Example of Custom Loss\nMULT_NO4PERC = 5.0 \nMAX_WEIGHTS = 2.0\nlen_preds = 458913//5  # OOF preds\ntop4_position = len_preds*0.04\n\nplt.plot(1+np.exp(-MULT_NO4PERC*np.linspace(MAX_WEIGHTS-1,0,len_preds)))\nplt.vlines(x=top4_position, ymin=1, ymax=MAX_WEIGHTS, color='r')\nplt.text(0, 1.5, 'top 4%', color='r', rotation=90)\nplt.title('Weights')","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:24:13.909081Z","iopub.execute_input":"2022-08-13T15:24:13.909445Z","iopub.status.idle":"2022-08-13T15:24:14.101892Z","shell.execute_reply.started":"2022-08-13T15:24:13.909419Z","shell.execute_reply":"2022-08-13T15:24:14.100743Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MULT_NO4PERC = 5.0 \nMAX_WEIGHTS = 2.0\nCUSTOM_LOSS = weighted_logloss\nscore, all_eval_dict = train_and_evaluate(train, test)\nprint(score)\n\nif ALL_DATA:\n    sub = pd.read_csv(f'/kaggle/working/submission.csv')\n    sub.to_csv(f'submission_MULT_N04_{MULT_NO4PERC}_score{score}.csv', index = False)","metadata":{"execution":{"iopub.status.busy":"2022-08-13T15:24:21.733304Z","iopub.execute_input":"2022-08-13T15:24:21.733707Z","iopub.status.idle":"2022-08-13T15:41:57.285956Z","shell.execute_reply.started":"2022-08-13T15:24:21.733679Z","shell.execute_reply":"2022-08-13T15:41:57.283506Z"},"trusted":true},"execution_count":null,"outputs":[]}]}