{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Preprocessing","metadata":{"papermill":{"duration":0.002645,"end_time":"2022-07-06T15:09:30.803906","exception":false,"start_time":"2022-07-06T15:09:30.801261","status":"completed"},"pycharm":{"name":"#%% md\n"},"tags":[]}},{"cell_type":"code","source":"import gc\nimport os\nimport joblib\nimport random\nimport warnings\nimport itertools\nimport scipy as sp\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport lightgbm as lgb\nfrom itertools import combinations\npd.set_option('display.width', 1000)\npd.set_option('display.max_rows', 500)\npd.set_option('display.max_columns', 500)\nfrom sklearn.preprocessing import LabelEncoder\nimport warnings; warnings.filterwarnings('ignore')\nfrom sklearn.model_selection import StratifiedKFold, train_test_split\n\ndef get_difference(data, num_features):\n    df1 = []\n    customer_ids = []\n    for customer_id, df in tqdm(data.groupby(['customer_ID'])):\n        diff_df1 = df[num_features].diff(1).iloc[[-1]].values.astype(np.float32)\n        df1.append(diff_df1)\n        customer_ids.append(customer_id)\n    df1 = np.concatenate(df1, axis = 0)\n    df1 = pd.DataFrame(df1, columns = [col + '_diff1' for col in df[num_features].columns])\n    df1['customer_ID'] = customer_ids\n    return df1\n\ndef read_preprocess_data():\n    train = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/train.parquet')\n    train.describe()\n    train.plot()\n    features = train.drop(['customer_ID', 'S_2'], axis = 1).columns.to_list()\n    features.describe()\n    features.plot()\n    cat_features = [\n        \"B_30\",\n        \"B_38\",\n        \"D_114\",\n        \"D_116\",\n        \"D_117\",\n        \"D_120\",\n        \"D_126\",\n        \"D_63\",\n        \"D_64\",\n        \"D_66\",\n        \"D_68\",\n    ]\n    num_features = [col for col in features if col not in cat_features]\n    print('Starting training feature engineer...')\n    train_num_agg = train.groupby(\"customer_ID\")[num_features].agg(['first', 'mean', 'std', 'min', 'max', 'last'])\n    train_num_agg.describe()\n    train_num_agg.plot()\n    train_num_agg.columns = ['_'.join(x) for x in train_num_agg.columns]\n    train_num_agg.reset_index(inplace = True)\n\n    # Lag Features\n    for col in train_num_agg:\n        if 'last' in col and col.replace('last', 'first') in train_num_agg:\n            train_num_agg[col + '_lag_sub'] = train_num_agg[col] - train_num_agg[col.replace('last', 'first')]\n            train_num_agg[col + '_lag_div'] = train_num_agg[col] / train_num_agg[col.replace('last', 'first')]\n\n    train_cat_agg = train.groupby(\"customer_ID\")[cat_features].agg(['count', 'first', 'last', 'nunique'])\n    train_num_agg.describe()\n    train_num_agg.plot()\n    train_cat_agg.columns = ['_'.join(x) for x in train_cat_agg.columns]\n    train_cat_agg.reset_index(inplace = True)\n    \n    train_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\n    train_labels.describe()\n    train_labels.plot()\n    # Transform float64 columns to float32\n    cols = list(train_num_agg.dtypes[train_num_agg.dtypes == 'float64'].index)\n    for col in tqdm(cols):\n        train_num_agg[col] = train_num_agg[col].astype(np.float32)\n    # Transform int64 columns to int32\n    cols = list(train_cat_agg.dtypes[train_cat_agg.dtypes == 'int64'].index)\n    for col in tqdm(cols):\n        train_cat_agg[col] = train_cat_agg[col].astype(np.int32)\n    # Get the difference\n    train_diff = get_difference(train, num_features)\n    train = train_num_agg.merge(train_cat_agg, how = 'inner', on = 'customer_ID').merge(train_diff, how = 'inner', on = 'customer_ID').merge(train_labels, how = 'inner', on = 'customer_ID')\n    train.describe()\n    train.plot()\n    del train_num_agg, train_cat_agg, train_diff\n    gc.collect()\n    \n    # Test FE\n    test = pd.read_parquet('../input/amex-data-integer-dtypes-parquet-format/test.parquet')\n    print('Starting test feature engineer...')\n    test_num_agg = test.groupby(\"customer_ID\")[num_features].agg(['first', 'mean', 'std', 'min', 'max', 'last'])\n    test_num_agg.describe()\n    test_num_agg.plot()\n    test_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\n    test_num_agg.reset_index(inplace = True)\n\n    # Lag Features\n    for col in test_num_agg:\n        if 'last' in col and col.replace('last', 'first') in test_num_agg:\n            test_num_agg[col + '_lag_sub'] = test_num_agg[col] - test_num_agg[col.replace('last', 'first')]\n            test_num_agg[col + '_lag_div'] = test_num_agg[col] / test_num_agg[col.replace('last', 'first')]\n\n    test_cat_agg = test.groupby(\"customer_ID\")[cat_features].agg(['count', 'first', 'last', 'nunique'])\n    test_num_agg.describe()\n    test_num_agg.plot()\n    test_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\n    test_cat_agg.reset_index(inplace = True)\n    # Transform float64 columns to float32\n    cols = list(test_num_agg.dtypes[test_num_agg.dtypes == 'float64'].index)\n    for col in tqdm(cols):\n        test_num_agg[col] = test_num_agg[col].astype(np.float32)\n    # Transform int64 columns to int32\n    cols = list(test_cat_agg.dtypes[test_cat_agg.dtypes == 'int64'].index)\n    for col in tqdm(cols):\n        test_cat_agg[col] = test_cat_agg[col].astype(np.int32)\n    # Get the difference\n    test_diff = get_difference(test, num_features)\n    test = test_num_agg.merge(test_cat_agg, how = 'inner', on = 'customer_ID').merge(test_diff, how = 'inner', on = 'customer_ID')\n    test.describe()\n    test.plot()\n    del test_num_agg, test_cat_agg, test_diff\n    gc.collect()\n    # Save files to disk\n\n    train.to_parquet('train_fe_plus_plus.parquet')\n    test.to_parquet('test_fe_plus_plus.parquet')\n    \n# Read & Preprocess Data\n# read_preprocess_data()","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.191822,"end_time":"2022-07-06T15:09:30.998606","exception":false,"start_time":"2022-07-06T15:09:30.806784","status":"completed"},"pycharm":{"name":"#%%\n"},"tags":[],"execution":{"iopub.status.busy":"2022-08-09T08:00:21.967428Z","iopub.execute_input":"2022-08-09T08:00:21.968148Z","iopub.status.idle":"2022-08-09T08:00:24.192912Z","shell.execute_reply.started":"2022-08-09T08:00:21.968024Z","shell.execute_reply":"2022-08-09T08:00:24.191826Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training & Inference","metadata":{"papermill":{"duration":0.002685,"end_time":"2022-07-06T15:09:31.004476","exception":false,"start_time":"2022-07-06T15:09:31.001791","status":"completed"},"pycharm":{"name":"#%% md\n"},"tags":[]}},{"cell_type":"code","source":"class CFG:\n    seed = 42\n    n_folds = 5\n    target = 'target'\n    input_dir = '../input/amex-fe/'\n\ndef seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n\ndef read_data():\n    train = pd.read_parquet(CFG.input_dir + 'train_fe_plus_plus.parquet')\n    test = pd.read_parquet(CFG.input_dir + 'test_fe_plus_plus.parquet')\n    return train, test\n\ndef amex_metric(y_true, y_pred):\n    labels = np.transpose(np.array([y_true, y_pred]))\n    labels = labels[labels[:, 1].argsort()[::-1]]\n    weights = np.where(labels[:,0]==0, 20, 1)\n    cut_vals = labels[np.cumsum(weights) <= int(0.04012353 * np.sum(weights))]\n    top_four = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n    gini = [0,0]\n    for i in [1,0]:\n        labels = np.transpose(np.array([y_true, y_pred]))\n        labels = labels[labels[:, i].argsort()[::-1]]\n        weight = np.where(labels[:,0]==0, 20, 1)\n        weight_random = np.cumsum(weight / np.sum(weight))\n        total_pos = np.sum(labels[:, 0] *  weight)\n        cum_pos_found = np.cumsum(labels[:, 0] * weight)\n        lorentz = cum_pos_found / total_pos\n        gini[i] = np.sum((lorentz - weight_random) * weight)\n    return 0.5 * (gini[1]/gini[0] + top_four)\n\ndef amex_metric_np(preds, target):\n    indices = np.argsort(preds)[::-1]\n    preds, target = preds[indices], target[indices]\n    weight = 20.0 - target * 19.0\n    cum_norm_weight = (weight / weight.sum()).cumsum()\n    four_pct_mask = cum_norm_weight <= 0.04\n    d = np.sum(target[four_pct_mask]) / np.sum(target)\n    weighted_target = target * weight\n    lorentz = (weighted_target / weighted_target.sum()).cumsum()\n    gini = ((lorentz - cum_norm_weight) * weight).sum()\n    n_pos = np.sum(target)\n    n_neg = target.shape[0] - n_pos\n    gini_max = 10 * n_neg * (n_pos + 20 * n_neg - 19) / (n_pos + 20 * n_neg)\n    g = gini / gini_max\n    return 0.5 * (g + d)","metadata":{"papermill":{"duration":2.673873,"end_time":"2022-07-06T15:09:33.682611","exception":false,"start_time":"2022-07-06T15:09:31.008738","status":"completed"},"pycharm":{"name":"#%%\n"},"tags":[],"execution":{"iopub.status.busy":"2022-08-09T08:00:24.195892Z","iopub.execute_input":"2022-08-09T08:00:24.196847Z","iopub.status.idle":"2022-08-09T08:00:24.215249Z","shell.execute_reply.started":"2022-08-09T08:00:24.1968Z","shell.execute_reply":"2022-08-09T08:00:24.214266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training LightGBM (DART) Model\n\n- Final predictions output uploaded as a public dataset. ","metadata":{"papermill":{"duration":0.003511,"end_time":"2022-07-06T15:09:33.690195","exception":false,"start_time":"2022-07-06T15:09:33.686684","status":"completed"},"tags":[],"pycharm":{"name":"#%% md\n"}}},{"cell_type":"code","source":"def lgb_amex_metric(y_pred, y_true):\n    y_true = y_true.get_label()\n    return 'amex_metric', amex_metric(y_true, y_pred), True\n\ndef train_and_evaluate(train, test):\n    # Label encode categorical features\n    cat_features = [\n        \"B_30\",\n        \"B_38\",\n        \"D_114\",\n        \"D_116\",\n        \"D_117\",\n        \"D_120\",\n        \"D_126\",\n        \"D_63\",\n        \"D_64\",\n        \"D_66\",\n        \"D_68\"\n    ]\n    cat_features = [f\"{cf}_last\" for cf in cat_features]\n    for cat_col in cat_features:\n        encoder = LabelEncoder()\n        train[cat_col] = encoder.fit_transform(train[cat_col])\n        test[cat_col] = encoder.transform(test[cat_col])\n    # Round last float features to 2 decimal place\n    num_cols = list(train.dtypes[(train.dtypes == 'float32') | (train.dtypes == 'float64')].index)\n    num_cols = [col for col in num_cols if 'last' in col]\n    for col in num_cols:\n        train[col + '_round2'] = train[col].round(2)\n        test[col + '_round2'] = test[col].round(2)\n    # Get the difference between last and mean\n    num_cols = [col for col in train.columns if 'last' in col]\n    num_cols = [col[:-5] for col in num_cols if 'round' not in col]\n    for col in num_cols:\n        try:\n            train[f'{col}_last_mean_diff'] = train[f'{col}_last'] - train[f'{col}_mean']\n            test[f'{col}_last_mean_diff'] = test[f'{col}_last'] - test[f'{col}_mean']\n        except:\n            pass\n    # Transform float64 and float32 to float16\n    num_cols = list(train.dtypes[(train.dtypes == 'float32') | (train.dtypes == 'float64')].index)\n    for col in tqdm(num_cols):\n        train[col] = train[col].astype(np.float16)\n        test[col] = test[col].astype(np.float16)\n    # Get feature list\n    features = [col for col in train.columns if col not in ['customer_ID', CFG.target]]\n    params = {\n        'objective': 'binary',\n        'metric': \"binary_logloss\",\n        'boosting': 'dart',\n        'seed': CFG.seed,\n        'num_leaves': 100,\n        'learning_rate': 0.01,\n        'feature_fraction': 0.50,\n        'bagging_freq': 10,\n        'bagging_fraction': 0.80,\n        'n_jobs': -1,\n        'lambda_l2': 2,\n        'min_data_in_leaf': 40\n        }\n    # Create a numpy array to store test predictions\n    test_predictions = np.zeros(len(test))\n    # Create a numpy array to store out of folds predictions\n    oof_predictions = np.zeros(len(train))\n    kfold = StratifiedKFold(n_splits = CFG.n_folds, shuffle = True, random_state = CFG.seed)\n    for fold, (trn_ind, val_ind) in enumerate(kfold.split(train, train[CFG.target])):\n        print(' ')\n        print('-'*50)\n        print(f'Training fold {fold} with {len(features)} features...')\n        x_train, x_val = train[features].iloc[trn_ind], train[features].iloc[val_ind]\n        y_train, y_val = train[CFG.target].iloc[trn_ind], train[CFG.target].iloc[val_ind]\n        lgb_train = lgb.Dataset(x_train, y_train, categorical_feature = cat_features)\n        lgb_valid = lgb.Dataset(x_val, y_val, categorical_feature = cat_features)\n        model = lgb.train(\n            params = params,\n            train_set = lgb_train,\n            num_boost_round = 10500,\n            valid_sets = [lgb_train, lgb_valid],\n            early_stopping_rounds = 100,\n            verbose_eval = 500,\n            feval = lgb_amex_metric\n            )\n        # Save best model\n        joblib.dump(model, f'lgbm_fold{fold}_seed{CFG.seed}.pkl')\n        # Predict validation\n        val_pred = model.predict(x_val)\n        # Add to out of folds array\n        oof_predictions[val_ind] = val_pred\n        # Predict the test set\n        test_pred = model.predict(test[features])\n        test_predictions += test_pred / CFG.n_folds\n        # Compute fold metric\n        score = amex_metric(y_val, val_pred)\n        print(f'Our fold {fold} CV score is {score}')\n        del x_train, x_val, y_train, y_val, lgb_train, lgb_valid\n        gc.collect()\n    # Compute out of folds metric\n    score = amex_metric(train[CFG.target], oof_predictions)\n    print(f'Our out of folds CV score is {score}')\n    # Create a dataframe to store out of folds predictions\n    oof_df = pd.DataFrame({'customer_ID': train['customer_ID'], 'target': train[CFG.target], 'prediction': oof_predictions})\n    oof_df.to_csv(f'oof_lgbm_baseline_{CFG.n_folds}fold_seed{CFG.seed}.csv', index = False)\n    # Create a dataframe to store test prediction\n    test_df = pd.DataFrame({'customer_ID': test['customer_ID'], 'prediction': test_predictions})\n    test_df.to_csv(f'test_lgbm_baseline_{CFG.n_folds}fold_seed{CFG.seed}.csv', index = False)\n\n# seed_everything(CFG.seed)\n# train, test = read_data()\n# train_and_evaluate(train, test)","metadata":{"collapsed":false,"jupyter":{"outputs_hidden":false},"papermill":{"duration":0.028439,"end_time":"2022-07-06T15:09:33.721871","exception":false,"start_time":"2022-07-06T15:09:33.693432","status":"completed"},"pycharm":{"name":"#%%\n"},"tags":[],"execution":{"iopub.status.busy":"2022-08-09T08:00:24.217878Z","iopub.execute_input":"2022-08-09T08:00:24.218963Z","iopub.status.idle":"2022-08-09T08:00:24.246621Z","shell.execute_reply.started":"2022-08-09T08:00:24.218899Z","shell.execute_reply":"2022-08-09T08:00:24.24555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Prediction\n\n- Replace / comment-out this to use your own predictions from the model in the above cell.","metadata":{"papermill":{"duration":0.002816,"end_time":"2022-07-06T15:09:33.728277","exception":false,"start_time":"2022-07-06T15:09:33.725461","status":"completed"},"pycharm":{"name":"#%% md\n"},"tags":[]}},{"cell_type":"code","source":"import os\nimport pandas as pd\n\ndf_1 = pd.read_csv('../input/amex-all-feats-preds/test_lgbm_v3_5fold_seed42.csv')\ndf_1.to_csv('submission.csv', index=False)\ndf_1.describe()\ndf_1.plot()","metadata":{"papermill":{"duration":5.342453,"end_time":"2022-07-06T15:09:39.07373","exception":false,"start_time":"2022-07-06T15:09:33.731277","status":"completed"},"pycharm":{"name":"#%%\n"},"tags":[],"execution":{"iopub.status.busy":"2022-08-09T08:00:24.250786Z","iopub.execute_input":"2022-08-09T08:00:24.251655Z","iopub.status.idle":"2022-08-09T08:00:33.633324Z","shell.execute_reply.started":"2022-08-09T08:00:24.251621Z","shell.execute_reply":"2022-08-09T08:00:33.632227Z"},"trusted":true},"execution_count":null,"outputs":[]}]}