{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\n\nfrom sklearn.model_selection import train_test_split\n\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-12T21:57:41.434625Z","iopub.execute_input":"2022-08-12T21:57:41.435145Z","iopub.status.idle":"2022-08-12T21:57:43.029074Z","shell.execute_reply.started":"2022-08-12T21:57:41.43505Z","shell.execute_reply":"2022-08-12T21:57:43.027923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    random_state = 4222\n    kaggle = True","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:57:45.427581Z","iopub.execute_input":"2022-08-12T21:57:45.427994Z","iopub.status.idle":"2022-08-12T21:57:45.434518Z","shell.execute_reply.started":"2022-08-12T21:57:45.42796Z","shell.execute_reply":"2022-08-12T21:57:45.432778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocessing(df, cat_features,num_features, i = 'train'):\n    \n    cid = pd.Categorical(df.pop('customer_ID'), ordered = True)\n    last = (cid != np.roll(cid, -1)) \n    penul = np.roll(last, -1)\n    lt2 = (cid != np.roll(cid, -2))\n    features = cat_features + num_features\n    \n    if 'target' in df.columns:\n        df.drop(columns=['target'], inplace=True)\n    gc.collect()\n    print('Read', i)\n        \n    df_last = (df.loc[last,features]\n              .rename(columns={f: f\"{f}_lt\" for f in features})\n              .set_index(np.asarray(cid[last]))\n             )\n    gc.collect()\n    print('Computed last', i)\n    \n    df_std = (df\n              .groupby(cid)\n              .std()[num_features]\n              .rename(columns={f: f\"{f}_std\" for f in num_features})\n            )\n    gc.collect()\n    print(\"computed std\", i)\n\n    df_avg = (df\n              .groupby(cid)\n              .mean()[num_features]\n              .rename(columns={f: f\"{f}_avg\" for f in num_features})\n            )\n    gc.collect()\n    print(\"computed avg\", i)\n        \n    df = pd.concat([df_last,df_std, df_avg], axis=1)\n    \n    del df_last, df_std, df_avg,cid, last, penul, lt2, features\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:57:46.949361Z","iopub.execute_input":"2022-08-12T21:57:46.950545Z","iopub.status.idle":"2022-08-12T21:57:46.962845Z","shell.execute_reply.started":"2022-08-12T21:57:46.950508Z","shell.execute_reply":"2022-08-12T21:57:46.961448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_parquet(f'/kaggle/input/amex-data-integer-dtypes-parquet-format/train.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:57:48.498912Z","iopub.execute_input":"2022-08-12T21:57:48.499359Z","iopub.status.idle":"2022-08-12T21:58:21.17242Z","shell.execute_reply.started":"2022-08-12T21:57:48.499308Z","shell.execute_reply":"2022-08-12T21:58:21.170402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test = pd.read_parquet(f'/kaggle/input/amex-data-integer-dtypes-parquet-format/test.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:58:21.176355Z","iopub.execute_input":"2022-08-12T21:58:21.177132Z","iopub.status.idle":"2022-08-12T21:58:21.18844Z","shell.execute_reply.started":"2022-08-12T21:58:21.177061Z","shell.execute_reply":"2022-08-12T21:58:21.186382Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nnum_features = [col for col in train.columns if col not in cat_features + [\"target\", \"customer_ID\", \"S_2\"] ]","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:58:21.191557Z","iopub.execute_input":"2022-08-12T21:58:21.192861Z","iopub.status.idle":"2022-08-12T21:58:21.206286Z","shell.execute_reply.started":"2022-08-12T21:58:21.192803Z","shell.execute_reply":"2022-08-12T21:58:21.204313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[cat_features] = train[cat_features].fillna('other')","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:58:21.212Z","iopub.execute_input":"2022-08-12T21:58:21.212797Z","iopub.status.idle":"2022-08-12T21:58:21.803752Z","shell.execute_reply.started":"2022-08-12T21:58:21.212733Z","shell.execute_reply":"2022-08-12T21:58:21.801592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"null_val = train.isnull().sum().sort_values(ascending=False)/train.shape[0]","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:58:21.805845Z","iopub.execute_input":"2022-08-12T21:58:21.807251Z","iopub.status.idle":"2022-08-12T21:58:25.011857Z","shell.execute_reply.started":"2022-08-12T21:58:21.807202Z","shell.execute_reply":"2022-08-12T21:58:25.010608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cols_drop = list(null_val[null_val>0.5].index)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:58:25.013582Z","iopub.execute_input":"2022-08-12T21:58:25.014146Z","iopub.status.idle":"2022-08-12T21:58:25.021295Z","shell.execute_reply.started":"2022-08-12T21:58:25.014085Z","shell.execute_reply":"2022-08-12T21:58:25.019419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[cols_drop] = train[cols_drop].fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:58:25.023436Z","iopub.execute_input":"2022-08-12T21:58:25.024097Z","iopub.status.idle":"2022-08-12T21:58:25.721189Z","shell.execute_reply.started":"2022-08-12T21:58:25.024009Z","shell.execute_reply":"2022-08-12T21:58:25.719812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[num_features] = train[num_features].fillna(train[num_features].median())","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:58:25.72284Z","iopub.execute_input":"2022-08-12T21:58:25.723233Z","iopub.status.idle":"2022-08-12T21:58:48.71364Z","shell.execute_reply.started":"2022-08-12T21:58:25.723198Z","shell.execute_reply":"2022-08-12T21:58:48.712415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = preprocessing(train, cat_features, num_features)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:58:48.715463Z","iopub.execute_input":"2022-08-12T21:58:48.715859Z","iopub.status.idle":"2022-08-12T21:59:36.934398Z","shell.execute_reply.started":"2022-08-12T21:58:48.715816Z","shell.execute_reply":"2022-08-12T21:59:36.933057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.fillna(0)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:59:36.938081Z","iopub.execute_input":"2022-08-12T21:59:36.938494Z","iopub.status.idle":"2022-08-12T21:59:38.991895Z","shell.execute_reply.started":"2022-08-12T21:59:36.938461Z","shell.execute_reply":"2022-08-12T21:59:38.990766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = [feat for feat in train.columns if feat != 'customer_ID' and feat != 'target' and feat != \"S_2\"]\nlen(features)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:59:38.993175Z","iopub.execute_input":"2022-08-12T21:59:38.993583Z","iopub.status.idle":"2022-08-12T21:59:39.007555Z","shell.execute_reply.started":"2022-08-12T21:59:38.993547Z","shell.execute_reply":"2022-08-12T21:59:39.005911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv').target.values\nprint(f\"target shape: {target.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:59:39.008963Z","iopub.execute_input":"2022-08-12T21:59:39.009405Z","iopub.status.idle":"2022-08-12T21:59:39.978297Z","shell.execute_reply.started":"2022-08-12T21:59:39.009345Z","shell.execute_reply":"2022-08-12T21:59:39.9766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    if isinstance(y_true, np.ndarray):\n            y_true = pd.DataFrame(y_true, columns = [\"target\"])\n    \n    if isinstance(y_pred, np.ndarray):\n            y_pred = pd.DataFrame(y_pred, columns = [\"prediction\"])\n            #y_pred[\"prediction\"] = y_pred\n    \n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n      \n        df['weight'] = df[\"target\"].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df[\"target\"] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df[\"target\"].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df[\"target\"] * df['weight']).sum()\n        df['cum_pos_found'] = (df[\"target\"] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    d = top_four_percent_captured(y_true, y_pred)\n    g = normalized_weighted_gini(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:59:39.979984Z","iopub.execute_input":"2022-08-12T21:59:39.980658Z","iopub.status.idle":"2022-08-12T21:59:39.999498Z","shell.execute_reply.started":"2022-08-12T21:59:39.980604Z","shell.execute_reply":"2022-08-12T21:59:39.997948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X = pd.DataFrame(X_train_pca).copy()\n# y = target.copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:59:40.001584Z","iopub.execute_input":"2022-08-12T21:59:40.00203Z","iopub.status.idle":"2022-08-12T21:59:40.01703Z","shell.execute_reply.started":"2022-08-12T21:59:40.001994Z","shell.execute_reply":"2022-08-12T21:59:40.015511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lgb_amex_metric(y_true, y_pred):\n    \"\"\"The competition metric with lightgbm's calling convention\"\"\"\n    return ('amex',\n            amex_metric(y_true, y_pred),\n            True)","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:59:40.018736Z","iopub.execute_input":"2022-08-12T21:59:40.01913Z","iopub.status.idle":"2022-08-12T21:59:40.028614Z","shell.execute_reply.started":"2022-08-12T21:59:40.019082Z","shell.execute_reply":"2022-08-12T21:59:40.027042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import GridSearchCV\nfrom sklearn.metrics import make_scorer\nfrom sklearn.ensemble import RandomForestClassifier\nparam_grid = { \n    'n_estimators': [200, 500, 350],\n    'max_features': ['auto', 'sqrt', 'log2'],\n    'max_depth' : [4,5,6,7,8, None],\n    'criterion' :['gini', 'entropy']\n}\n\nscore=make_scorer(amex_metric,greater_is_better=True)\nclf=RandomForestClassifier()\nmnn= GridSearchCV(clf, param_grid=param_grid, cv= 5, scoring=score)\nknn = mnn.fit(train,target) ","metadata":{"execution":{"iopub.status.busy":"2022-08-12T21:59:40.030526Z","iopub.execute_input":"2022-08-12T21:59:40.030902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mnn.best_params_","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import optuna  # pip install optuna\nfrom sklearn.metrics import log_loss\nfrom sklearn.model_selection import StratifiedKFold\nfrom optuna.integration import LightGBMPruningCallback\nimport lightgbm as lgbm\n\ndef objective(trial, X, y):\n    param_grid = {\n        #         \"device_type\": trial.suggest_categorical(\"device_type\", ['gpu']),\n        \"n_estimators\": trial.suggest_categorical(\"n_estimators\", [10000]),\n        \"learning_rate\": trial.suggest_float(\"learning_rate\", 0.01, 0.3),\n        \"num_leaves\": trial.suggest_int(\"num_leaves\", 20, 3000, step=20),\n        \"max_depth\": trial.suggest_int(\"max_depth\", 3, 12),\n        \"min_data_in_leaf\": trial.suggest_int(\"min_data_in_leaf\", 200, 10000, step=100),\n        \"max_bin\": trial.suggest_int(\"max_bin\", 200, 300),\n        \"lambda_l1\": trial.suggest_int(\"lambda_l1\", 0, 100, step=5),\n        \"lambda_l2\": trial.suggest_int(\"lambda_l2\", 0, 100, step=5),\n        \"min_gain_to_split\": trial.suggest_float(\"min_gain_to_split\", 0, 15),\n        \"bagging_fraction\": trial.suggest_float(\n            \"bagging_fraction\", 0.2, 0.95, step=0.1\n        ),\n        \"bagging_freq\": trial.suggest_categorical(\"bagging_freq\", [1]),\n        \"used_ram_limit\": \"3gb\",\n        \"feature_fraction\": trial.suggest_float(\n            \"feature_fraction\", 0.2, 0.95, step=0.1\n        ),\n    } # to be filled in lataer\n    cv = StratifiedKFold(n_splits=5, shuffle=True, random_state=1121218)\n\n    cv_scores = np.empty(5)\n    for idx, (train_idx, test_idx) in enumerate(cv.split(train, target)):\n        X_train, X_test = train.iloc[train_idx], train.iloc[test_idx]\n        y_train, y_test = target[train_idx], target[test_idx]\n\n        model = lgbm.LGBMClassifier(objective=\"binary\", **param_grid)\n        model.fit(\n            X_train,\n            y_train,\n            eval_set=[(X_test, y_test)],\n            eval_metric=lgb_amex_metric,\n            early_stopping_rounds=100,\n            callbacks=[log_evaluation(50)],  # Add a pruning callback\n        )\n        preds = model.predict_proba(X_test)\n        cv_scores[idx] = log_loss(y_test, preds)\n\n    return np.mean(cv_scores)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study = optuna.create_study(direction=\"minimize\", study_name=\"LGBM Classifier\")\nfunc = lambda trial: objective(trial, train, target)\nstudy.optimize(func, n_trials=20)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"\\tBest value (rmse): {study.best_value:.5f}\")\nprint(f\"\\tBest params:\")\n\nfor key, value in study.best_params.items():\n    print(f\"\\t\\t{key}: {value}\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import catboost as cb\nimport optuna\ndef objective2(trial, train, target):\n    train_x, valid_x, train_y, valid_y = train_test_split(train,target, test_size=0.3)\n\n    param = {\n        \"objective\": trial.suggest_categorical(\"objective\", [\"Logloss\", \"CrossEntropy\"]),\n        \"colsample_bylevel\": trial.suggest_float(\"colsample_bylevel\", 0.01, 0.1),\n        \"depth\": trial.suggest_int(\"depth\", 1, 12),\n        \"boosting_type\": trial.suggest_categorical(\"boosting_type\", [\"Ordered\", \"Plain\"]),\n        \"bootstrap_type\": trial.suggest_categorical(\n            \"bootstrap_type\", [\"Bayesian\", \"Bernoulli\", \"MVS\"]\n        ),\n#         \"used_ram_limit\": \"3gb\",\n    }\n\n    if param[\"bootstrap_type\"] == \"Bayesian\":\n        param[\"bagging_temperature\"] = trial.suggest_float(\"bagging_temperature\", 0, 10)\n    elif param[\"bootstrap_type\"] == \"Bernoulli\":\n        param[\"subsample\"] = trial.suggest_float(\"subsample\", 0.1, 1)\n\n    gbm = cb.CatBoostClassifier(**param, eval_metric=lgb_amex_metric)\n\n    gbm.fit(\n           train_x,\n            train_y,\n            eval_set=[(valid_x, valid_y)],\n            early_stopping_rounds=100,\n            callbacks=[log_evaluation(50)] )\n\n    preds = gbm.predict(valid_x)\n    pred_labels = np.rint(preds)\n    accuracy = accuracy_score(valid_y, pred_labels)\n    return accuracy","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"study2 = optuna.create_study(direction=\"minimize\", study_name=\"Catboost Classifier\")\nfunc2 = lambda trial: objective2(trial, train, target)\nstudy2.optimize(func2, n_trials=50)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Number of finished trials: {}\".format(len(study2.trials)))\n\nprint(\"Best trial:\")\ntrial = study2.best_trial\n\nprint(\"  Value: {}\".format(trial.value))\n\nprint(\"  Params: \")\nfor key, value in trial.params.items():\n    print(\"    {}: {}\".format(key, value))","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"search_params = { \n    'learning_rate' : 0.065,\n    'lambda_l1': 8.481514197781607,\n    'lambda_l2': 0, #0.0004266339834880936,\n    'num_leaves': 27,\n    'feature_fraction': 0.484,\n    'bagging_fraction': 0.8477112190030014,\n    'bagging_freq': 2,\n    'min_child_samples': 20\n}\n\nfixed_params={\n    'objective': 'binary',\n    'metric': 'custom', #'binay_logloss',\n    'boosting_type' : 'gbdt',\n    #'force_row_wise' : True,\n    #'device': 'gpu',\n    'random_state' : config.random_state,\n    #'extra_trees' : True,\n    #'feature_pre_filter': False,\n    'n_estimators': 600,\n    'early_stopping_round': 50\n}","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_modelo(df,target,features):\n    \n    x = df[features]\n    y = pd.Series(target)\n    \n    #enc = OrdinalEncoder()\n    #x[cat_features] = enc.fit_transform(x[cat_features])\n\n    X_train, X_test, y_train, y_test = train_test_split(x,y,test_size = 0.3,\n                                random_state = config.random_state, stratify = y)\n    \n    model = LGBMClassifier(**fixed_params, **search_params)\n    \n    model.fit(\n        X_train, y_train, \n        eval_set=[(X_test,y_test)],\n        eval_metric= lgb_amex_metric,\n        callbacks=[log_evaluation(50)]\n    )\n    \n    del x,y,X_train, y_train\n    \n    return model, X_test, y_test","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel, X_test, y_test = train_modelo(train,target,features)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = pd.DataFrame(y_test, columns = [\"target\"])\ny_pred = pd.DataFrame(y_test.copy(), columns = [\"prediction\"])\n\ny_pred[\"prediction\"] = model.predict_proba(X_test)[:,1]\namex_metric(y_test, y_pred)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train, target, X_test, y_test, y_pred\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n    if usecols is not None: df = pd.read_parquet(path,columns = usecols)\n    else: df = pd.read_parquet(path)\n   \n    #df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = pd.to_datetime( df.S_2 )\n    #df = df.fillna(NAN_VALUE) \n    print('shape of data:', df.shape)\n    \n    return df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_rows(customers, test, NUM_PARTS = 4, verbose = ''):\n    chunk = len(customers)//NUM_PARTS\n    if verbose != '':\n        print(f'We will process {verbose} data as {NUM_PARTS} separate parts.')\n        print(f'There will be {chunk} customers in each part (except the last part).')\n        print('Below are number of rows in each part:')\n    rows = []\n\n    for k in range(NUM_PARTS):\n        if k == NUM_PARTS-1: cc = customers[k*chunk:]\n        else: cc = customers[k*chunk:(k+1)*chunk]\n        s = test.loc[test.customer_ID.isin(cc)].shape[0]\n        rows.append(s)\n    \n    if verbose != '': print( rows )\n    \n    return rows,chunk","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_PARTS = 4\nTEST_PATH =  '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\n\nprint(f'Reading test data...')\ntest = read_file(path = TEST_PATH, usecols = ['customer_ID','S_2'])\n\ncustomers = test[['customer_ID']].drop_duplicates().sort_index().values.flatten()\n\nrows,num_cust = get_rows(customers,test[['customer_ID']], NUM_PARTS = NUM_PARTS, verbose = 'test')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER TEST DATA IN PARTS\nskip_rows = 0\nskip_cust = 0\ntest_preds = []\n\nfor k in range(NUM_PARTS):\n    print(f'\\nReading test data...')\n    test = read_file(path = TEST_PATH)\n    test = test.iloc[skip_rows:skip_rows + rows[k]]\n    skip_rows += rows[k]\n    print(f'=> Test part {k+1} has shape', test.shape )\n          \n    test = preprocessing(test, cat_features, num_features, i = 'test')\n    if k == 0: \n        features = [feat for feat in test.columns if feat != 'customer_ID' and feat != 'target' and feat != \"S_2\"]\n \n    if k == NUM_PARTS - 1: test = test.loc[customers[skip_cust:]]\n    else: test = test.loc[customers[skip_cust:skip_cust+num_cust]]\n    skip_cust += num_cust\n\n# Clean Memory\ndel test\n_ = gc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predictions = np.concatenate\n\nsubmission = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\nsubmission.loc[:, \"prediction\"] = test_predictions\n\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}