{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\n\nfrom lightgbm import LGBMClassifier, early_stopping, log_evaluation\n\nfrom sklearn.model_selection import train_test_split\n\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T14:53:29.923920Z","iopub.execute_input":"2022-08-10T14:53:29.924895Z","iopub.status.idle":"2022-08-10T14:53:29.932095Z","shell.execute_reply.started":"2022-08-10T14:53:29.924841Z","shell.execute_reply":"2022-08-10T14:53:29.931044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    random_state = 4222\n    kaggle = True","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:03:25.430095Z","iopub.execute_input":"2022-08-10T14:03:25.430565Z","iopub.status.idle":"2022-08-10T14:03:25.437504Z","shell.execute_reply.started":"2022-08-10T14:03:25.430529Z","shell.execute_reply":"2022-08-10T14:03:25.435934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocessing(df, cat_features,num_features, i = 'train'):\n    \n    cid = pd.Categorical(df.pop('customer_ID'), ordered = True)\n    last = (cid != np.roll(cid, -1)) \n    penul = np.roll(last, -1)\n    lt2 = (cid != np.roll(cid, -2))\n    features = cat_features + num_features\n    \n    if 'target' in df.columns:\n        df.drop(columns=['target'], inplace=True)\n    gc.collect()\n    print('Read', i)\n        \n    df_last = (df.loc[last,features]\n              .rename(columns={f: f\"{f}_lt\" for f in features})\n              .set_index(np.asarray(cid[last]))\n             )\n    gc.collect()\n    print('Computed last', i)\n    \n    df_std = (df\n              .groupby(cid)\n              .std()[num_features]\n              .rename(columns={f: f\"{f}_std\" for f in num_features})\n            )\n    gc.collect()\n    print(\"computed std\", i)\n\n    df_avg = (df\n              .groupby(cid)\n              .mean()[num_features]\n              .rename(columns={f: f\"{f}_avg\" for f in num_features})\n            )\n    gc.collect()\n    print(\"computed avg\", i)\n        \n    df = pd.concat([df_last,df_std, df_avg], axis=1)\n    \n    del df_last, df_std, df_avg,cid, last, penul, lt2, features\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:13:22.679646Z","iopub.execute_input":"2022-08-10T14:13:22.680047Z","iopub.status.idle":"2022-08-10T14:13:22.692912Z","shell.execute_reply.started":"2022-08-10T14:13:22.680016Z","shell.execute_reply":"2022-08-10T14:13:22.691335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_parquet(f'../input/amex-data-integer-dtypes-parquet-format/train.parquet')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:16:03.504165Z","iopub.execute_input":"2022-08-10T14:16:03.504733Z","iopub.status.idle":"2022-08-10T14:16:32.122139Z","shell.execute_reply.started":"2022-08-10T14:16:03.504684Z","shell.execute_reply":"2022-08-10T14:16:32.117822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_features = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nnum_features = [col for col in train.columns if col not in cat_features + [\"target\", \"customer_ID\", \"S_2\"] ]","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:16:57.529047Z","iopub.execute_input":"2022-08-10T14:16:57.529533Z","iopub.status.idle":"2022-08-10T14:16:57.536997Z","shell.execute_reply.started":"2022-08-10T14:16:57.529497Z","shell.execute_reply":"2022-08-10T14:16:57.536141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = preprocessing(train, cat_features, num_features)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:17:09.092436Z","iopub.execute_input":"2022-08-10T14:17:09.092858Z","iopub.status.idle":"2022-08-10T14:18:00.174597Z","shell.execute_reply.started":"2022-08-10T14:17:09.092824Z","shell.execute_reply":"2022-08-10T14:18:00.172603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = [feat for feat in train.columns if feat != 'customer_ID' and feat != 'target' and feat != \"S_2\"]\nlen(features)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:12.059250Z","iopub.execute_input":"2022-08-10T14:18:12.059722Z","iopub.status.idle":"2022-08-10T14:18:12.069206Z","shell.execute_reply.started":"2022-08-10T14:18:12.059684Z","shell.execute_reply":"2022-08-10T14:18:12.067900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target = pd.read_csv('../input/amex-default-prediction/train_labels.csv').target.values\nprint(f\"target shape: {target.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:18:38.849978Z","iopub.execute_input":"2022-08-10T14:18:38.850513Z","iopub.status.idle":"2022-08-10T14:18:39.826462Z","shell.execute_reply.started":"2022-08-10T14:18:38.850472Z","shell.execute_reply":"2022-08-10T14:18:39.824868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    if isinstance(y_true, np.ndarray):\n            y_true = pd.DataFrame(y_true, columns = [\"target\"])\n    \n    if isinstance(y_pred, np.ndarray):\n            y_pred = pd.DataFrame(y_pred, columns = [\"prediction\"])\n            #y_pred[\"prediction\"] = y_pred\n    \n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n      \n        df['weight'] = df[\"target\"].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df[\"target\"] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df[\"target\"].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df[\"target\"] * df['weight']).sum()\n        df['cum_pos_found'] = (df[\"target\"] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    d = top_four_percent_captured(y_true, y_pred)\n    g = normalized_weighted_gini(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:19:24.400495Z","iopub.execute_input":"2022-08-10T14:19:24.401417Z","iopub.status.idle":"2022-08-10T14:19:24.417685Z","shell.execute_reply.started":"2022-08-10T14:19:24.401374Z","shell.execute_reply":"2022-08-10T14:19:24.416255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lgb_amex_metric(y_true, y_pred):\n    \"\"\"The competition metric with lightgbm's calling convention\"\"\"\n    return ('amex',\n            amex_metric(y_true, y_pred),\n            True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:19:49.187317Z","iopub.execute_input":"2022-08-10T14:19:49.187772Z","iopub.status.idle":"2022-08-10T14:19:49.194749Z","shell.execute_reply.started":"2022-08-10T14:19:49.187738Z","shell.execute_reply":"2022-08-10T14:19:49.193113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"search_params = { \n    'learning_rate' : 0.065,\n    'lambda_l1': 8.481514197781607,\n    'lambda_l2': 0, #0.0004266339834880936,\n    'num_leaves': 27,\n    'feature_fraction': 0.484,\n    'bagging_fraction': 0.8477112190030014,\n    'bagging_freq': 2,\n    'min_child_samples': 20\n}\n\nfixed_params={\n    'objective': 'binary',\n    'metric': 'custom', #'binay_logloss',\n    'boosting_type' : 'gbdt',\n    #'force_row_wise' : True,\n    #'device': 'gpu',\n    'random_state' : config.random_state,\n    #'extra_trees' : True,\n    #'feature_pre_filter': False,\n    'n_estimators': 600,\n    'early_stopping_round': 50\n}","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:20:16.618895Z","iopub.execute_input":"2022-08-10T14:20:16.619364Z","iopub.status.idle":"2022-08-10T14:20:16.627424Z","shell.execute_reply.started":"2022-08-10T14:20:16.619328Z","shell.execute_reply":"2022-08-10T14:20:16.626057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_modelo(df,target,features):\n    \n    x = df[features]\n    y = pd.Series(target)\n    \n    #enc = OrdinalEncoder()\n    #x[cat_features] = enc.fit_transform(x[cat_features])\n\n    X_train, X_test, y_train, y_test = train_test_split(x,y,test_size = 0.3,\n                                random_state = config.random_state, stratify = y)\n    \n    model = LGBMClassifier(**fixed_params, **search_params)\n    \n    model.fit(\n        X_train, y_train, \n        eval_set=[(X_test,y_test)],\n        eval_metric= lgb_amex_metric,\n        callbacks=[log_evaluation(50)]\n    )\n    \n    del x,y,X_train, y_train\n    \n    return model, X_test, y_test","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:20:51.998407Z","iopub.execute_input":"2022-08-10T14:20:51.998896Z","iopub.status.idle":"2022-08-10T14:20:52.008128Z","shell.execute_reply.started":"2022-08-10T14:20:51.998862Z","shell.execute_reply":"2022-08-10T14:20:52.006556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel, X_test, y_test = train_modelo(train,target,features)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:21:06.598511Z","iopub.execute_input":"2022-08-10T14:21:06.598966Z","iopub.status.idle":"2022-08-10T14:25:25.677051Z","shell.execute_reply.started":"2022-08-10T14:21:06.598932Z","shell.execute_reply":"2022-08-10T14:25:25.675674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = pd.DataFrame(y_test, columns = [\"target\"])\ny_pred = pd.DataFrame(y_test.copy(), columns = [\"prediction\"])\n\ny_pred[\"prediction\"] = model.predict_proba(X_test)[:,1]\namex_metric(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:25:38.499675Z","iopub.execute_input":"2022-08-10T14:25:38.500195Z","iopub.status.idle":"2022-08-10T14:25:41.722525Z","shell.execute_reply.started":"2022-08-10T14:25:38.500152Z","shell.execute_reply":"2022-08-10T14:25:41.720910Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train, target, X_test, y_test, y_pred\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:26:13.582586Z","iopub.execute_input":"2022-08-10T14:26:13.583266Z","iopub.status.idle":"2022-08-10T14:26:13.789692Z","shell.execute_reply.started":"2022-08-10T14:26:13.583220Z","shell.execute_reply":"2022-08-10T14:26:13.788475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_file(path = '', usecols = None):\n    if usecols is not None: df = pd.read_parquet(path,columns = usecols)\n    else: df = pd.read_parquet(path)\n   \n    #df['customer_ID'] = df['customer_ID'].str[-16:].str.hex_to_int().astype('int64')\n    df.S_2 = pd.to_datetime( df.S_2 )\n    #df = df.fillna(NAN_VALUE) \n    print('shape of data:', df.shape)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:26:55.116361Z","iopub.execute_input":"2022-08-10T14:26:55.116817Z","iopub.status.idle":"2022-08-10T14:26:55.123783Z","shell.execute_reply.started":"2022-08-10T14:26:55.116784Z","shell.execute_reply":"2022-08-10T14:26:55.122816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_rows(customers, test, NUM_PARTS = 4, verbose = ''):\n    chunk = len(customers)//NUM_PARTS\n    if verbose != '':\n        print(f'We will process {verbose} data as {NUM_PARTS} separate parts.')\n        print(f'There will be {chunk} customers in each part (except the last part).')\n        print('Below are number of rows in each part:')\n    rows = []\n\n    for k in range(NUM_PARTS):\n        if k == NUM_PARTS-1: cc = customers[k*chunk:]\n        else: cc = customers[k*chunk:(k+1)*chunk]\n        s = test.loc[test.customer_ID.isin(cc)].shape[0]\n        rows.append(s)\n    \n    if verbose != '': print( rows )\n    \n    return rows,chunk","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:27:19.890638Z","iopub.execute_input":"2022-08-10T14:27:19.891610Z","iopub.status.idle":"2022-08-10T14:27:19.901211Z","shell.execute_reply.started":"2022-08-10T14:27:19.891559Z","shell.execute_reply":"2022-08-10T14:27:19.899798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_PARTS = 4\nTEST_PATH =  '../input/amex-data-integer-dtypes-parquet-format/test.parquet'\n\nprint(f'Reading test data...')\ntest = read_file(path = TEST_PATH, usecols = ['customer_ID','S_2'])\n\ncustomers = test[['customer_ID']].drop_duplicates().sort_index().values.flatten()\n\nrows,num_cust = get_rows(customers,test[['customer_ID']], NUM_PARTS = NUM_PARTS, verbose = 'test')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:38:40.897129Z","iopub.execute_input":"2022-08-10T14:38:40.897564Z","iopub.status.idle":"2022-08-10T14:38:51.578037Z","shell.execute_reply.started":"2022-08-10T14:38:40.897530Z","shell.execute_reply":"2022-08-10T14:38:51.577133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:39:14.838593Z","iopub.execute_input":"2022-08-10T14:39:14.839387Z","iopub.status.idle":"2022-08-10T14:39:15.040039Z","shell.execute_reply.started":"2022-08-10T14:39:14.839344Z","shell.execute_reply":"2022-08-10T14:39:15.038909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# INFER TEST DATA IN PARTS\nskip_rows = 0\nskip_cust = 0\ntest_preds = []\n\nfor k in range(NUM_PARTS):\n    print(f'\\nReading test data...')\n    test = read_file(path = TEST_PATH)\n    test = test.iloc[skip_rows:skip_rows + rows[k]]\n    skip_rows += rows[k]\n    print(f'=> Test part {k+1} has shape', test.shape )\n          \n    test = preprocessing(test, cat_features, num_features, i = 'test')\n    if k == 0: \n        features = [feat for feat in test.columns if feat != 'customer_ID' and feat != 'target' and feat != \"S_2\"]\n \n    if k == NUM_PARTS - 1: test = test.loc[customers[skip_cust:]]\n    else: test = test.loc[customers[skip_cust:skip_cust+num_cust]]\n    skip_cust += num_cust\n\n# Clean Memory\ndel test\n_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:49:39.324477Z","iopub.execute_input":"2022-08-10T14:49:39.325192Z","iopub.status.idle":"2022-08-10T14:52:53.465001Z","shell.execute_reply.started":"2022-08-10T14:49:39.325147Z","shell.execute_reply":"2022-08-10T14:52:53.462517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predictions = np.concatenate\n\nsubmission = pd.read_csv(\"../input/amex-default-prediction/sample_submission.csv\")\nsubmission.loc[:, \"prediction\"] = test_predictions\n\nsubmission.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T14:56:02.562223Z","iopub.execute_input":"2022-08-10T14:56:02.562673Z","iopub.status.idle":"2022-08-10T14:56:07.397882Z","shell.execute_reply.started":"2022-08-10T14:56:02.562635Z","shell.execute_reply":"2022-08-10T14:56:07.396843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}