{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nimport glob\nimport numpy as np\nimport pandas as pd\nimport xgboost as xgb\nSEED = 42","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-08T17:20:19.601395Z","iopub.execute_input":"2022-06-08T17:20:19.601802Z","iopub.status.idle":"2022-06-08T17:20:20.337085Z","shell.execute_reply.started":"2022-06-08T17:20:19.601724Z","shell.execute_reply":"2022-06-08T17:20:20.336312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv('../input/amex-default-prediction/train_labels.csv')\ntrain_labels['customer_ID'] = train_labels['customer_ID'].apply(lambda x: int(x[-16:], 16)).astype(np.int64)\ntrain_labels = train_labels.set_axis(train_labels['customer_ID'])\ntrain_labels = train_labels.drop(['customer_ID'], axis=1)\n\ntrain_pkls = sorted(glob.glob('../input/amex-processed-dataset/train_data_*'))\ntest_pkls = sorted(glob.glob('../input/amex-processed-dataset/test_data_*'))\n\ntrain_df = pd.read_pickle(train_pkls[0]).astype(np.float32)\nprint(train_pkls[0])\nfor i in train_pkls[1:]:\n    print(i)\n    train_df = train_df.append(pd.read_pickle(i))\n    train_df = train_df.astype(np.float32)\n    gc.collect()\n    \ny = train_labels.loc[train_df.index.values].values.astype(np.int8)\ntrain_df = train_df.drop(['D_64_-1', 'D_66_0.0', 'D_68_0.0'], axis=1).astype(np.float32)\nprint(train_df.shape, y.shape)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T17:20:20.338691Z","iopub.execute_input":"2022-06-08T17:20:20.339051Z","iopub.status.idle":"2022-06-08T17:20:51.998858Z","shell.execute_reply.started":"2022-06-08T17:20:20.339015Z","shell.execute_reply":"2022-06-08T17:20:51.997806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T17:20:52.000747Z","iopub.execute_input":"2022-06-08T17:20:52.001118Z","iopub.status.idle":"2022-06-08T17:20:52.006243Z","shell.execute_reply.started":"2022-06-08T17:20:52.00108Z","shell.execute_reply":"2022-06-08T17:20:52.004495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def amex_metric(y_true, y_pred):\n    labels     = np.transpose(np.array([y_true, y_pred]))\n    labels     = labels[labels[:, 1].argsort()[::-1]]\n    weights    = np.where(labels[:,0]==0, 20, 1)\n    cut_vals   = labels[np.cumsum(weights) <= int(0.04 * np.sum(weights))]\n    top_four   = np.sum(cut_vals[:,0]) / np.sum(labels[:,0])\n\n    gini = [0,0]\n    for i in [1,0]:\n        labels         = np.transpose(np.array([y_true, y_pred]))\n        labels         = labels[labels[:, i].argsort()[::-1]]\n        weight         = np.where(labels[:,0]==0, 20, 1)\n        weight_random  = np.cumsum(weight / np.sum(weight))\n        total_pos      = np.sum(labels[:, 0] *  weight)\n        cum_pos_found  = np.cumsum(labels[:, 0] * weight)\n        lorentz        = cum_pos_found / total_pos\n        gini[i]        = np.sum((lorentz - weight_random) * weight)\n\n    return 0.5 * (gini[1]/gini[0] + top_four)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T17:20:52.00765Z","iopub.execute_input":"2022-06-08T17:20:52.007991Z","iopub.status.idle":"2022-06-08T17:20:52.087695Z","shell.execute_reply.started":"2022-06-08T17:20:52.007956Z","shell.execute_reply":"2022-06-08T17:20:52.086371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_val, y_train, y_val = train_test_split(train_df, y,\n                                                    stratify=y, \n                                                    test_size=0.25)\n\ndel train_df, y\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T17:20:52.092502Z","iopub.execute_input":"2022-06-08T17:20:52.093054Z","iopub.status.idle":"2022-06-08T17:20:58.339417Z","shell.execute_reply.started":"2022-06-08T17:20:52.093016Z","shell.execute_reply":"2022-06-08T17:20:58.338534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = xgb.XGBClassifier(\n        n_estimators = 5000,\n        max_depth = 3,\n        learning_rate = 0.05, \n        subsample = 1,\n        colsample_bytree = 0.2, \n        tree_method ='gpu_hist',\n        predictor = 'gpu_predictor',\n        eval_metric = amex_metric,\n        random_state = SEED\n    )","metadata":{"execution":{"iopub.status.busy":"2022-06-08T17:21:02.529927Z","iopub.execute_input":"2022-06-08T17:21:02.530277Z","iopub.status.idle":"2022-06-08T17:21:02.535354Z","shell.execute_reply.started":"2022-06-08T17:21:02.530248Z","shell.execute_reply":"2022-06-08T17:21:02.534278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(X_train, y_train,eval_set=[(X_train, y_train), (X_val, y_val)],verbose=50)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T17:21:03.118154Z","iopub.execute_input":"2022-06-08T17:21:03.118874Z","iopub.status.idle":"2022-06-08T17:45:47.006281Z","shell.execute_reply.started":"2022-06-08T17:21:03.118836Z","shell.execute_reply":"2022-06-08T17:45:47.005423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/amex-default-prediction/sample_submission.csv')\nsubmission['customer_ID_encoded'] = train_labels['customer_ID'] = submission['customer_ID'].apply(lambda x: int(x[-16:], 16)).astype(np.int64)\nsubmission.set_axis(submission['customer_ID_encoded'], inplace=True)\nsubmission = submission.drop(['customer_ID_encoded'], axis=1)\nsubmission['prediction'] = submission['prediction'].astype(np.float32)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T17:47:56.515342Z","iopub.execute_input":"2022-06-08T17:47:56.515716Z","iopub.status.idle":"2022-06-08T17:47:58.936107Z","shell.execute_reply.started":"2022-06-08T17:47:56.515686Z","shell.execute_reply":"2022-06-08T17:47:58.935279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ncustomer_ids_list = []\npreds_list = []\nfor t in test_pkls:\n    test_df = pd.read_pickle(t)\n    customer_ids = test_df.axes[0].values\n    customer_ids = submission.loc[customer_ids]['customer_ID'].values\n    customer_ids_list.extend(customer_ids)\n    preds = model.predict_proba(test_df)[:, 1]\n    preds_list.extend(preds)\n    gc.collect()\n\npreds_list = np.array(preds_list).reshape(-1, 1)\ncustomer_ids_list = np.array(customer_ids_list).reshape(-1, 1)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T17:47:58.937747Z","iopub.execute_input":"2022-06-08T17:47:58.938101Z","iopub.status.idle":"2022-06-08T17:49:43.50085Z","shell.execute_reply.started":"2022-06-08T17:47:58.938067Z","shell.execute_reply":"2022-06-08T17:49:43.500028Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame(data=np.concatenate([customer_ids_list, preds_list], axis=1), columns=['customer_ID', 'prediction'])\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T17:50:07.301055Z","iopub.execute_input":"2022-06-08T17:50:07.301392Z","iopub.status.idle":"2022-06-08T17:50:11.96836Z","shell.execute_reply.started":"2022-06-08T17:50:07.301363Z","shell.execute_reply":"2022-06-08T17:50:11.967473Z"},"trusted":true},"execution_count":null,"outputs":[]}]}