{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Cloned from\n- [First submission using CatBoost](https://www.kaggle.com/code/aninda/first-submission-using-catboost)\n- The original dataset: [https://www.kaggle.com/code/aninda/creating-smaller-train-test-data](https://www.kaggle.com/code/aninda/creating-smaller-train-test-data)\n\n\nReplace with the [AE Credit ID Encoded Dataset [FP16]](https://www.kaggle.com/competitions/amex-default-prediction/discussion/327228) dataset.\n\nThe whole purpose here is to cross-validate different dataset compression methods to see if we have done something wrong.","metadata":{}},{"cell_type":"markdown","source":"# Creating a basic submission using CatBoost Model","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-26T10:49:49.383231Z","iopub.execute_input":"2022-05-26T10:49:49.383850Z","iopub.status.idle":"2022-05-26T10:49:49.414861Z","shell.execute_reply.started":"2022-05-26T10:49:49.383743Z","shell.execute_reply":"2022-05-26T10:49:49.413818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_pickle(\"../input/ae-credit-id-encoded-dataset-fp16/id_encoded_fp16_train_data.pkl\")\nlabels = pd.read_pickle(\"../input/ae-credit-id-encoded-dataset-fp16/id_encoded_train_labels.pkl\")","metadata":{"execution":{"iopub.status.busy":"2022-05-26T10:49:49.418718Z","iopub.execute_input":"2022-05-26T10:49:49.419372Z","iopub.status.idle":"2022-05-26T10:50:08.265682Z","shell.execute_reply.started":"2022-05-26T10:49:49.419312Z","shell.execute_reply":"2022-05-26T10:50:08.264718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327094\ntrain =  (train\n            .groupby('customer_ID')\n            .tail(1)\n            .set_index('customer_ID', drop=True)\n            .sort_index()\n            .drop(['S_2'], axis='columns'))","metadata":{"execution":{"iopub.status.busy":"2022-05-26T10:50:08.267326Z","iopub.execute_input":"2022-05-26T10:50:08.268014Z","iopub.status.idle":"2022-05-26T10:50:17.269452Z","shell.execute_reply.started":"2022-05-26T10:50:08.267977Z","shell.execute_reply":"2022-05-26T10:50:17.267968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.merge(train, labels, how=\"left\", on=\"customer_ID\")","metadata":{"execution":{"iopub.status.busy":"2022-05-26T10:50:17.271437Z","iopub.execute_input":"2022-05-26T10:50:17.271829Z","iopub.status.idle":"2022-05-26T10:50:19.195989Z","shell.execute_reply.started":"2022-05-26T10:50:17.271792Z","shell.execute_reply":"2022-05-26T10:50:19.194930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-26T10:50:19.198452Z","iopub.execute_input":"2022-05-26T10:50:19.199010Z","iopub.status.idle":"2022-05-26T10:50:19.208528Z","shell.execute_reply.started":"2022-05-26T10:50:19.198972Z","shell.execute_reply":"2022-05-26T10:50:19.207498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_cols = train.columns.to_list()\n\ncat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\nnum_cols = [col for col in all_cols if col not in cat_cols + [\"target\", \"customer_ID\"]]","metadata":{"execution":{"iopub.status.busy":"2022-05-26T10:50:19.209785Z","iopub.execute_input":"2022-05-26T10:50:19.210167Z","iopub.status.idle":"2022-05-26T10:50:19.220601Z","shell.execute_reply.started":"2022-05-26T10:50:19.210135Z","shell.execute_reply":"2022-05-26T10:50:19.219516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_cols","metadata":{"execution":{"iopub.status.busy":"2022-05-26T10:50:19.223497Z","iopub.execute_input":"2022-05-26T10:50:19.224097Z","iopub.status.idle":"2022-05-26T10:50:19.237802Z","shell.execute_reply.started":"2022-05-26T10:50:19.224062Z","shell.execute_reply":"2022-05-26T10:50:19.236965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X = train[cat_cols + num_cols]\ntrain_y = pd.DataFrame(train[\"target\"])","metadata":{"execution":{"iopub.status.busy":"2022-05-26T10:50:19.239879Z","iopub.execute_input":"2022-05-26T10:50:19.240372Z","iopub.status.idle":"2022-05-26T10:50:19.437138Z","shell.execute_reply.started":"2022-05-26T10:50:19.240294Z","shell.execute_reply":"2022-05-26T10:50:19.436086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in cat_cols:\n    train_X[col] = train_X[col].astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T10:50:19.440765Z","iopub.execute_input":"2022-05-26T10:50:19.441094Z","iopub.status.idle":"2022-05-26T10:50:21.625178Z","shell.execute_reply.started":"2022-05-26T10:50:19.441065Z","shell.execute_reply":"2022-05-26T10:50:21.624106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Competition metric**","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T10:50:21.626833Z","iopub.execute_input":"2022-05-26T10:50:21.627378Z","iopub.status.idle":"2022-05-26T10:50:21.642772Z","shell.execute_reply.started":"2022-05-26T10:50:21.627319Z","shell.execute_reply":"2022-05-26T10:50:21.642004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"http://Testing the competition metric","metadata":{}},{"cell_type":"code","source":"y_pred = train_y.copy(deep=True).rename(columns={'P_2': 'prediction'}).drop(\"target\",axis=1)\ny_pred[\"prediction\"] = 0\ny_pred.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T10:50:21.646026Z","iopub.execute_input":"2022-05-26T10:50:21.647086Z","iopub.status.idle":"2022-05-26T10:50:21.680544Z","shell.execute_reply.started":"2022-05-26T10:50:21.647044Z","shell.execute_reply":"2022-05-26T10:50:21.679432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric(train_y,y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T10:50:21.681761Z","iopub.execute_input":"2022-05-26T10:50:21.682082Z","iopub.status.idle":"2022-05-26T10:50:22.248657Z","shell.execute_reply.started":"2022-05-26T10:50:21.682052Z","shell.execute_reply":"2022-05-26T10:50:22.247567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Using CatBoostClassifier to make a basic submission**","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\nfrom catboost import CatBoostClassifier\nfrom sklearn.model_selection import train_test_split\n\nX_train,X_test,y_train,y_test = train_test_split(train_X,train_y,stratify=train_y)\nclf = CatBoostClassifier(iterations=1000)\n\nclf.fit(X_train,y_train,eval_set=[(X_test,y_test)],cat_features=cat_cols,verbose=100)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T10:50:22.250621Z","iopub.execute_input":"2022-05-26T10:50:22.251435Z","iopub.status.idle":"2022-05-26T11:00:03.384088Z","shell.execute_reply.started":"2022-05-26T10:50:22.251389Z","shell.execute_reply":"2022-05-26T11:00:03.382441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = y_test.copy(deep=True)\ny_pred = y_pred.rename(columns={\"target\":\"prediction\"})\ny_pred[\"prediction\"] = clf.predict_proba(X_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-05-26T11:00:03.385937Z","iopub.execute_input":"2022-05-26T11:00:03.386454Z","iopub.status.idle":"2022-05-26T11:00:08.917402Z","shell.execute_reply.started":"2022-05-26T11:00:03.386406Z","shell.execute_reply":"2022-05-26T11:00:08.916528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric(y_test,y_pred) # Metric calculation on validation set","metadata":{"execution":{"iopub.status.busy":"2022-05-26T11:00:08.918735Z","iopub.execute_input":"2022-05-26T11:00:08.919066Z","iopub.status.idle":"2022-05-26T11:00:09.100390Z","shell.execute_reply.started":"2022-05-26T11:00:08.919035Z","shell.execute_reply":"2022-05-26T11:00:09.099414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making prediction on Competition Test data","metadata":{}},{"cell_type":"code","source":"del train, train_X, train_y, X_train, X_test, y_train, y_test\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T11:00:09.102183Z","iopub.execute_input":"2022-05-26T11:00:09.102823Z","iopub.status.idle":"2022-05-26T11:00:09.626293Z","shell.execute_reply.started":"2022-05-26T11:00:09.102775Z","shell.execute_reply":"2022-05-26T11:00:09.625402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_pickle(\"../input/ae-credit-id-encoded-dataset-fp16/id_encoded_fp16_test_data.pkl\")","metadata":{"execution":{"iopub.status.busy":"2022-05-26T11:00:09.627468Z","iopub.execute_input":"2022-05-26T11:00:09.627988Z","iopub.status.idle":"2022-05-26T11:00:50.675850Z","shell.execute_reply.started":"2022-05-26T11:00:09.627952Z","shell.execute_reply":"2022-05-26T11:00:50.674861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test =  (test\n            .groupby('customer_ID')\n            .tail(1)\n            .set_index('customer_ID', drop=True)\n            .sort_index()\n            .drop(['S_2'], axis='columns'))","metadata":{"execution":{"iopub.status.busy":"2022-05-26T11:00:50.677311Z","iopub.execute_input":"2022-05-26T11:00:50.678099Z","iopub.status.idle":"2022-05-26T11:01:04.056435Z","shell.execute_reply.started":"2022-05-26T11:00:50.678061Z","shell.execute_reply":"2022-05-26T11:01:04.055537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in cat_cols:\n    test[col] = test[col].astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T11:01:04.058735Z","iopub.execute_input":"2022-05-26T11:01:04.059210Z","iopub.status.idle":"2022-05-26T11:01:07.858862Z","shell.execute_reply.started":"2022-05-26T11:01:04.059164Z","shell.execute_reply":"2022-05-26T11:01:07.857880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"prediction\"] = clf.predict_proba(test[cat_cols + num_cols])[:,1]\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T11:01:07.860163Z","iopub.execute_input":"2022-05-26T11:01:07.860972Z","iopub.status.idle":"2022-05-26T11:01:50.780161Z","shell.execute_reply.started":"2022-05-26T11:01:07.860935Z","shell.execute_reply":"2022-05-26T11:01:50.779059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nloaded_encoder = LabelEncoder()\nloaded_encoder.classes_ = np.load(f\"../input/ae-credit-id-encoded-dataset-fp16/id_encodings.npy\", allow_pickle=True)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T11:01:50.781943Z","iopub.execute_input":"2022-05-26T11:01:50.782404Z","iopub.status.idle":"2022-05-26T11:01:53.334398Z","shell.execute_reply.started":"2022-05-26T11:01:50.782358Z","shell.execute_reply":"2022-05-26T11:01:53.333441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test.reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T11:01:53.335692Z","iopub.execute_input":"2022-05-26T11:01:53.336395Z","iopub.status.idle":"2022-05-26T11:01:53.502674Z","shell.execute_reply.started":"2022-05-26T11:01:53.336313Z","shell.execute_reply":"2022-05-26T11:01:53.501630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"customer_ID\"] = loaded_encoder.inverse_transform(test[\"customer_ID\"])\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T11:01:53.504201Z","iopub.execute_input":"2022-05-26T11:01:53.504979Z","iopub.status.idle":"2022-05-26T11:01:53.655323Z","shell.execute_reply.started":"2022-05-26T11:01:53.504934Z","shell.execute_reply":"2022-05-26T11:01:53.654409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[[\"customer_ID\", \"prediction\"]].to_csv(\"submission_first.csv\",index=False) #Creating submission file","metadata":{"execution":{"iopub.status.busy":"2022-05-26T11:04:19.866899Z","iopub.execute_input":"2022-05-26T11:04:19.867458Z","iopub.status.idle":"2022-05-26T11:04:23.703052Z","shell.execute_reply.started":"2022-05-26T11:04:19.867407Z","shell.execute_reply":"2022-05-26T11:04:23.702204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}