{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Creating a basic submission using CatBoost Model","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-26T09:04:08.043961Z","iopub.execute_input":"2022-05-26T09:04:08.044825Z","iopub.status.idle":"2022-05-26T09:04:08.04907Z","shell.execute_reply.started":"2022-05-26T09:04:08.044785Z","shell.execute_reply":"2022-05-26T09:04:08.048378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_pickle(\"../input/creating-smaller-train-test-data/amex_train_data.pkl\")","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:05:45.173012Z","iopub.execute_input":"2022-05-26T09:05:45.17343Z","iopub.status.idle":"2022-05-26T09:06:01.523925Z","shell.execute_reply.started":"2022-05-26T09:05:45.173398Z","shell.execute_reply":"2022-05-26T09:06:01.522786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.kaggle.com/competitions/amex-default-prediction/discussion/327094\n\ntrain =  (train\n            .groupby('customer_ID')\n            .tail(1)\n            .set_index('customer_ID', drop=True)\n            .sort_index()\n            .drop(['S_2'], axis='columns'))","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:07:35.489695Z","iopub.execute_input":"2022-05-26T09:07:35.490115Z","iopub.status.idle":"2022-05-26T09:07:44.120377Z","shell.execute_reply.started":"2022-05-26T09:07:35.490075Z","shell.execute_reply":"2022-05-26T09:07:44.11937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:07:51.825713Z","iopub.execute_input":"2022-05-26T09:07:51.82609Z","iopub.status.idle":"2022-05-26T09:07:51.832413Z","shell.execute_reply.started":"2022-05-26T09:07:51.826061Z","shell.execute_reply":"2022-05-26T09:07:51.83166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_cols = train.columns.to_list()\n\ncat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\nnum_cols = [col for col in all_cols if col not in cat_cols + [\"target\"]]","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:08:09.563353Z","iopub.execute_input":"2022-05-26T09:08:09.563958Z","iopub.status.idle":"2022-05-26T09:08:09.569232Z","shell.execute_reply.started":"2022-05-26T09:08:09.563922Z","shell.execute_reply":"2022-05-26T09:08:09.568489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_X = train[cat_cols + num_cols]\ntrain_y = pd.DataFrame(train[\"target\"])","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:12:23.15144Z","iopub.execute_input":"2022-05-26T09:12:23.151859Z","iopub.status.idle":"2022-05-26T09:12:23.160077Z","shell.execute_reply.started":"2022-05-26T09:12:23.151823Z","shell.execute_reply":"2022-05-26T09:12:23.158817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in cat_cols:\n    train_X[col] = train_X[col].astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:09:45.376196Z","iopub.execute_input":"2022-05-26T09:09:45.377176Z","iopub.status.idle":"2022-05-26T09:09:48.125546Z","shell.execute_reply.started":"2022-05-26T09:09:45.37714Z","shell.execute_reply":"2022-05-26T09:09:48.124555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Competition metric**","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n        \n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x==0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:12:31.007395Z","iopub.execute_input":"2022-05-26T09:12:31.007932Z","iopub.status.idle":"2022-05-26T09:12:31.021905Z","shell.execute_reply.started":"2022-05-26T09:12:31.007901Z","shell.execute_reply":"2022-05-26T09:12:31.021045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Testing the competition metric","metadata":{}},{"cell_type":"code","source":"y_pred = train_y.copy(deep=True).rename(columns={'P_2': 'prediction'}).drop(\"target\",axis=1)\ny_pred[\"prediction\"] = 0\ny_pred.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:12:35.395995Z","iopub.execute_input":"2022-05-26T09:12:35.396795Z","iopub.status.idle":"2022-05-26T09:12:35.416026Z","shell.execute_reply.started":"2022-05-26T09:12:35.396742Z","shell.execute_reply":"2022-05-26T09:12:35.414753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric(train_y,y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:12:38.64805Z","iopub.execute_input":"2022-05-26T09:12:38.64864Z","iopub.status.idle":"2022-05-26T09:12:39.899874Z","shell.execute_reply.started":"2022-05-26T09:12:38.648599Z","shell.execute_reply":"2022-05-26T09:12:39.898429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Using CatBoostClassifier to make a basic submission**","metadata":{}},{"cell_type":"code","source":"import lightgbm as lgb\nfrom catboost import CatBoostClassifier\nfrom sklearn.model_selection import train_test_split\n\nX_train,X_test,y_train,y_test = train_test_split(train_X,train_y,stratify=train_y)\nclf = CatBoostClassifier()\n\nclf.fit(X_train,y_train,eval_set=[(X_test,y_test)],cat_features=cat_cols,verbose=100)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:12:49.133464Z","iopub.execute_input":"2022-05-26T09:12:49.133862Z","iopub.status.idle":"2022-05-26T09:23:44.585843Z","shell.execute_reply.started":"2022-05-26T09:12:49.13383Z","shell.execute_reply":"2022-05-26T09:23:44.584707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = y_test.copy(deep=True)\ny_pred = y_pred.rename(columns={\"target\":\"prediction\"})\ny_pred[\"prediction\"] = clf.predict_proba(X_test)[:,1]","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:27:15.500845Z","iopub.execute_input":"2022-05-26T09:27:15.501321Z","iopub.status.idle":"2022-05-26T09:27:24.681674Z","shell.execute_reply.started":"2022-05-26T09:27:15.501272Z","shell.execute_reply":"2022-05-26T09:27:24.680579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"amex_metric(y_test,y_pred) # Metric calculation on validation set","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:27:43.015345Z","iopub.execute_input":"2022-05-26T09:27:43.015747Z","iopub.status.idle":"2022-05-26T09:27:43.486389Z","shell.execute_reply.started":"2022-05-26T09:27:43.015718Z","shell.execute_reply":"2022-05-26T09:27:43.485631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Making prediction on Competition Test data","metadata":{}},{"cell_type":"code","source":"test = pd.read_pickle(\"../input/creating-smaller-train-test-data/amex_test_data.pkl\")","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:27:56.8715Z","iopub.execute_input":"2022-05-26T09:27:56.872096Z","iopub.status.idle":"2022-05-26T09:28:42.008716Z","shell.execute_reply.started":"2022-05-26T09:27:56.872063Z","shell.execute_reply":"2022-05-26T09:28:42.007592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test =  (test\n            .groupby('customer_ID')\n            .tail(1)\n            .set_index('customer_ID', drop=True)\n            .sort_index()\n            .drop(['S_2'], axis='columns'))","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:29:22.511726Z","iopub.execute_input":"2022-05-26T09:29:22.512126Z","iopub.status.idle":"2022-05-26T09:29:45.934642Z","shell.execute_reply.started":"2022-05-26T09:29:22.512096Z","shell.execute_reply":"2022-05-26T09:29:45.933616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for col in cat_cols:\n    test[col] = test[col].astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:30:56.665103Z","iopub.execute_input":"2022-05-26T09:30:56.666048Z","iopub.status.idle":"2022-05-26T09:30:56.973Z","shell.execute_reply.started":"2022-05-26T09:30:56.666004Z","shell.execute_reply":"2022-05-26T09:30:56.971865Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"prediction\"] = clf.predict_proba(test[cat_cols + num_cols])[:,1]\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:35:11.336348Z","iopub.execute_input":"2022-05-26T09:35:11.337254Z","iopub.status.idle":"2022-05-26T09:36:29.505963Z","shell.execute_reply.started":"2022-05-26T09:35:11.337082Z","shell.execute_reply":"2022-05-26T09:36:29.504886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test[\"prediction\"].to_csv(\"submission_first.csv\",index=True) #Creating submission file","metadata":{"execution":{"iopub.status.busy":"2022-05-26T09:37:27.896842Z","iopub.execute_input":"2022-05-26T09:37:27.897297Z","iopub.status.idle":"2022-05-26T09:37:32.681818Z","shell.execute_reply.started":"2022-05-26T09:37:27.897249Z","shell.execute_reply":"2022-05-26T09:37:32.680704Z"},"trusted":true},"execution_count":null,"outputs":[]}]}