{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom catboost import CatBoostClassifier\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.preprocessing import LabelEncoder\nimport gc; gc.enable()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-15T06:59:52.174708Z","iopub.execute_input":"2022-06-15T06:59:52.175615Z","iopub.status.idle":"2022-06-15T06:59:53.438011Z","shell.execute_reply.started":"2022-06-15T06:59:52.175537Z","shell.execute_reply":"2022-06-15T06:59:53.437086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load pickle file as dataset\n\n\n> Compressed data in Pickle Format for American Express - Default Prediction Competition\n\n> float64 categories converted to float16\n\n> int64 categories converted to int8\n\n> object categories converted to category\n\n> num cols agg stats -> 'mean', 'std', 'min', 'max', 'last'\n\n> cat cols agg stats -> 'count', 'last', 'nunique'","metadata":{}},{"cell_type":"code","source":"sub1 = pd.read_csv('../input/d/datasets/bhavikardeshna/avg-weights/submission.csv')\ntraini = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', parse_dates=['S_2'], chunksize=900_000, iterator=True)\ntesti = pd.read_csv('/kaggle/input/amex-default-prediction/test_data.csv', parse_dates=['S_2'], chunksize=500_000, iterator=True) \nlabels = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-13T14:40:46.054643Z","iopub.execute_input":"2022-06-13T14:40:46.057231Z","iopub.status.idle":"2022-06-13T14:40:49.490372Z","shell.execute_reply.started":"2022-06-13T14:40:46.057187Z","shell.execute_reply":"2022-06-13T14:40:49.488769Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain = []\nfor df in traini:\n    if len(train)>0: train = pd.concat([train, df])\n    else: train = df[:]\n    train.sort_values(by=['S_2'], inplace=True)\n    train.reset_index(drop=True, inplace=True)\n    train.drop_duplicates(subset=['customer_ID'], keep='last', inplace=True)\n    del df; gc.collect()\ntrain = pd.merge(train, labels, how='inner', on=['customer_ID'])\ndel labels; gc.collect()\ncol = [c for c in train if c not in ['customer_ID', 'target','S_2']]\ntrain.fillna(0).to_csv('train.csv', index=False)\ndel train; del traini; gc.collect()\n\ntest = []\nfor df in testi:\n    if len(test)>0: test = pd.concat([test, df])\n    else: test = df[:]\n    test.sort_values(by=['S_2'], inplace=True)\n    test.reset_index(drop=True, inplace=True)\n    test.drop_duplicates(subset=['customer_ID'], keep='last', inplace=True)\n    del df; gc.collect()\ntest.fillna(0).to_csv('test.csv', index=False)\ndel test; del testi; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-13T14:40:49.493313Z","iopub.execute_input":"2022-06-13T14:40:49.49439Z","iopub.status.idle":"2022-06-13T14:42:03.579206Z","shell.execute_reply.started":"2022-06-13T14:40:49.494345Z","shell.execute_reply":"2022-06-13T14:42:03.578286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## AMEX Metrics\n(From discusion)","metadata":{}},{"cell_type":"code","source":"def amex_metric(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n    def top_four_percent_captured(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        four_pct_cutoff = int(0.04 * df['weight'].sum())\n        df['weight_cumsum'] = df['weight'].cumsum()\n        df_cutoff = df.loc[df['weight_cumsum'] <= four_pct_cutoff]\n        return (df_cutoff['target'] == 1).sum() / (df['target'] == 1).sum()\n\n    def weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        df = (pd.concat([y_true, y_pred], axis='columns')\n              .sort_values('prediction', ascending=False))\n        df['weight'] = df['target'].apply(lambda x: 20 if x == 0 else 1)\n        df['random'] = (df['weight'] / df['weight'].sum()).cumsum()\n        total_pos = (df['target'] * df['weight']).sum()\n        df['cum_pos_found'] = (df['target'] * df['weight']).cumsum()\n        df['lorentz'] = df['cum_pos_found'] / total_pos\n        df['gini'] = (df['lorentz'] - df['random']) * df['weight']\n        return df['gini'].sum()\n\n    def normalized_weighted_gini(y_true: pd.DataFrame, y_pred: pd.DataFrame) -> float:\n        y_true_pred = y_true.rename(columns={'target': 'prediction'})\n        return weighted_gini(y_true, y_pred) / weighted_gini(y_true, y_true_pred)\n\n    g = normalized_weighted_gini(y_true, y_pred)\n    d = top_four_percent_captured(y_true, y_pred)\n\n    return 0.5 * (g + d)","metadata":{"execution":{"iopub.status.busy":"2022-06-13T14:42:03.581546Z","iopub.execute_input":"2022-06-13T14:42:03.582053Z","iopub.status.idle":"2022-06-13T14:42:03.59458Z","shell.execute_reply.started":"2022-06-13T14:42:03.582016Z","shell.execute_reply":"2022-06-13T14:42:03.593733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training CatBoostClassifier ","metadata":{}},{"cell_type":"code","source":"train = pd.read_csv('train.csv')\ncat_features = ['B_30', 'B_31', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nfor c in cat_features: train[c] = train[c].astype(str)","metadata":{"execution":{"iopub.status.busy":"2022-06-13T14:42:03.59583Z","iopub.execute_input":"2022-06-13T14:42:03.596266Z","iopub.status.idle":"2022-06-13T14:42:03.632435Z","shell.execute_reply.started":"2022-06-13T14:42:03.596228Z","shell.execute_reply":"2022-06-13T14:42:03.630873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x1, x2, y1, y2 = train_test_split(train[col], train.target, test_size=0.20, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-06-13T14:42:03.633373Z","iopub.status.idle":"2022-06-13T14:42:03.634391Z","shell.execute_reply.started":"2022-06-13T14:42:03.634152Z","shell.execute_reply":"2022-06-13T14:42:03.634177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = CatBoostClassifier(iterations=10000, random_state=42, nan_mode='Min')\nclf.fit(x1, y1, eval_set=[(x2, y2)], cat_features=cat_features,  verbose=50, early_stopping_rounds=20)\npreds = clf.predict_proba(x2)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-06-13T14:42:03.635647Z","iopub.status.idle":"2022-06-13T14:42:03.636309Z","shell.execute_reply.started":"2022-06-13T14:42:03.636077Z","shell.execute_reply":"2022-06-13T14:42:03.6361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_ = gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-13T14:42:03.637636Z","iopub.status.idle":"2022-06-13T14:42:03.638431Z","shell.execute_reply.started":"2022-06-13T14:42:03.638138Z","shell.execute_reply":"2022-06-13T14:42:03.638185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os","metadata":{"execution":{"iopub.status.busy":"2022-06-13T14:42:03.639645Z","iopub.status.idle":"2022-06-13T14:42:03.640289Z","shell.execute_reply.started":"2022-06-13T14:42:03.64006Z","shell.execute_reply":"2022-06-13T14:42:03.640083Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('test.csv')\nfor c in cat_features: test[c] = test[c].astype(str)\ntest['prediction'] = clf.predict_proba(test[col])[:, 1]\nsub2 = test[['customer_ID', 'prediction']]\ndel test;  gc.collect()\nos.remove ('test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-06-13T14:42:03.641509Z","iopub.status.idle":"2022-06-13T14:42:03.64217Z","shell.execute_reply.started":"2022-06-13T14:42:03.641933Z","shell.execute_reply":"2022-06-13T14:42:03.641967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub2.columns = ['customer_ID', 'prediction2']\nblend = pd.merge(sub1, sub2, how='inner', on='customer_ID')\nblend.prediction = (blend.prediction * 0.955 + blend.prediction2 * 0.045)\nblend[['customer_ID', 'prediction']].to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-13T14:42:03.643362Z","iopub.status.idle":"2022-06-13T14:42:03.644018Z","shell.execute_reply.started":"2022-06-13T14:42:03.64378Z","shell.execute_reply":"2022-06-13T14:42:03.643802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub2.columns = ['customer_ID','prediction2']\nblend = pd.merge(sub1, sub2, how='inner', on='customer_ID')\nblend.prediction = (blend.prediction * 0.955 + blend.prediction2 * 0.045)\nblend[['customer_ID','prediction']].to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-13T14:42:03.645232Z","iopub.status.idle":"2022-06-13T14:42:03.645903Z","shell.execute_reply.started":"2022-06-13T14:42:03.645672Z","shell.execute_reply":"2022-06-13T14:42:03.645695Z"},"trusted":true},"execution_count":null,"outputs":[]}]}