{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"}],"dockerImageVersionId":30197,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-09T04:04:10.514864Z","iopub.execute_input":"2022-07-09T04:04:10.515738Z","iopub.status.idle":"2022-07-09T04:04:10.544777Z","shell.execute_reply.started":"2022-07-09T04:04:10.515609Z","shell.execute_reply":"2022-07-09T04:04:10.544032Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from catboost import CatBoostClassifier\nimport gc; gc.enable()\nfrom sklearn import *\nimport pandas as pd\nimport numpy as np\nimport os\ntraini = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', parse_dates=['S_2'], chunksize=400_000, iterator=True)\ntesti = pd.read_csv('/kaggle/input/amex-default-prediction/test_data.csv', parse_dates=['S_2'], chunksize=400_000, iterator=True) \nlabels = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T04:04:10.546323Z","iopub.execute_input":"2022-07-09T04:04:10.546664Z","iopub.status.idle":"2022-07-09T04:04:14.506287Z","shell.execute_reply.started":"2022-07-09T04:04:10.546632Z","shell.execute_reply":"2022-07-09T04:04:14.505088Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = []\nfor df in traini:\n    if len(train)>0: train = pd.concat([train, df])\n    else: train = df[:]\n    train.sort_values(by=['S_2'], inplace=True)\n    train.reset_index(drop=True, inplace=True)\n    train.drop_duplicates(subset=['customer_ID'], keep='last', inplace=True)\n    del df; gc.collect()\n\n\ntrain = pd.merge(train, labels, 'inner', on=['customer_ID'])\ncol = [c for c in train if c not in ['customer_ID', 'target','S_2']]\ntrain.fillna(0).to_csv('train.csv', index=False)\n\ntest = []\nfor df in testi:\n    if len(test)>0: test = pd.concat([test, df])\n    else: test = df[:]\n    test.sort_values(by=['S_2'], inplace=True)\n    test.reset_index(drop=True, inplace=True)\n    test.drop_duplicates(subset=['customer_ID'], keep='last', inplace=True)\n\ntest.fillna(0).to_csv('test.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T04:04:14.508457Z","iopub.execute_input":"2022-07-09T04:04:14.508944Z","iopub.status.idle":"2022-07-09T04:35:33.37631Z","shell.execute_reply.started":"2022-07-09T04:04:14.508882Z","shell.execute_reply":"2022-07-09T04:35:33.374994Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Inspired by TowardsDataScience\nclass Metric(object):\n    def get_final_error(self, error, weight):\n        return error\n    def is_max_optimal(self):\n        return True\n    def evaluate(self, y_pred, y_true, weight):\n        y_pred = y_pred[0]\n        indices = np.argsort(y_pred)[::-1]\n        preds, target = y_pred[indices], y_true[indices]\n        weight = 20.0 - target * 19.0\n        cum_norm_weight = (weight / weight.sum()).cumsum()\n        four_pct_mask = cum_norm_weight <= 0.04\n        d = np.sum(target[four_pct_mask]) / np.sum(target)\n        weighted_target = target * weight\n        lorentz = (weighted_target / weighted_target.sum()).cumsum()\n        gini = ((lorentz - cum_norm_weight) * weight).sum()\n        n_pos = np.sum(target)\n        n_neg = target.shape[0] - n_pos\n        gini_max = 10 * n_neg * (n_pos + 20 * n_neg - 19) / (n_pos + 20 * n_neg)\n        g = gini / gini_max\n        return 0.5 * (g + d), 0","metadata":{"execution":{"iopub.status.busy":"2022-07-09T04:35:33.378716Z","iopub.execute_input":"2022-07-09T04:35:33.379607Z","iopub.status.idle":"2022-07-09T04:35:33.393322Z","shell.execute_reply.started":"2022-07-09T04:35:33.379554Z","shell.execute_reply":"2022-07-09T04:35:33.392392Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_csv('train.csv')\ncat_features = ['B_30', 'B_31', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\nfor c in cat_features: train[c] = train[c].astype(str)\nx1, x2, y1, y2 = model_selection.train_test_split(train[col], train.target, test_size=0.20, random_state=22)\n\nmodel = CatBoostClassifier(iterations=5000, random_state=101, nan_mode='Min', eval_metric=Metric())\nmodel.fit(x1, y1, eval_set=[(x2, y2)], cat_features=cat_features,  verbose=50)\npreds = model.predict_proba(x2)[:, 1]","metadata":{"execution":{"iopub.status.busy":"2022-07-09T04:51:43.643624Z","iopub.execute_input":"2022-07-09T04:51:43.644304Z","iopub.status.idle":"2022-07-09T04:51:43.691097Z","shell.execute_reply.started":"2022-07-09T04:51:43.644249Z","shell.execute_reply":"2022-07-09T04:51:43.689278Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test = pd.read_csv('test.csv')\nfor c in cat_features: test[c] = test[c].astype(str)\ntest['prediction'] = model.predict_proba(test[col])[:, 1]\nsub2 = test[['customer_ID', 'prediction']]\nsub2.columns = ['customer_ID', 'prediction']\nsub2.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T04:47:02.594452Z","iopub.status.idle":"2022-07-09T04:47:02.594991Z","shell.execute_reply.started":"2022-07-09T04:47:02.594789Z","shell.execute_reply":"2022-07-09T04:47:02.594807Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-07-09T04:47:02.595954Z","iopub.status.idle":"2022-07-09T04:47:02.596447Z","shell.execute_reply.started":"2022-07-09T04:47:02.596278Z","shell.execute_reply":"2022-07-09T04:47:02.596296Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}