{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-09T21:13:54.148501Z","iopub.execute_input":"2022-08-09T21:13:54.148985Z","iopub.status.idle":"2022-08-09T21:13:54.187700Z","shell.execute_reply.started":"2022-08-09T21:13:54.148886Z","shell.execute_reply":"2022-08-09T21:13:54.186647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from catboost import CatBoostClassifier, Pool, cv, to_classifier\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-08-09T21:13:54.189630Z","iopub.execute_input":"2022-08-09T21:13:54.190367Z","iopub.status.idle":"2022-08-09T21:13:56.258217Z","shell.execute_reply.started":"2022-08-09T21:13:54.190298Z","shell.execute_reply":"2022-08-09T21:13:56.256875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_size = 1_500_000\nbatch_size = 100_000\n\ndef add_dt_columns(df):\n    \n    datetime_ = pd.to_datetime(df['S_2']).dt\n    \n    df['date'] = datetime_.year.values * 10_000 + datetime_.month.values * 100 + datetime_.day.values\n    df['month'] = datetime_.month.values\n    df['year'] = datetime_.year.values\n    df['day'] = datetime_.day.values\n    \n    df['date_diff'] = df.groupby('customer_ID')['date'].transform(np.max) - df.groupby('customer_ID')['date'].transform(np.min)\n\n    df.drop('S_2', axis=1, inplace=True)\n\ndef preprocess(df, labels = {}, csv_name='preprocessed.csv', isFirst=True):\n    \n    add_dt_columns(df)\n    df['count'] = df.groupby('customer_ID')['date'].transform('count')\n    \n    df = df.drop([k for k, v in df.dtypes.to_dict().items() if v == 'object'], axis = 1).round(4).astype(np.float32)\n\n    if len(labels) > 0:\n        df['label'] = df.reset_index().customer_ID.map(labels).fillna(0).values\n        df['label'] = df['label'].astype(int)\n        df = df.reset_index().drop('customer_ID', axis = 1)\n\n    #     df = df.reset_index().groupby('customer_ID').agg([np.min, np.max, np.std, np.mean])\n    \n        df = df[['label'] + [c for c in df.columns if c != 'label']]\n    else:\n        return df\n    \n    if isFirst:\n        df.to_csv(csv_name, header=False, index=False)\n    else:\n        df.to_csv('preprocessed.csv', mode='a', header=False, index=False)\n\ndef load_preprocess():\n    train = pd.read_csv(\n        \"/kaggle/input/amex-default-prediction/train_data.csv\", \n        chunksize=batch_size,\n#         nrows=10_000,\n        index_col='customer_ID'\n    )\n    train_labels = pd.read_csv(\n        \"/kaggle/input/amex-default-prediction/train_labels.csv\", \n        index_col='customer_ID'\n    )\n    \n    train_labels = {k:v for k, v in train_labels['target'].to_dict().items() if v > 0}\n    \n    result = []\n    \n    isFirst = True\n    for train_ in tqdm(train, total = int(5.53 * 1e6) // batch_size + 1):\n        preprocess(train_, train_labels, isFirst=isFirst)\n        isFirst=False\n        \n#         if len(result) < 1:\n#             result = df\n#         else:\n#             result = pd.concat([result, df], copy=False)\n    \n# #     result = pd.concat(train_full_list)\n    \n#     print(f\"input size: {result.shape}\")\n    \n#     return result","metadata":{"execution":{"iopub.status.busy":"2022-08-09T21:13:56.260063Z","iopub.execute_input":"2022-08-09T21:13:56.260515Z","iopub.status.idle":"2022-08-09T21:13:56.281441Z","shell.execute_reply.started":"2022-08-09T21:13:56.260480Z","shell.execute_reply":"2022-08-09T21:13:56.280070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nclass_weights = {0.0: 20.0, 1.0: 1.0}\n\n\ndef get_train_val():\n    \n#     X = train_full.values[:,:-1]\n#     y = train_full.values[:,-1]\n\n#     val_size = len(train_full) // 3\n\n#     X_train = train_full.values[:-val_size,:-1]\n#     y_train = train_full.values[:-val_size,-1]\n\n#     X_val = train_full.values[-val_size:,:-1]\n#     y_val = train_full.values[-val_size:,-1]\n\n#     train_pool = Pool(data=X_train, label=y_train)\n#     val_pool = Pool(data=X_val, label=y_val)\n\n    pool = Pool('preprocessed.csv', delimiter=',')\n    \n    return pool\n\n","metadata":{"execution":{"iopub.status.busy":"2022-08-09T21:13:56.284322Z","iopub.execute_input":"2022-08-09T21:13:56.285253Z","iopub.status.idle":"2022-08-09T21:13:56.303623Z","shell.execute_reply.started":"2022-08-09T21:13:56.285201Z","shell.execute_reply":"2022-08-09T21:13:56.301402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_full = load_preprocess()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-09T21:13:56.305675Z","iopub.execute_input":"2022-08-09T21:13:56.306198Z","iopub.status.idle":"2022-08-09T21:13:56.314576Z","shell.execute_reply.started":"2022-08-09T21:13:56.306148Z","shell.execute_reply":"2022-08-09T21:13:56.312751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"params = {\"iterations\": 4_000,\n          \"depth\": 8,\n          \"loss_function\": \"Logloss\",\n          \"verbose\": 100,\n          \"learning_rate\": 0.1,\n          \"scale_pos_weight\": 10.0,\n#           \"task_type\": \"GPU\"\n         }\n\n\n\n\ndef produce_cat():\n    \n    load_preprocess()\n    print(\"load done\")\n    cv_dataset = get_train_val()\n    print(\"split done\")\n    \n    \n    scores, models = cv(cv_dataset,\n            params,\n            fold_count=3, \n            return_models=True\n    )\n\n#     cat.fit(train_pool, eval_set=val_pool)\n    \n    return scores, [to_classifier(m) for m in models]\n\nscores, models = produce_cat()","metadata":{"execution":{"iopub.status.busy":"2022-08-09T21:17:29.662402Z","iopub.execute_input":"2022-08-09T21:17:29.662891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"result = []\n\n\nn_iter = 11363761 // batch_size + 1\ntest_iterator = pd.read_csv(\"/kaggle/input/amex-default-prediction/test_data.csv\", chunksize=batch_size, index_col='customer_ID')\n\nfor test in tqdm(test_iterator, total=n_iter):\n    \n    test_full = preprocess(test)\n    test_full['prediction'] = 0\n    \n    for cat in models:\n        test_full['prediction'] += cat.predict_proba(test_full.values)[:,1]\n    \n    result.append(test_full['prediction'] / len(models))\n","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:08:11.688341Z","iopub.execute_input":"2022-08-08T13:08:11.688760Z","iopub.status.idle":"2022-08-08T13:08:17.547095Z","shell.execute_reply.started":"2022-08-08T13:08:11.688721Z","shell.execute_reply":"2022-08-08T13:08:17.545875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.DataFrame(pd.concat(result, copy=False)).reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:08:30.528413Z","iopub.execute_input":"2022-08-08T13:08:30.528857Z","iopub.status.idle":"2022-08-08T13:08:30.539424Z","shell.execute_reply.started":"2022-08-08T13:08:30.528820Z","shell.execute_reply":"2022-08-08T13:08:30.538020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:08:39.314853Z","iopub.execute_input":"2022-08-08T13:08:39.315953Z","iopub.status.idle":"2022-08-08T13:08:39.340690Z","shell.execute_reply.started":"2022-08-08T13:08:39.315912Z","shell.execute_reply":"2022-08-08T13:08:39.339448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = sub.groupby('customer_ID').max().reset_index()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:08:43.492311Z","iopub.execute_input":"2022-08-08T13:08:43.493400Z","iopub.status.idle":"2022-08-08T13:08:43.525362Z","shell.execute_reply.started":"2022-08-08T13:08:43.493356Z","shell.execute_reply":"2022-08-08T13:08:43.524093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:08:45.207177Z","iopub.execute_input":"2022-08-08T13:08:45.207730Z","iopub.status.idle":"2022-08-08T13:08:45.221294Z","shell.execute_reply.started":"2022-08-08T13:08:45.207655Z","shell.execute_reply":"2022-08-08T13:08:45.220181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:08:08.712862Z","iopub.status.idle":"2022-08-08T13:08:08.713256Z","shell.execute_reply.started":"2022-08-08T13:08:08.713067Z","shell.execute_reply":"2022-08-08T13:08:08.713085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T13:08:08.714446Z","iopub.status.idle":"2022-08-08T13:08:08.715016Z","shell.execute_reply.started":"2022-08-08T13:08:08.714725Z","shell.execute_reply":"2022-08-08T13:08:08.714763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}