{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## American Express: Create Training Pickle File","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport gc","metadata":{"execution":{"iopub.status.busy":"2022-10-08T08:28:54.472589Z","iopub.execute_input":"2022-10-08T08:28:54.472918Z","iopub.status.idle":"2022-10-08T08:28:54.477468Z","shell.execute_reply.started":"2022-10-08T08:28:54.472894Z","shell.execute_reply":"2022-10-08T08:28:54.475666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_data = pd.DataFrame()\nwith pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', chunksize=10**5) as reader:\n    for counter, chunk in enumerate(reader):\n        for column in chunk.columns:\n            if str(chunk[column].dtype) == \"float64\":\n                chunk[column] = chunk[column].astype(np.float32)\n            if str(chunk[column].dtype) == \"int64\":\n                chunk[column] = chunk[column].astype(np.int32)\n        train_data = pd.concat([train_data, chunk])\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-08T08:28:56.407351Z","iopub.execute_input":"2022-10-08T08:28:56.407736Z","iopub.status.idle":"2022-10-08T08:38:04.14336Z","shell.execute_reply.started":"2022-10-08T08:28:56.407711Z","shell.execute_reply":"2022-10-08T08:38:04.14227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_data = pd.DataFrame()\nwith pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', chunksize=10**5) as reader:\n    for counter, chunk in enumerate(reader):\n#         for column in chunk.columns:\n#             if str(chunk[column].dtype) == \"float64\":\n#                 chunk[column] = chunk[column].astype(np.float32)\n#             if str(chunk[column].dtype) == \"int64\":\n#                 chunk[column] = chunk[column].astype(np.int32)\n        train_data = pd.concat([train_data, chunk])\n        gc.collect()\ntrain_data.memory_usage().sum() / 1024**2","metadata":{"execution":{"iopub.status.busy":"2022-10-08T08:48:51.375301Z","iopub.execute_input":"2022-10-08T08:48:51.375722Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data.to_pickle(\"train_data.pkl\")","metadata":{"execution":{"iopub.status.busy":"2022-10-08T07:06:47.441307Z","iopub.execute_input":"2022-10-08T07:06:47.442248Z","iopub.status.idle":"2022-10-08T07:07:02.849333Z","shell.execute_reply.started":"2022-10-08T07:06:47.442191Z","shell.execute_reply":"2022-10-08T07:07:02.848327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels = pd.read_csv(\"/kaggle/input/amex-default-prediction/train_labels.csv\")\ntrain_labels.to_pickle(\"train_labels.pkl\")","metadata":{"execution":{"iopub.status.busy":"2022-10-08T07:10:51.344069Z","iopub.execute_input":"2022-10-08T07:10:51.344491Z","iopub.status.idle":"2022-10-08T07:10:52.609997Z","shell.execute_reply.started":"2022-10-08T07:10:51.344454Z","shell.execute_reply":"2022-10-08T07:10:52.608662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_mem_usage(df, verbose=True):\n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage().sum() / 1024**2    \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)    \n    end_mem = df.memory_usage().sum() / 1024**2\n    if verbose: print('Mem. usage decreased to {:5.2f} Mb ({:.1f}% reduction)'.format(end_mem, 100 * (start_mem - end_mem) / start_mem))\n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-08T08:43:03.5536Z","iopub.execute_input":"2022-10-08T08:43:03.553977Z","iopub.status.idle":"2022-10-08T08:43:03.567766Z","shell.execute_reply.started":"2022-10-08T08:43:03.55395Z","shell.execute_reply":"2022-10-08T08:43:03.566783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_mem_red = reduce_mem_usage(train_data)","metadata":{"execution":{"iopub.status.busy":"2022-10-08T08:44:16.631525Z","iopub.execute_input":"2022-10-08T08:44:16.631998Z","iopub.status.idle":"2022-10-08T08:44:48.898153Z","shell.execute_reply.started":"2022-10-08T08:44:16.631956Z","shell.execute_reply":"2022-10-08T08:44:48.896763Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = reduce_mem_usage(train)\ntest = reduce_mem_usage(test)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"存储数据","metadata":{}},{"cell_type":"code","source":"%%time\nif SAVE_PREPROCESSED_DATA:\n    train.to_pickle(\"train_agg_data_{}.pkl\".format(data_version), compression=\"gzip\")\n    test.to_pickle(\"test_agg_data_{}.pkl\".format(data_version), compression=\"gzip\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-08T07:19:54.905341Z","iopub.execute_input":"2022-10-08T07:19:54.906777Z","iopub.status.idle":"2022-10-08T07:19:55.038897Z","shell.execute_reply.started":"2022-10-08T07:19:54.906727Z","shell.execute_reply":"2022-10-08T07:19:55.038006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train_data.merge(train_labels, left_on='customer_ID', right_on='customer_ID')","metadata":{"execution":{"iopub.status.busy":"2022-10-08T07:18:39.890522Z","iopub.execute_input":"2022-10-08T07:18:39.89154Z","iopub.status.idle":"2022-10-08T07:18:47.639213Z","shell.execute_reply.started":"2022-10-08T07:18:39.891494Z","shell.execute_reply":"2022-10-08T07:18:47.638111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest_data = pd.DataFrame()\nwith pd.read_csv('/kaggle/input/amex-default-prediction/test_data.csv', chunksize=10**5) as reader:\n    for counter, chunk in enumerate(reader):\n        for column in chunk.columns:\n            if str(chunk[column].dtype) == \"float64\":\n                chunk[column] = chunk[column].astype(np.float32)\n            if str(chunk[column].dtype) == \"int64\":\n                chunk[column] = chunk[column].astype(np.int32)\n        test_data = pd.concat([test_data, chunk])\n        gc.collect()\ntest_data.to_pickle(\"test_data.pkl\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = test_data","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-08T07:25:47.336657Z","iopub.execute_input":"2022-10-08T07:25:47.33722Z","iopub.status.idle":"2022-10-08T07:25:47.444917Z","shell.execute_reply.started":"2022-10-08T07:25:47.337184Z","shell.execute_reply":"2022-10-08T07:25:47.443688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = train.drop(['customer_ID', 'S_2', 'target'], axis=1).columns.to_list()\ncat_features = [\n    \"B_30\",\n    \"B_38\",\n    \"D_114\",\n    \"D_116\",\n    \"D_117\",\n    \"D_120\",\n    \"D_126\",\n    \"D_63\",\n    \"D_64\",\n    \"D_66\",\n    \"D_68\",\n]\nnum_features = [col for col in features if col not in cat_features]","metadata":{"execution":{"iopub.status.busy":"2022-10-08T07:35:40.163465Z","iopub.execute_input":"2022-10-08T07:35:40.164162Z","iopub.status.idle":"2022-10-08T07:35:40.261427Z","shell.execute_reply.started":"2022-10-08T07:35:40.164027Z","shell.execute_reply":"2022-10-08T07:35:40.260281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_num_agg = train.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\ntrain_num_agg.columns = ['_'.join(x) for x in train_num_agg.columns]\ntrain_cat_agg = train.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\ntrain_cat_agg.columns = ['_'.join(x) for x in train_cat_agg.columns]\ntrain_target = (train.groupby(\"customer_ID\").tail(1).set_index('customer_ID', drop=True).sort_index()[\"target\"])\ntrain = pd.concat([train_num_agg, train_cat_agg, train_target], axis=1)\n\ntrain.to_pickle(\"../data/train_agg.pkl\", compression=\"gzip\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_num_agg = test.groupby(\"customer_ID\")[num_features].agg(['mean', 'std', 'min', 'max', 'last'])\ntest_num_agg.columns = ['_'.join(x) for x in test_num_agg.columns]\ntest_cat_agg = test.groupby(\"customer_ID\")[cat_features].agg(['count', 'last', 'nunique'])\ntest_cat_agg.columns = ['_'.join(x) for x in test_cat_agg.columns]\ntest = pd.concat([test_num_agg, test_cat_agg], axis=1)\n\ntest.to_pickle(\"../data/test_agg.pkl\", compression=\"gzip\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}