{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport gc\nimport pandas as pd\nfrom humanize import naturalsize","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-06-21T18:12:50.984173Z","iopub.execute_input":"2022-06-21T18:12:50.984661Z","iopub.status.idle":"2022-06-21T18:12:51.016405Z","shell.execute_reply.started":"2022-06-21T18:12:50.984563Z","shell.execute_reply":"2022-06-21T18:12:51.01526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = '/kaggle/input/amex-default-prediction'\n\nfor file in os.listdir(data_dir):\n    size = os.path.getsize(os.path.join(data_dir, file))\n    size = naturalsize(size)\n    print('{}: {}'.format(file, size))","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:12:51.731288Z","iopub.execute_input":"2022-06-21T18:12:51.731903Z","iopub.status.idle":"2022-06-21T18:12:51.740797Z","shell.execute_reply.started":"2022-06-21T18:12:51.731869Z","shell.execute_reply":"2022-06-21T18:12:51.739996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntrain_df = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', nrows=1000)","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:12:52.327407Z","iopub.execute_input":"2022-06-21T18:12:52.328288Z","iopub.status.idle":"2022-06-21T18:12:52.428667Z","shell.execute_reply.started":"2022-06-21T18:12:52.328239Z","shell.execute_reply":"2022-06-21T18:12:52.427866Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"naturalsize(train_df.memory_usage(deep=True).sum())","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:12:53.118854Z","iopub.execute_input":"2022-06-21T18:12:53.119743Z","iopub.status.idle":"2022-06-21T18:12:53.151416Z","shell.execute_reply.started":"2022-06-21T18:12:53.119659Z","shell.execute_reply":"2022-06-21T18:12:53.150515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_cols = train_df.columns.to_list()\ncat_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\ndtype = {col: 'float16' for col in all_cols if col not in cat_cols + ['customer_ID', 'S_2']}\n\nfor col in cat_cols + ['customer_ID']:\n    dtype[col] = 'category'\n    ","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:12:54.06119Z","iopub.execute_input":"2022-06-21T18:12:54.061853Z","iopub.status.idle":"2022-06-21T18:12:54.069127Z","shell.execute_reply.started":"2022-06-21T18:12:54.061807Z","shell.execute_reply":"2022-06-21T18:12:54.068301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntrain_df = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', nrows=1000, dtype=dtype)\ntrain_df.S_2 = pd.to_datetime(train_df.S_2)","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:12:57.493391Z","iopub.execute_input":"2022-06-21T18:12:57.493832Z","iopub.status.idle":"2022-06-21T18:12:57.576351Z","shell.execute_reply.started":"2022-06-21T18:12:57.493798Z","shell.execute_reply":"2022-06-21T18:12:57.575225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"naturalsize(train_df.memory_usage(deep=True).sum())","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:12:58.656361Z","iopub.execute_input":"2022-06-21T18:12:58.656777Z","iopub.status.idle":"2022-06-21T18:12:58.679019Z","shell.execute_reply.started":"2022-06-21T18:12:58.656744Z","shell.execute_reply":"2022-06-21T18:12:58.677983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:13:00.20513Z","iopub.execute_input":"2022-06-21T18:13:00.205536Z","iopub.status.idle":"2022-06-21T18:13:00.312348Z","shell.execute_reply.started":"2022-06-21T18:13:00.205504Z","shell.execute_reply":"2022-06-21T18:13:00.311227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train data","metadata":{}},{"cell_type":"code","source":"def process(df):\n    \n    df.S_2 = pd.to_datetime(df.S_2)\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:13:04.721089Z","iopub.execute_input":"2022-06-21T18:13:04.72147Z","iopub.status.idle":"2022-06-21T18:13:04.72667Z","shell.execute_reply.started":"2022-06-21T18:13:04.721438Z","shell.execute_reply":"2022-06-21T18:13:04.725757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nFILE = '/kaggle/input/amex-default-prediction/train_data.csv'\nCHUNKSIZE = 400000\ntrain_df = pd.DataFrame([])\n\nwith pd.read_csv(FILE, chunksize=CHUNKSIZE, dtype=dtype) as reader:\n    for i, chunk in enumerate(reader):\n        print('processing chunk {}'.format(i + 1))\n        train_df = train_df.append(process(chunk), ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:03:51.446424Z","iopub.execute_input":"2022-06-21T18:03:51.44685Z","iopub.status.idle":"2022-06-21T18:10:26.555514Z","shell.execute_reply.started":"2022-06-21T18:03:51.446789Z","shell.execute_reply":"2022-06-21T18:10:26.554311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"naturalsize(train_df.memory_usage(deep=True).sum())","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:10:26.557448Z","iopub.execute_input":"2022-06-21T18:10:26.558457Z","iopub.status.idle":"2022-06-21T18:10:27.629252Z","shell.execute_reply.started":"2022-06-21T18:10:26.558395Z","shell.execute_reply":"2022-06-21T18:10:27.628011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntrain_df.to_pickle('train_data.pkl')","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:11:11.300041Z","iopub.execute_input":"2022-06-21T18:11:11.300606Z","iopub.status.idle":"2022-06-21T18:11:16.100258Z","shell.execute_reply.started":"2022-06-21T18:11:11.300563Z","shell.execute_reply":"2022-06-21T18:11:16.09931Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:11:20.45659Z","iopub.execute_input":"2022-06-21T18:11:20.457043Z","iopub.status.idle":"2022-06-21T18:11:20.633417Z","shell.execute_reply.started":"2022-06-21T18:11:20.457004Z","shell.execute_reply":"2022-06-21T18:11:20.632424Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test data","metadata":{}},{"cell_type":"code","source":"%%time\n\nFILE = '/kaggle/input/amex-default-prediction/test_data.csv'\nCHUNKSIZE = 400000\ntest_df = pd.DataFrame([])\n\nwith pd.read_csv(FILE, chunksize=CHUNKSIZE, dtype=dtype) as reader:\n    for i, chunk in enumerate(reader):\n        print('processing chunk {}'.format(i + 1))\n        test_df = test_df.append(process(chunk), ignore_index=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:13:10.271712Z","iopub.execute_input":"2022-06-21T18:13:10.272101Z","iopub.status.idle":"2022-06-21T18:26:38.039965Z","shell.execute_reply.started":"2022-06-21T18:13:10.272068Z","shell.execute_reply":"2022-06-21T18:26:38.038607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"naturalsize(test_df.memory_usage(deep=True).sum())","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:26:38.042133Z","iopub.execute_input":"2022-06-21T18:26:38.042617Z","iopub.status.idle":"2022-06-21T18:26:40.042264Z","shell.execute_reply.started":"2022-06-21T18:26:38.042571Z","shell.execute_reply":"2022-06-21T18:26:40.041434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\ntest_df.to_pickle('test_data.pkl')","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:28:34.216463Z","iopub.execute_input":"2022-06-21T18:28:34.216886Z","iopub.status.idle":"2022-06-21T18:28:49.967133Z","shell.execute_reply.started":"2022-06-21T18:28:34.216854Z","shell.execute_reply":"2022-06-21T18:28:49.965956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del test_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-06-21T18:32:30.605379Z","iopub.execute_input":"2022-06-21T18:32:30.607040Z","iopub.status.idle":"2022-06-21T18:32:30.711620Z","shell.execute_reply.started":"2022-06-21T18:32:30.606986Z","shell.execute_reply":"2022-06-21T18:32:30.710147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}