{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"'''\n1. In this dataset only the last customer statement is taken because it is more important then other 12 statements for predicting default.  \n2. The converted datasets could be download from 'https://www.kaggle.com/datasets/ydvaakash/amex-train-dataset-vaex-format?datasetId=2391740&sortBy=dateRun&tab=profile'\n'''","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport gc; gc.enable()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-06T13:57:15.335411Z","iopub.execute_input":"2022-08-06T13:57:15.335973Z","iopub.status.idle":"2022-08-06T13:57:15.352642Z","shell.execute_reply.started":"2022-08-06T13:57:15.335853Z","shell.execute_reply":"2022-08-06T13:57:15.350831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"traini = pd.read_csv('/kaggle/input/amex-default-prediction/train_data.csv', parse_dates=['S_2'], chunksize=400_000, iterator=True)\nlabels = pd.read_csv('/kaggle/input/amex-default-prediction/train_labels.csv')\ntesti = pd.read_csv('/kaggle/input/amex-default-prediction/test_data.csv', parse_dates=['S_2'],chunksize=400_000, iterator=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:57:20.058181Z","iopub.execute_input":"2022-08-06T13:57:20.058715Z","iopub.status.idle":"2022-08-06T13:57:21.139459Z","shell.execute_reply.started":"2022-08-06T13:57:20.058673Z","shell.execute_reply":"2022-08-06T13:57:21.138112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = []\nfor df in testi:\n    if len(test)>0: test = pd.concat([test, df])\n    else: test = df[:]\n    test.sort_values(by=['S_2'], inplace=True)\n    test.reset_index(drop=True, inplace=True)\n    test.drop_duplicates(subset=['customer_ID'], keep='last', inplace=True)\n    del df; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T13:57:42.289034Z","iopub.execute_input":"2022-08-06T13:57:42.289540Z","iopub.status.idle":"2022-08-06T14:14:08.094125Z","shell.execute_reply.started":"2022-08-06T13:57:42.289498Z","shell.execute_reply":"2022-08-06T14:14:08.092626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = []\nfor df in traini:\n    if len(train)>0: train = pd.concat([train, df])\n    else: train = df[:]\n    train.sort_values(by=['S_2'], inplace=True)\n    train.reset_index(drop=True, inplace=True)\n    train.drop_duplicates(subset=['customer_ID'], keep='last', inplace=True)\n    del df; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T14:14:08.096526Z","iopub.execute_input":"2022-08-06T14:14:08.096950Z","iopub.status.idle":"2022-08-06T14:21:30.374815Z","shell.execute_reply.started":"2022-08-06T14:14:08.096910Z","shell.execute_reply":"2022-08-06T14:21:30.373749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.merge(train, labels, how='inner', on=['customer_ID'])\ndel labels; gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-08-06T14:21:30.376150Z","iopub.execute_input":"2022-08-06T14:21:30.380320Z","iopub.status.idle":"2022-08-06T14:21:32.837533Z","shell.execute_reply.started":"2022-08-06T14:21:30.380258Z","shell.execute_reply":"2022-08-06T14:21:32.836276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import vaex as vx","metadata":{"execution":{"iopub.status.busy":"2022-08-06T14:31:31.622872Z","iopub.execute_input":"2022-08-06T14:31:31.624372Z","iopub.status.idle":"2022-08-06T14:31:33.701376Z","shell.execute_reply.started":"2022-08-06T14:31:31.624310Z","shell.execute_reply":"2022-08-06T14:31:33.700045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#converting test pandas dataframe into vaex dataframe \ndf=vx.from_pandas(test)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.export_hdf5('AMEX_TEST_DATASET.hdf5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.chdir(r'/kaggle/working')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#downloading on local system\nfrom IPython.display import FileLink\nFileLink(r'./AMEX_TEST_DATASET.hdf5')\n","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train=vx.from_pandas(train)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.export_hdf5('AMEX_TRAIN_DATASET.hdf5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FileLink(r'./AMEX_TRAIN_DATASET.hdf5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x=os.stat('AMEX_TRAIN_DATASET.hdf5')\ny=os.stat('AMEX_TEST_DATASET.hdf5')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print((x.st_size)/1073741824)\nprint((y.st_size)/1073741824)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Size of train and test dataset is reduced to around 670 MB and 1.35 GB respectively.","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}