{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Compress data to Parquet format\n\n- The train.csv file is about 2.3GB; Converting it to Parquet format will decrease the size to around 177MB.\n\n- Parquet format will also make it possible to read data by columns and speed up feature engineering process. ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-21T13:19:02.746583Z","iopub.execute_input":"2023-02-21T13:19:02.747331Z","iopub.status.idle":"2023-02-21T13:19:02.780812Z","shell.execute_reply.started":"2023-02-21T13:19:02.747243Z","shell.execute_reply":"2023-02-21T13:19:02.779866Z"}}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport gc\nimport copy\nimport os\nimport sys\n\nfrom pathlib import Path\nfrom datetime import datetime, date, time, timedelta\nfrom dateutil import relativedelta\n\nimport pyarrow.parquet as pq\nimport pyarrow as pa","metadata":{"execution":{"iopub.status.busy":"2023-02-21T14:24:47.163184Z","iopub.execute_input":"2023-02-21T14:24:47.164967Z","iopub.status.idle":"2023-02-21T14:24:47.174599Z","shell.execute_reply.started":"2023-02-21T14:24:47.164905Z","shell.execute_reply":"2023-02-21T14:24:47.173247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#https://stackoverflow.com/questions/25962114/how-do-i-read-a-large-csv-file-with-pandas\n#https://www.kaggle.com/code/xxxxyyyy80008/jpx-data-processing-csv2parquet\n\ndef process_big_csv(chunk, dest_file):\n    #---convert float64 to float32--------\n    float64_cols = chunk.select_dtypes(include=['float64']).columns.tolist()\n    chunk[float64_cols] = np.float32(chunk[float64_cols].values)\n    #---convert int64 to int32\n    int64_cols = chunk.select_dtypes(include=['int64']).columns.tolist()\n    chunk[int64_cols] = np.int32(chunk[int64_cols].values)\n    \n    #-- save to parquet file\n    table = pa.Table.from_pandas(chunk)\n    pq.write_table(table, dest_file, compression = 'GZIP')\n    \n    del table, chunk\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T14:25:08.437756Z","iopub.execute_input":"2023-02-21T14:25:08.438133Z","iopub.status.idle":"2023-02-21T14:25:08.445853Z","shell.execute_reply.started":"2023-02-21T14:25:08.438101Z","shell.execute_reply":"2023-02-21T14:25:08.445068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_label_file = r\"/kaggle/input/predict-student-performance-from-game-play/train_labels.csv\"\ntrain_file = r\"/kaggle/input/predict-student-performance-from-game-play/train.csv\"","metadata":{"execution":{"iopub.status.busy":"2023-02-21T14:25:55.465043Z","iopub.execute_input":"2023-02-21T14:25:55.465562Z","iopub.status.idle":"2023-02-21T14:25:55.471791Z","shell.execute_reply.started":"2023-02-21T14:25:55.465527Z","shell.execute_reply":"2023-02-21T14:25:55.470285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nfor raw_file, file_name in zip([train_file, train_label_file], \n                               ['train', 'train_labels']):\n    df = pd.read_csv(raw_file, low_memory=True)\n    mem_size = (df.memory_usage(deep=True).sum())/2473897 #convert memory usage to MB size\n    print(f\"file: {file_name}\\t file size: {mem_size:,.1f}\")         \n\n    dest_file = f'{file_name}.parquet'\n\n    process_big_csv(df, dest_file)\n\n    del df\n    gc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-02-21T14:35:47.166687Z","iopub.execute_input":"2023-02-21T14:35:47.167621Z","iopub.status.idle":"2023-02-21T14:37:51.417536Z","shell.execute_reply.started":"2023-02-21T14:35:47.167571Z","shell.execute_reply":"2023-02-21T14:37:51.415706Z"},"trusted":true},"execution_count":null,"outputs":[]}]}