{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd, numpy as np\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-22T14:28:08.306215Z","iopub.execute_input":"2023-03-22T14:28:08.306929Z","iopub.status.idle":"2023-03-22T14:28:08.317790Z","shell.execute_reply.started":"2023-03-22T14:28:08.306887Z","shell.execute_reply":"2023-03-22T14:28:08.316736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reduce Memory Usage\ndef reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n#                     if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n#                         df[col] = df[col].astype(np.int8)\n                    if c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n#                     if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n#                         df[col] = df[col].astype(np.float16)\n                    if c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-03-22T14:28:08.318923Z","iopub.execute_input":"2023-03-22T14:28:08.319620Z","iopub.status.idle":"2023-03-22T14:28:08.331793Z","shell.execute_reply.started":"2023-03-22T14:28:08.319583Z","shell.execute_reply":"2023-03-22T14:28:08.330567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = '/kaggle/input/predict-student-performance-from-game-play/train.csv'\ntrain_cols = ['session_id', 'index', 'elapsed_time', 'event_name', 'name', 'level', 'page', \\\n              'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration', \\\n              'text', 'fqid', 'room_fqid', 'text_fqid', 'level_group']\n\ntmp = pd.read_csv(train_path, usecols=train_cols, chunksize=10000000)\n\ntrain = pd.DataFrame()\nfor data in tmp:\n    data = reduce_memory_usage(data)\n    train = pd.concat([train, data], axis=0)\n    print( train.shape )\n    del data\n    gc.collect()\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-22T14:28:08.333042Z","iopub.execute_input":"2023-03-22T14:28:08.333923Z","iopub.status.idle":"2023-03-22T14:29:58.617824Z","shell.execute_reply.started":"2023-03-22T14:28:08.333884Z","shell.execute_reply":"2023-03-22T14:29:58.616587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.to_parquet('train.parquet')","metadata":{"execution":{"iopub.status.busy":"2023-03-22T14:29:58.620154Z","iopub.execute_input":"2023-03-22T14:29:58.620586Z","iopub.status.idle":"2023-03-22T14:30:14.415855Z","shell.execute_reply.started":"2023-03-22T14:29:58.620550Z","shell.execute_reply":"2023-03-22T14:30:14.414091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}