{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#  UMP: Create Pickle Dataset\n## Import Packages","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np\nimport gc\nimport math","metadata":{"execution":{"iopub.status.busy":"2022-01-23T13:43:17.188499Z","iopub.execute_input":"2022-01-23T13:43:17.188774Z","iopub.status.idle":"2022-01-23T13:43:17.210458Z","shell.execute_reply.started":"2022-01-23T13:43:17.188694Z","shell.execute_reply":"2022-01-23T13:43:17.209716Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Utilities","metadata":{}},{"cell_type":"code","source":"def reduce_memory_usage(df, features):\n    for feature in features:\n        item = df[feature].astype(np.float16)\n        df[feature] = item\n        del item\n        gc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-01-23T13:43:20.607186Z","iopub.execute_input":"2022-01-23T13:43:20.607458Z","iopub.status.idle":"2022-01-23T13:43:20.612391Z","shell.execute_reply.started":"2022-01-23T13:43:20.607429Z","shell.execute_reply":"2022-01-23T13:43:20.611549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import dataset","metadata":{}},{"cell_type":"code","source":"%%time\nn_features = 300\nfeatures = [f'f_{i}' for i in range(n_features)]\nfeature_columns = ['investment_id', 'time_id'] + features\ntrain = pd.read_parquet('../input/ubiquant-parquet/train_low_mem.parquet', columns=feature_columns + [\"target\"])\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-23T13:43:45.822761Z","iopub.execute_input":"2022-01-23T13:43:45.823636Z","iopub.status.idle":"2022-01-23T13:44:20.527955Z","shell.execute_reply.started":"2022-01-23T13:43:45.823597Z","shell.execute_reply":"2022-01-23T13:44:20.527344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Reducing Memories\nThere are totally 3141410 records and each record has 303 columns. If we convert all data type to int16 and float16, then the total memory of training data will be  (3141410 x 303 x 2)  / (1024^3) G, which is about 1.8G.","metadata":{}},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-01-23T13:44:45.464661Z","iopub.execute_input":"2022-01-23T13:44:45.464924Z","iopub.status.idle":"2022-01-23T13:44:45.493067Z","shell.execute_reply.started":"2022-01-23T13:44:45.464899Z","shell.execute_reply":"2022-01-23T13:44:45.492502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nreduce_memory_usage(train, features + [\"target\"])","metadata":{"execution":{"iopub.status.busy":"2022-01-23T13:44:54.321777Z","iopub.execute_input":"2022-01-23T13:44:54.322318Z","iopub.status.idle":"2022-01-23T13:48:17.664094Z","shell.execute_reply.started":"2022-01-23T13:44:54.322281Z","shell.execute_reply":"2022-01-23T13:48:17.663253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"execution":{"iopub.status.busy":"2022-01-23T13:48:21.741564Z","iopub.execute_input":"2022-01-23T13:48:21.742186Z","iopub.status.idle":"2022-01-23T13:48:21.766976Z","shell.execute_reply.started":"2022-01-23T13:48:21.742149Z","shell.execute_reply":"2022-01-23T13:48:21.766191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.to_pickle(\"train.pkl\")","metadata":{"execution":{"iopub.status.busy":"2022-01-23T13:52:25.027833Z","iopub.execute_input":"2022-01-23T13:52:25.0284Z","iopub.status.idle":"2022-01-23T13:52:28.745576Z","shell.execute_reply.started":"2022-01-23T13:52:25.02833Z","shell.execute_reply":"2022-01-23T13:52:28.744781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-01-23T13:53:56.140412Z","iopub.execute_input":"2022-01-23T13:53:56.141222Z","iopub.status.idle":"2022-01-23T13:53:56.14552Z","shell.execute_reply.started":"2022-01-23T13:53:56.141171Z","shell.execute_reply":"2022-01-23T13:53:56.144784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Read Pickle file","metadata":{}},{"cell_type":"code","source":"train = pd.read_pickle(\"/kaggle/working/train.pkl\")\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-01-23T13:56:55.116686Z","iopub.execute_input":"2022-01-23T13:56:55.117501Z","iopub.status.idle":"2022-01-23T13:56:56.660883Z","shell.execute_reply.started":"2022-01-23T13:56:55.117449Z","shell.execute_reply":"2022-01-23T13:56:56.66007Z"},"trusted":true},"execution_count":null,"outputs":[]}]}