{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Reduce Memory 2GB ==> 500MB","metadata":{}},{"cell_type":"markdown","source":"This is a modified version of [@Mohamed Eltayeb](https://www.kaggle.com/mohammad2012191)'s notebook:\n\n[https://www.kaggle.com/code/mohammad2012191/reduce-memory-usage-2gb-780mb](https://www.kaggle.com/code/mohammad2012191/reduce-memory-usage-2gb-780mb).\n\nThere are two steps to reduce memory usage:\n\n- Don't load `fullscreen`,`hq`,`music` columns. These features are full of NA values.\n\n- Use the `reduce_memory_usage` function. This function is copied from Mohamed Eltayeb's notebook. Thanks to the original author of this function [@ArjanGroen](https://www.kaggle.com/arjanso).","metadata":{}},{"cell_type":"markdown","source":"Train data after memory reduction is here: [https://www.kaggle.com/datasets/curiosity30/sp-reduce-mem-train](https://www.kaggle.com/datasets/curiosity30/sp-reduce-mem-train).\n\nFor ease of use, I have uploaded this dataset to this notebook. You can directly copy this notebook to start your work.\n\n**Please upvote if this notebook helps you. Thank you for your support!**","metadata":{}},{"cell_type":"markdown","source":"# Code","metadata":{}},{"cell_type":"code","source":"import pandas as pd, numpy as np\nfrom sklearn.model_selection import KFold, GroupKFold\nimport lightgbm as lgb\nfrom xgboost import XGBClassifier\nfrom sklearn.metrics import f1_score\nimport os\nimport gc","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-15T03:38:03.813839Z","iopub.execute_input":"2023-02-15T03:38:03.814326Z","iopub.status.idle":"2023-02-15T03:38:05.308356Z","shell.execute_reply.started":"2023-02-15T03:38:03.814229Z","shell.execute_reply":"2023-02-15T03:38:05.307384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('BEFORE: Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"AFTER: Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-02-15T03:38:05.310189Z","iopub.execute_input":"2023-02-15T03:38:05.310540Z","iopub.status.idle":"2023-02-15T03:38:05.324525Z","shell.execute_reply.started":"2023-02-15T03:38:05.310506Z","shell.execute_reply":"2023-02-15T03:38:05.323666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntrain_path = '/kaggle/input/predict-student-performance-from-game-play/train.csv'\ntrain_cols = ['session_id', 'index', 'elapsed_time', 'event_name', 'name', 'level', 'page', \\\n              'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration', \\\n              'text', 'fqid', 'room_fqid', 'text_fqid', 'level_group']\ntrain = pd.read_csv(train_path, usecols=train_cols)\nprint(train.shape)","metadata":{"execution":{"iopub.status.busy":"2023-02-15T03:38:05.325941Z","iopub.execute_input":"2023-02-15T03:38:05.326683Z","iopub.status.idle":"2023-02-15T03:38:56.100569Z","shell.execute_reply.started":"2023-02-15T03:38:05.326648Z","shell.execute_reply":"2023-02-15T03:38:56.099732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = reduce_memory_usage(train)\ntrain.to_pickle('reduce_train.pkl')\nprint('OK!')","metadata":{"execution":{"iopub.status.busy":"2023-02-15T03:38:56.102437Z","iopub.execute_input":"2023-02-15T03:38:56.103413Z","iopub.status.idle":"2023-02-15T03:39:09.513884Z","shell.execute_reply.started":"2023-02-15T03:38:56.103376Z","shell.execute_reply":"2023-02-15T03:39:09.512981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.dtypes)","metadata":{"execution":{"iopub.status.busy":"2023-02-15T03:39:09.517756Z","iopub.execute_input":"2023-02-15T03:39:09.519731Z","iopub.status.idle":"2023-02-15T03:39:09.529141Z","shell.execute_reply.started":"2023-02-15T03:39:09.519692Z","shell.execute_reply":"2023-02-15T03:39:09.528098Z"},"trusted":true},"execution_count":null,"outputs":[]}]}