{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45533,"databundleVersionId":5748852,"sourceType":"competition"}],"dockerImageVersionId":30380,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### This function is widely used in Kaggle and excels at minimizing the datatype sizes of features without losing any data. I'm highlighting it here especially for those interested in the efficiency prize, as it significantly reduces RAM requirements. I would also like to express my appreciation to the original creator of this function, [@arjanso](https://www.kaggle.com/arjanso).","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd ","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-02-06T14:44:24.584591Z","iopub.execute_input":"2024-02-06T14:44:24.585503Z","iopub.status.idle":"2024-02-06T14:44:24.611905Z","shell.execute_reply.started":"2024-02-06T14:44:24.585408Z","shell.execute_reply":"2024-02-06T14:44:24.611019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Reduce Memory Usage\ndef reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2\n    print('Memory usage of dataframe is {:.2f} MB'.format(start_mem))\n    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2 \n    print(\"Memory usage became: \",mem_usg,\" MB\")\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2024-02-06T14:44:24.613839Z","iopub.execute_input":"2024-02-06T14:44:24.614513Z","iopub.status.idle":"2024-02-06T14:44:24.632462Z","shell.execute_reply.started":"2024-02-06T14:44:24.614476Z","shell.execute_reply":"2024-02-06T14:44:24.631206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv')\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-06T21:52:44.904802Z","iopub.execute_input":"2023-02-06T21:52:44.905154Z","iopub.status.idle":"2023-02-06T21:53:30.100924Z","shell.execute_reply.started":"2023-02-06T21:52:44.905119Z","shell.execute_reply":"2023-02-06T21:53:30.099916Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = reduce_memory_usage(train_df)","metadata":{"execution":{"iopub.status.busy":"2023-02-06T21:53:57.947345Z","iopub.execute_input":"2023-02-06T21:53:57.947718Z","iopub.status.idle":"2023-02-06T21:54:08.680019Z","shell.execute_reply.started":"2023-02-06T21:53:57.947689Z","shell.execute_reply":"2023-02-06T21:54:08.678374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-02-06T21:54:08.681468Z","iopub.execute_input":"2023-02-06T21:54:08.681802Z","iopub.status.idle":"2023-02-06T21:54:08.700137Z","shell.execute_reply.started":"2023-02-06T21:54:08.681772Z","shell.execute_reply":"2023-02-06T21:54:08.698125Z"},"trusted":true},"execution_count":null,"outputs":[]}]}