{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":45533,"databundleVersionId":5748852,"sourceType":"competition"}],"dockerImageVersionId":30474,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-22T11:34:40.100032Z","iopub.execute_input":"2023-05-22T11:34:40.100451Z","iopub.status.idle":"2023-05-22T11:34:40.141674Z","shell.execute_reply.started":"2023-05-22T11:34:40.100404Z","shell.execute_reply":"2023-05-22T11:34:40.140795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reduce_memory_usage(df):\n    \n    start_mem = df.memory_usage().sum() / 1024**2    \n    for col in df.columns:\n        col_type = df[col].dtype.name\n        if ((col_type != 'datetime64[ns]') & (col_type != 'category')):\n            if (col_type != 'object'):\n                c_min = df[col].min()\n                c_max = df[col].max()\n\n                if str(col_type)[:3] == 'int':\n                    if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                        df[col] = df[col].astype(np.int8)\n                    elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                        df[col] = df[col].astype(np.int16)\n                    elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                        df[col] = df[col].astype(np.int32)\n                    elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                        df[col] = df[col].astype(np.int64)\n\n                else:\n                    if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                        df[col] = df[col].astype(np.float16)\n                    elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                        df[col] = df[col].astype(np.float32)\n                    else:\n                        pass\n            else:\n                df[col] = df[col].astype('category')\n    mem_usg = df.memory_usage().sum() / 1024**2     \n    return df","metadata":{"execution":{"iopub.status.busy":"2023-05-22T11:35:03.051084Z","iopub.execute_input":"2023-05-22T11:35:03.051499Z","iopub.status.idle":"2023-05-22T11:35:03.065713Z","shell.execute_reply.started":"2023-05-22T11:35:03.051468Z","shell.execute_reply":"2023-05-22T11:35:03.064791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nchunk_size = 5000 # Adjust the chunk size based on your system's memory capacity\ntrain_df=[]\nfor chunk_train in pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/train.csv', chunksize=chunk_size):\n    train_df.append(reduce_memory_usage(chunk_train))\ntest_df=[]\nfor chunk_test in pd.read_csv('/kaggle/input/predict-student-performance-from-game-play/test.csv', chunksize=chunk_size):\n    test_df.append(reduce_memory_usage(chunk_test))\ntrain = pd.concat(train_df, ignore_index=True)\ntest  = pd.concat(test_df, ignore_index=True)\ntrain.to_csv(\"new_train.csv\",index=False)\ntest.to_csv(\"new_test.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-22T11:35:04.444798Z","iopub.execute_input":"2023-05-22T11:35:04.445221Z","iopub.status.idle":"2023-05-22T11:47:27.642498Z","shell.execute_reply.started":"2023-05-22T11:35:04.445189Z","shell.execute_reply":"2023-05-22T11:47:27.641121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.info()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2023-05-22T11:31:04.416956Z","iopub.execute_input":"2023-05-22T11:31:04.417361Z","iopub.status.idle":"2023-05-22T11:31:04.582712Z","shell.execute_reply.started":"2023-05-22T11:31:04.417328Z","shell.execute_reply":"2023-05-22T11:31:04.581794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}