{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 💡 TPS-OCT22, Sequential Dataset Loader & Memory Reduction\nThis Notebook demonstrate an example to use the reduce memory function in a sequential for loop to create a merged dataset for training; the results are stored at the end on a pickle dataset...","metadata":{}},{"cell_type":"markdown","source":"## 📚 1.0 Importing Required Libraries...","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport gc # garbage collector\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-10-01T15:45:14.985277Z","iopub.execute_input":"2022-10-01T15:45:14.985634Z","iopub.status.idle":"2022-10-01T15:45:15.002211Z","shell.execute_reply.started":"2022-10-01T15:45:14.985549Z","shell.execute_reply":"2022-10-01T15:45:15.000803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## ⚙️ 2.0 Configuring the Notebook...","metadata":{}},{"cell_type":"code","source":"%%time\n# I like to disable my Notebook warnings to reduce noice.\nimport warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-10-01T15:45:15.003603Z","iopub.execute_input":"2022-10-01T15:45:15.003918Z","iopub.status.idle":"2022-10-01T15:45:15.019966Z","shell.execute_reply.started":"2022-10-01T15:45:15.003893Z","shell.execute_reply":"2022-10-01T15:45:15.018668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Notebook Configuration.\n\n# Amount of data we want to load into the model from Pandas.\nDATA_ROWS = None\n\n# Dataframe, the amount of rows and cols to visualize.\nNROWS = 15\nNCOLS = 25\n\n# Main data location base path.\nBASE_PATH = '...'","metadata":{"execution":{"iopub.status.busy":"2022-10-01T15:45:15.021828Z","iopub.execute_input":"2022-10-01T15:45:15.022445Z","iopub.status.idle":"2022-10-01T15:45:15.031903Z","shell.execute_reply.started":"2022-10-01T15:45:15.022407Z","shell.execute_reply":"2022-10-01T15:45:15.030471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Configure notebook display settings to only use 2 decimal places, tables look nicer and compressed.\npd.options.display.float_format = '{:,.1f}'.format\npd.set_option('display.max_columns', NCOLS) \npd.set_option('display.max_rows', NROWS)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T15:45:15.034760Z","iopub.execute_input":"2022-10-01T15:45:15.035079Z","iopub.status.idle":"2022-10-01T15:45:15.043067Z","shell.execute_reply.started":"2022-10-01T15:45:15.035054Z","shell.execute_reply":"2022-10-01T15:45:15.042002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 🚀 3.0 Defining Auxiliary Functions...","metadata":{}},{"cell_type":"code","source":"%%time\ndef reduce_mem_usage(df, verbose = True):\n    '''\n    The following function reduce the size of the dataset...\n    '''\n    \n    numerics = ['int16', 'int32', 'int64', 'float16', 'float32', 'float64']\n    start_mem = df.memory_usage(deep=True).sum() / 1024 ** 2 # just added \n    \n    for col in df.columns:\n        col_type = df[col].dtypes\n        if col_type in numerics:\n            c_min = df[col].min()\n            c_max = df[col].max()\n            if str(col_type)[:3] == 'int':\n                if c_min > np.iinfo(np.int8).min and c_max < np.iinfo(np.int8).max:\n                    df[col] = df[col].astype(np.int8)\n                elif c_min > np.iinfo(np.int16).min and c_max < np.iinfo(np.int16).max:\n                    df[col] = df[col].astype(np.int16)\n                elif c_min > np.iinfo(np.int32).min and c_max < np.iinfo(np.int32).max:\n                    df[col] = df[col].astype(np.int32)\n                elif c_min > np.iinfo(np.int64).min and c_max < np.iinfo(np.int64).max:\n                    df[col] = df[col].astype(np.int64)  \n            else:\n                if c_min > np.finfo(np.float16).min and c_max < np.finfo(np.float16).max:\n                    df[col] = df[col].astype(np.float16)\n                elif c_min > np.finfo(np.float32).min and c_max < np.finfo(np.float32).max:\n                    df[col] = df[col].astype(np.float32)\n                else:\n                    df[col] = df[col].astype(np.float64)    \n    \n    end_mem = df.memory_usage(deep=True).sum() / 1024 ** 2\n    percent = 100 * (start_mem - end_mem) / start_mem\n    \n    print('Mem. usage decreased from {:5.2f} Mb to {:5.2f} Mb ({:.1f}% reduction)'.format(start_mem, end_mem, percent))\n    \n    return df","metadata":{"execution":{"iopub.status.busy":"2022-10-01T15:45:15.045599Z","iopub.execute_input":"2022-10-01T15:45:15.045932Z","iopub.status.idle":"2022-10-01T15:45:15.058168Z","shell.execute_reply.started":"2022-10-01T15:45:15.045899Z","shell.execute_reply":"2022-10-01T15:45:15.056750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 💾 4.0 Loading the Datasets...","metadata":{}},{"cell_type":"code","source":"%%time\n# Load the datasets sequentilly and apply memory optimizaiton function.\ndtypes_df = pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_dtypes.csv')\ndtypes = {k: v for (k, v) in zip(dtypes_df.column, dtypes_df.dtype)}\nmerged_df = pd.DataFrame()\n\nfor file_number in range(0,10):\n    print(f'Loading file number {file_number} & optimizing memory usage ... ')\n    tmp = pd.read_csv(f'/kaggle/input/tabular-playground-series-oct-2022/train_{file_number}.csv', dtype = dtypes)\n    tmp = reduce_mem_usage(tmp)\n    tmp['team_scoring_next'] = tmp['team_scoring_next'].astype('category')\n    merged_df = merged_df.append(tmp)\n    print('')\n\ndel tmp\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-10-01T15:45:15.059158Z","iopub.execute_input":"2022-10-01T15:45:15.059429Z","iopub.status.idle":"2022-10-01T15:50:06.216791Z","shell.execute_reply.started":"2022-10-01T15:45:15.059405Z","shell.execute_reply":"2022-10-01T15:50:06.215670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 🧭 5.0 Exploring the Loaded Information...","metadata":{}},{"cell_type":"code","source":"%%time\nmerged_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-10-01T15:50:06.218396Z","iopub.execute_input":"2022-10-01T15:50:06.219104Z","iopub.status.idle":"2022-10-01T15:50:06.242422Z","shell.execute_reply.started":"2022-10-01T15:50:06.219036Z","shell.execute_reply":"2022-10-01T15:50:06.240588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Display the first 5 rows of the dataset, it provides a general idea of the data.\nmerged_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-10-01T15:50:06.243927Z","iopub.execute_input":"2022-10-01T15:50:06.244430Z","iopub.status.idle":"2022-10-01T15:52:05.289473Z","shell.execute_reply.started":"2022-10-01T15:50:06.244403Z","shell.execute_reply":"2022-10-01T15:52:05.287649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 📥 6.0 Exporting the Merged Train Dataset...\nLocation of the optimized datset for future loading:\n\nhttps://www.kaggle.com/datasets/cv13j0/tpsoct22-memory-optimized-dataset","metadata":{}},{"cell_type":"code","source":"%%time\n# Export the merged dataset to a pickle file\npath = './merged_train_dataset.pkl'\nmerged_df.to_pickle(path, compression = 'infer', protocol = 4, storage_options = None)","metadata":{"execution":{"iopub.status.busy":"2022-10-01T15:52:05.290885Z","iopub.execute_input":"2022-10-01T15:52:05.291313Z","iopub.status.idle":"2022-10-01T15:52:19.134654Z","shell.execute_reply.started":"2022-10-01T15:52:05.291286Z","shell.execute_reply":"2022-10-01T15:52:19.133463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n# Location of the optimized datset for future loading\n# https://www.kaggle.com/datasets/cv13j0/tpsoct22-memory-optimized-dataset","metadata":{"execution":{"iopub.status.busy":"2022-10-01T17:54:09.358586Z","iopub.execute_input":"2022-10-01T17:54:09.359615Z","iopub.status.idle":"2022-10-01T17:54:09.366340Z","shell.execute_reply.started":"2022-10-01T17:54:09.359577Z","shell.execute_reply":"2022-10-01T17:54:09.364774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}