{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-10-04T06:31:13.785609Z","iopub.execute_input":"2022-10-04T06:31:13.786050Z","iopub.status.idle":"2022-10-04T06:31:13.820678Z","shell.execute_reply.started":"2022-10-04T06:31:13.785961Z","shell.execute_reply":"2022-10-04T06:31:13.819637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Why Do We Need To Convert Dataset?**\n\n> The given Train Datasets are in the form of CSV and have a combined size of 9 GB.\n\n> The Datasets were converted into a pickle format for improving performance during loading and preprocessing.\n","metadata":{}},{"cell_type":"markdown","source":"**Let's check the time taken for loading a single CSV dataset!!**","metadata":{}},{"cell_type":"markdown","source":"> Total time taken is 19.4 sec.","metadata":{}},{"cell_type":"code","source":"%%time\ntrain_0=pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_0.csv')","metadata":{"execution":{"iopub.status.busy":"2022-10-04T06:31:13.822599Z","iopub.execute_input":"2022-10-04T06:31:13.823241Z","iopub.status.idle":"2022-10-04T06:31:46.592147Z","shell.execute_reply.started":"2022-10-04T06:31:13.823205Z","shell.execute_reply":"2022-10-04T06:31:46.591027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_0.shape","metadata":{"execution":{"iopub.status.busy":"2022-10-04T06:31:46.593545Z","iopub.execute_input":"2022-10-04T06:31:46.593877Z","iopub.status.idle":"2022-10-04T06:31:46.603090Z","shell.execute_reply.started":"2022-10-04T06:31:46.593845Z","shell.execute_reply":"2022-10-04T06:31:46.602045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Convert  CSV files to a pickle format & check the time taken for loading!**","metadata":{}},{"cell_type":"code","source":"%%time\n# Load the training dataset one by one from train_0 to train_9.\n# Convert each of them to a pandas pickle format.\n\n\ni=0\n\nwhile i <10:\n\n    df_train=pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/train_{}.csv'.format(i))\n    \n    df_train.to_pickle('train_{}.pickle'.format(i),compression='zip')\n    \n    i=i+1","metadata":{"execution":{"iopub.status.busy":"2022-10-04T06:31:46.605685Z","iopub.execute_input":"2022-10-04T06:31:46.606235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Converting test data to pickle format\n\ndf_test=pd.read_csv('/kaggle/input/tabular-playground-series-oct-2022/test.csv')\n    \ndf_test.to_pickle('test.pickle',compression='zip')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Loading the pickle file!**","metadata":{}},{"cell_type":"code","source":"%%time\n\npd.read_pickle('/kaggle/working/train_0.pickle', compression='zip')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**What we done here?**\n> Datasets are converted from CSV to Pickle Format for faster retrieval of data.\n> \n> The size of the total Datasets was reduced from 19.5GB to 3.6GB.\n>\n> The time taken for loading CSV file ... and for a pickle file....\n> \n> Pickle files are uploaded in Kaggle and datasets can be added to your notebook by using the below link.","metadata":{}},{"cell_type":"markdown","source":"**Download Pickle Datsets!!** Refer [https://www.kaggle.com/getting-started/168312](http://)\n\n>Sometimes its difficult to download kaggle output files, it actually struck!\n","metadata":{}},{"cell_type":"code","source":"import os \nos.chdir(r'/kaggle/working')","metadata":{"execution":{"iopub.status.busy":"2022-10-04T06:46:03.571087Z","iopub.execute_input":"2022-10-04T06:46:03.571537Z","iopub.status.idle":"2022-10-04T06:46:03.577261Z","shell.execute_reply.started":"2022-10-04T06:46:03.571494Z","shell.execute_reply":"2022-10-04T06:46:03.575850Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLink \n\ndef download_files(num):\n    train=FileLink(r'train_{}.pickle'.format(num))\n    return train    ","metadata":{"execution":{"iopub.status.busy":"2022-10-04T06:46:05.728368Z","iopub.execute_input":"2022-10-04T06:46:05.728837Z","iopub.status.idle":"2022-10-04T06:46:05.734869Z","shell.execute_reply.started":"2022-10-04T06:46:05.728802Z","shell.execute_reply":"2022-10-04T06:46:05.733527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i=0\nwhile i<10:\n    display(download_files(i))\n    i=i+1","metadata":{"execution":{"iopub.status.busy":"2022-10-04T06:46:06.747323Z","iopub.execute_input":"2022-10-04T06:46:06.747791Z","iopub.status.idle":"2022-10-04T06:46:06.774419Z","shell.execute_reply.started":"2022-10-04T06:46:06.747749Z","shell.execute_reply":"2022-10-04T06:46:06.773199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"FileLink(r'test.pickle')","metadata":{"execution":{"iopub.status.busy":"2022-10-04T07:03:59.260730Z","iopub.execute_input":"2022-10-04T07:03:59.261263Z","iopub.status.idle":"2022-10-04T07:03:59.269556Z","shell.execute_reply.started":"2022-10-04T07:03:59.261225Z","shell.execute_reply":"2022-10-04T07:03:59.268293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Click on below link and directly upload the dataset in your notebook!**","metadata":{}},{"cell_type":"markdown","source":"[Click Here!](http://www.kaggle.com/datasets/itspavansatish/tps-oct-2022-pickle)","metadata":{}}]}