{"cells":[{"metadata":{},"cell_type":"markdown","source":"Reading a csv file with pd.read_csv can take a long time.\nLoading With feather and Datatable very fast.\n\nThe feather format uses the following notebook output file.  \nhttps://www.kaggle.com/yamsam/riiid-feather-format\n\nI also found the following NOTEBOOK very helpful.  \nhttps://www.kaggle.com/rohanrao/tutorial-on-reading-large-datasets  \nhttps://www.kaggle.com/yihdarshieh/riiid-verifying-private-test-dataset-properties  "},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install ../input/python-datatable/datatable-0.11.0-cp37-cp37m-manylinux2010_x86_64.whl\n!mkdir ../tmp/","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from time import time\nfrom contextlib import contextmanager\nimport pandas as pd\nfrom tqdm.auto import tqdm\nimport gc\nimport pickle\nimport datatable as dt\n\ngc.enable()\n\n@contextmanager\ndef timer(name):\n    t0 = time()\n    yield\n    print(f'[{name}] done in {time() - t0:.2f} s')\n\ndef sizeof_fmt(num, suffix='B'):\n    for unit in ['','Ki','Mi','Gi','Ti','Pi','Ei','Zi']:\n        if abs(num) < 1024.0:\n            return \"%3.1f%s%s\" % (num, unit, suffix)\n        num /= 1024.0\n    return \"%.1f%s%s\" % (num, 'Yi', suffix)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# feather"},{"metadata":{"trusted":true},"cell_type":"code","source":"!du -h ../input/riiid-feather-format/train.feather","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# read the feather format 10 times.\nwith timer('feather'):\n    for _ in tqdm(range(10)):\n        train_df = pd.read_feather('../input/riiid-feather-format/train.feather')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(sizeof_fmt(train_df.memory_usage().sum()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with timer('feather save'):\n    train_df.to_feather('../tmp/train.feather')","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# pickle"},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"with timer('pickle save'):\n    with open('../tmp/train.pickle', 'wb') as f:\n        pickle.dump(train_df, f)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!du -h ../tmp/train.pickle","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with timer('pickle load'):\n    for _ in tqdm(range(10)):\n        with open('../tmp/train.pickle', 'rb') as f:\n            train_df = pickle.load(f)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Datatable\n\nhttps://datatable.readthedocs.io/en/latest/index.html"},{"metadata":{"trusted":true},"cell_type":"code","source":"with timer('DataFrame save'):\n    dt.Frame(train_df).to_jay(\"train.jay\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"del train_df","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with timer('Datatable'):\n    for _ in tqdm(range(10)):\n        train_dt = dt.fread('train.jay')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with timer('Datatable.Frame save'):\n    train_dt.to_jay('train.jay')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"type(train_dt)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"!du -h train.jay","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"import sys\nprint(sizeof_fmt(sys.getsizeof(train_dt)))\ndel train_dt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with timer('Datatable to pd.DataFrame'):\n    for _ in tqdm(range(10)):\n        train_df = dt.fread('train.jay').to_pandas()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"type(train_df)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df.dtypes","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# conclusion\n\nDatatable loads really fast. However, the conversion from Datatable to pandas.DataFrame is not fast.  \nThe feather format was able to load in about 2 seconds.  "}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}