{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-03T09:38:05.175438Z","iopub.execute_input":"2022-07-03T09:38:05.175832Z","iopub.status.idle":"2022-07-03T09:38:05.183694Z","shell.execute_reply.started":"2022-07-03T09:38:05.175798Z","shell.execute_reply":"2022-07-03T09:38:05.182629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Please make sure to upvote you use/ find interesting or useful**","metadata":{}},{"cell_type":"markdown","source":"The input data in this competition is ' Big data ' of very large size. To prevent memory over error while applying gradient boosting algorithms for AMEX time series data. We will be using 2 records for testing code. Let's begin by reading the files as feather.\n> Feather is a fast, lightweight, and easy-to-use binary file format for storing data frames. It has a few specific design goals: Lightweight, minimal API: make pushing data frames in and out of memory as simple as possible. Language agnostic: Feather files are the same whether written by Python or R code.\n\nThis is how you can save csv files as feather file inputs\n\n1. [Kaggle feather conversion](https://www.kaggle.com/code/kmader/convert-to-feather-for-use-in-other-kernels/notebook)\n2. [CSV to feather](https://www.kaggle.com/code/mathurinache/csv-to-feather/notebook)\n3. [Convert to feather for fast loading](https://www.kaggle.com/code/corochann/convert-to-feather-format-for-fast-data-loading/notebook)","metadata":{}},{"cell_type":"code","source":"%%time\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(filename)\n        try:\n            pd.read_csv(os.path.join(dirname, filename)).to_feather(filename[:-4]+'.feather')\n        except:\n            pass;\n        ","metadata":{"execution":{"iopub.status.busy":"2022-07-03T09:41:15.467763Z","iopub.execute_input":"2022-07-03T09:41:15.468503Z","iopub.status.idle":"2022-07-03T09:42:50.29644Z","shell.execute_reply.started":"2022-07-03T09:41:15.468462Z","shell.execute_reply":"2022-07-03T09:42:50.295336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nfrom pathlib import Path\n\noutdir = Path('.')\ndatadir = Path('../input/amex-default-prediction/')\n\nos.makedirs(str(outdir),exist_ok=True)\ntrain = pd.read_csv(datadir / 'train_data.csv').to_feather(outdir / 'train.feather')\ntest = pd.read_csv(datadir / 'test_data.csv').to_feather(outdir / 'test.feather')\nsubmission = pd.read_csv(datadir / 'sample_submission.csv').to_feather(outdir / 'sample_submission.feather')","metadata":{"execution":{"iopub.status.busy":"2022-07-03T09:39:22.272441Z","iopub.execute_input":"2022-07-03T09:39:22.273009Z","iopub.status.idle":"2022-07-03T09:40:24.866762Z","shell.execute_reply.started":"2022-07-03T09:39:22.272965Z","shell.execute_reply":"2022-07-03T09:40:24.865833Z"},"trusted":true},"execution_count":null,"outputs":[]}]}