{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# AMEX - Data Setup","metadata":{}},{"cell_type":"code","source":"import logging\nimport numpy as np\nimport pandas as pd \nimport pyarrow as pa\nimport pyarrow.parquet as pq\n\nlogging.basicConfig(format=\"[%(levelname)s] %(asctime)s: %(message)s\",\n                    datefmt=\"%Y-%m-%d %H:%M:%S\",\n                    level=logging.INFO)\nlogging.getLogger().addHandler(logging.StreamHandler())","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-24T07:07:18.592657Z","iopub.execute_input":"2022-07-24T07:07:18.593113Z","iopub.status.idle":"2022-07-24T07:07:18.624287Z","shell.execute_reply.started":"2022-07-24T07:07:18.593021Z","shell.execute_reply":"2022-07-24T07:07:18.623455Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def covert_to_parquet(\n    subset: str,\n    batch_size: int,\n    max_periods: int = 13,\n    max_batch: int or str = \"auto\"\n):\n    \n    data_dir = \"../input/amex-default-prediction\"\n    \n    if subset == \"train\":\n        y_path = f\"{data_dir}/train_labels.csv\"\n    else:\n        y_path = f\"{data_dir}/sample_submission.csv\"\n    \n    y = pd.read_csv(y_path)\n    \n    x_path = f\"{data_dir}/{subset}_data.csv\"\n    cols = pd.read_csv(x_path, nrows=1).columns\n    cats = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', \n            'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n    \n    batch_num = 0\n    row_start = 1\n    \n    if max_batch == \"auto\":\n        max_batch = np.ceil(y.shape[0] / batch_size)\n    \n    while batch_num < max_batch:\n        \n        df = pd.read_csv(x_path, \n                         header=None, \n                         skiprows=row_start, \n                         nrows=batch_size*max_periods, \n                         names=cols, \n                         parse_dates=[\"S_2\"])\n        \n        df = df[pd.factorize(df[\"customer_ID\"])[0] < batch_size].copy()\n        df.loc[:, \"batch\"] = batch_num\n        \n        for c in cats:\n            if df[c].dtypes == \"object\":\n                df.loc[:, c] = df[c].fillna(\"NA\")\n            else:\n                df.loc[:, c] = df[c].fillna(-9).astype(int)\n            df.loc[:, c] = df[c].astype(\"category\")\n                    \n        table = pa.Table.from_pandas(df.set_index([\"customer_ID\", \"S_2\"]))\n        pq.write_to_dataset(table, \n                            root_path=f\"{subset}_data\",\n                            partition_cols=[\"batch\"])\n        \n        logging.info(\"Batch %d complete\", batch_num)\n        batch_num += 1\n        row_start += df.shape[0]\n            \ncovert_to_parquet(subset=\"test\", batch_size=3e4)","metadata":{"execution":{"iopub.status.busy":"2022-07-24T07:07:18.626230Z","iopub.execute_input":"2022-07-24T07:07:18.626723Z"},"trusted":true},"execution_count":null,"outputs":[]}]}