{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.18","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":35332,"databundleVersionId":3723648,"sourceType":"competition"}],"dockerImageVersionId":31091,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install dask","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T23:17:18.344465Z","iopub.execute_input":"2025-09-29T23:17:18.344677Z","iopub.status.idle":"2025-09-29T23:17:24.089611Z","shell.execute_reply.started":"2025-09-29T23:17:18.344658Z","shell.execute_reply":"2025-09-29T23:17:24.085911Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!python -m pip install \"dask[distributed]\" --upgrade ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T23:17:27.978603Z","iopub.execute_input":"2025-09-29T23:17:27.978762Z","iopub.status.idle":"2025-09-29T23:17:30.907596Z","shell.execute_reply.started":"2025-09-29T23:17:27.978746Z","shell.execute_reply":"2025-09-29T23:17:30.904741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport dask.dataframe as dd\nimport pandas as pd\nimport os, shutil\nfrom dask.distributed import Client","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T23:47:52.176426Z","iopub.execute_input":"2025-09-29T23:47:52.178554Z","iopub.status.idle":"2025-09-29T23:47:52.189082Z","shell.execute_reply.started":"2025-09-29T23:47:52.178531Z","shell.execute_reply":"2025-09-29T23:47:52.183505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"client = Client(n_workers=8, threads_per_worker=2, memory_limit=\"32GB\")\nprint(client)\n\nID_COL = \"customer_ID\"\n\nTRAIN_PATH  = \"/kaggle/input/amex-default-prediction/train_data.csv\"\nTEST_PATH   = \"/kaggle/input/amex-default-prediction/test_data.csv\"\nLABELS_PATH = \"/kaggle/input/amex-default-prediction/train_labels.csv\"\n\nTRAIN_OUT       = \"/kaggle/working/train_agg.parquet\"\nTEST_OUT        = \"/kaggle/working/test_agg.parquet\"\nFINAL_TRAIN_OUT = \"/kaggle/working/train_final.parquet\"\n\nddf_train = dd.read_csv(TRAIN_PATH, blocksize=\"2GB\")\nddf_test  = dd.read_csv(TEST_PATH, blocksize=\"2GB\")\n\nnum_cols = ddf_train.select_dtypes(include=[\"number\"]).columns.tolist()\nprint(\"Numeric cols:\", len(num_cols))\n\naggs = [\"mean\", \"std\", \"min\", \"max\", \"last\"]\n\ndef aggregate(ddf, num_cols, id_col=ID_COL, split_out=128):\n    agg_ddf = (\n        ddf.groupby(id_col)[num_cols]\n        .agg(aggs, split_out=split_out)\n        .reset_index()\n    )\n    agg_ddf.columns = [\n        col if col == id_col else f\"{col[0]}_{col[1]}\"\n        for col in agg_ddf.columns.values\n    ]\n    return agg_ddf\n\ntrain_agg_ddf = aggregate(ddf_train, num_cols, split_out=256)\ntest_agg_ddf  = aggregate(ddf_test, num_cols, split_out=256)\n\ndef save_parquet(ddf, path):\n    if os.path.exists(path):\n        shutil.rmtree(path)\n    ddf.to_parquet(\n        path,\n        engine=\"pyarrow\",\n        compression=\"zstd\",  \n        write_index=False\n    )\n    print(f\"✔ Saved: {path}\")\n\nsave_parquet(train_agg_ddf, TRAIN_OUT)\nsave_parquet(test_agg_ddf, TEST_OUT)\n\ntrain_agg = dd.read_parquet(TRAIN_OUT).compute()\n\nid_col_in_parquet = [c for c in train_agg.columns if \"customer\" in c.lower()][0]\nprint(\"ID column in parquet:\", id_col_in_parquet)\n\nlabels = pd.read_csv(LABELS_PATH)\n\ntrain_final = train_agg.merge(labels, left_on=id_col_in_parquet, right_on=ID_COL, how=\"inner\")\n\nfloat_cols = train_final.select_dtypes(include=[\"float64\"]).columns\ntrain_final[float_cols] = train_final[float_cols].astype(\"float32\")\n\ntrain_final.to_parquet(\n    FINAL_TRAIN_OUT,\n    engine=\"pyarrow\",\n    compression=\"zstd\",\n    index=False\n)\n\nprint(\"Train final shape:\", train_final.shape)\nprint(f\"✔ Final train saved: {FINAL_TRAIN_OUT}\")\nprint(f\"✔ Test agg saved: {TEST_OUT}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T23:29:00.572515Z","iopub.execute_input":"2025-09-29T23:29:00.572798Z","iopub.status.idle":"2025-09-29T23:37:33.304967Z","shell.execute_reply.started":"2025-09-29T23:29:00.572780Z","shell.execute_reply":"2025-09-29T23:37:33.298219Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(train_agg.columns[:20])   \nprint(train_agg.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T23:20:39.236664Z","iopub.execute_input":"2025-09-29T23:20:39.237035Z","iopub.status.idle":"2025-09-29T23:20:39.257316Z","shell.execute_reply.started":"2025-09-29T23:20:39.237014Z","shell.execute_reply":"2025-09-29T23:20:39.252703Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(os.path.getsize(TRAIN_OUT) / 1e9, \"GB\")\nprint(os.path.getsize(FINAL_TRAIN_OUT) / 1e9, \"GB\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = pd.read_parquet(TRAIN_OUT)\ntest  = pd.read_parquet(FINAL_TRAIN_OUT)\n\nprint(train.shape)\nprint(test.shape)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T23:40:41.545396Z","iopub.execute_input":"2025-09-29T23:40:41.545888Z","iopub.status.idle":"2025-09-29T23:41:15.468249Z","shell.execute_reply.started":"2025-09-29T23:40:41.545869Z","shell.execute_reply":"2025-09-29T23:41:15.461788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(os.listdir(\".\"))  \nprint(os.listdir(\"/kaggle/working\"))  ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-29T22:17:28.071613Z","iopub.execute_input":"2025-09-29T22:17:28.071920Z","iopub.status.idle":"2025-09-29T22:17:28.080886Z","shell.execute_reply.started":"2025-09-29T22:17:28.071899Z","shell.execute_reply":"2025-09-29T22:17:28.077193Z"}},"outputs":[],"execution_count":null}]}