{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Solving Amex 💳 dataset with Lightning⚡Flash\n\nFlash makes complex AI recipes for over 15 tasks across 7 data domains accessible to all.\nIn a nutshell, Flash is the production grade research framework you always dreamed of but didn't have time to build.\n\nhttps://github.com/PyTorchLightning/lightning-flash","metadata":{"papermill":{"duration":0.015913,"end_time":"2022-06-03T09:47:24.889771","exception":false,"start_time":"2022-06-03T09:47:24.873858","status":"completed"},"tags":[]}},{"cell_type":"code","source":"! pip install -q \"pytorch-lightning>1.5\" lightning-flash[tabular] \"omegaconf==2.1.*\"\n! pip install -q 'https://github.com/PyTorchLightning/lightning-flash/archive/refs/heads/tabular/mean-std.zip#egg=lightning-flash[tabular]'\n! pip install -q \"matplotlib==3.1.1\" \"pandas==1.3.5\" --force-reinstall\n! pip uninstall -y torchtext\n! pip list | grep -e lightning -e torch -e tab","metadata":{"_kg_hide-output":true,"papermill":{"duration":84.268152,"end_time":"2022-06-03T09:51:37.051319","exception":false,"start_time":"2022-06-03T09:50:12.783167","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-09T23:47:15.931998Z","iopub.execute_input":"2022-06-09T23:47:15.934300Z","iopub.status.idle":"2022-06-09T23:49:32.695758Z","shell.execute_reply.started":"2022-06-09T23:47:15.932519Z","shell.execute_reply":"2022-06-09T23:49:32.691957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\n\nimport torch\nimport flash\nimport numpy as np\nimport pandas as pd\nimport dask.dataframe as dd\nfrom pprint import pprint\nfrom flash.tabular import TabularClassificationData, TabularClassifier","metadata":{"papermill":{"duration":11.010583,"end_time":"2022-06-03T09:51:48.08062","exception":false,"start_time":"2022-06-03T09:51:37.070037","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-08T10:50:12.043009Z","iopub.execute_input":"2022-06-08T10:50:12.043353Z","iopub.status.idle":"2022-06-08T10:50:23.249644Z","shell.execute_reply.started":"2022-06-08T10:50:12.043315Z","shell.execute_reply":"2022-06-08T10:50:23.248685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the dataset 🔎","metadata":{}},{"cell_type":"code","source":"df_labels = dd.read_csv(\"../input/amex-default-prediction/train_labels.csv\", dtype={\"target\": np.int16}).set_index('customer_ID')\ndisplay(df_labels.head())\nprint(len(df_labels))\ndf_labels[\"target\"].value_counts().compute().plot.pie()","metadata":{"execution":{"iopub.status.busy":"2022-06-08T10:50:23.250994Z","iopub.execute_input":"2022-06-08T10:50:23.252032Z","iopub.status.idle":"2022-06-08T10:50:28.385824Z","shell.execute_reply.started":"2022-06-08T10:50:23.251994Z","shell.execute_reply":"2022-06-08T10:50:28.38474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lut_ids = {id_: i for i, id_ in enumerate(df_labels.index)}\ndf_labels.index = df_labels.index.map(lut_ids)\ndisplay(df_labels.head())","metadata":{"execution":{"iopub.status.busy":"2022-06-08T10:50:28.388795Z","iopub.execute_input":"2022-06-08T10:50:28.389662Z","iopub.status.idle":"2022-06-08T10:50:40.31382Z","shell.execute_reply.started":"2022-06-08T10:50:28.389612Z","shell.execute_reply":"2022-06-08T10:50:40.312195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Load huge DataFrames\n\nwith the default Float64 it does not fit to memory so we lower the precision to Float16","metadata":{}},{"cell_type":"code","source":"with open(\"../input/amex-default-prediction/train_data.csv\") as fp:\n    head = fp.readline().strip().split(\",\")\n    # pprint(dict(zip(head, fp.readline().strip().split(\",\"))))\nprint(head)\ncol_dtypes = {c: np.float16 for c in head if c not in [\"customer_ID\", \"S_2\", \"D_63\", \"D_64\"]}","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-06-08T10:50:40.316784Z","iopub.execute_input":"2022-06-08T10:50:40.317455Z","iopub.status.idle":"2022-06-08T10:50:40.332186Z","shell.execute_reply.started":"2022-06-08T10:50:40.317405Z","shell.execute_reply":"2022-06-08T10:50:40.331315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = dd.read_csv(\"../input/amex-default-prediction/train_data.csv\", dtype=col_dtypes).set_index('customer_ID')\ndf_train.index = df_train.index.map(lut_ids)\ndisplay(df_train.head())\nprint(len(df_train))","metadata":{"papermill":{"duration":0.311232,"end_time":"2022-06-03T09:51:48.44692","exception":false,"start_time":"2022-06-03T09:51:48.135688","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-08T10:50:40.333587Z","iopub.execute_input":"2022-06-08T10:50:40.334208Z","iopub.status.idle":"2022-06-08T11:06:19.837962Z","shell.execute_reply.started":"2022-06-08T10:50:40.334173Z","shell.execute_reply":"2022-06-08T11:06:19.835549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Merge training data with targets","metadata":{}},{"cell_type":"code","source":"df_train = df_train.merge(df_labels, left_index=True, right_index=True).replace([np.inf, -np.inf], np.nan)\ndisplay(df_train.head())\nprint(len(df_train))\nprint(len(df_train.index.unique().compute()))\n\ndel df_labels","metadata":{"execution":{"iopub.status.busy":"2022-06-08T11:06:19.845088Z","iopub.execute_input":"2022-06-08T11:06:19.845685Z","iopub.status.idle":"2022-06-08T11:25:53.961422Z","shell.execute_reply.started":"2022-06-08T11:06:19.845639Z","shell.execute_reply":"2022-06-08T11:25:53.958505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col_counts = df_train.count().compute()\ncol_counts.sort_values().plot.bar(figsize=(14, 2), grid=True)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T11:25:53.967889Z","iopub.execute_input":"2022-06-08T11:25:53.968556Z","iopub.status.idle":"2022-06-08T11:34:06.297812Z","shell.execute_reply.started":"2022-06-08T11:25:53.968504Z","shell.execute_reply":"2022-06-08T11:34:06.295783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.index.value_counts().compute().hist(bins=50)","metadata":{"execution":{"iopub.status.busy":"2022-06-08T11:34:06.302722Z","iopub.execute_input":"2022-06-08T11:34:06.303307Z","iopub.status.idle":"2022-06-08T11:42:20.181818Z","shell.execute_reply.started":"2022-06-08T11:34:06.303236Z","shell.execute_reply":"2022-06-08T11:42:20.17922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train[\"S_2\"].value_counts().compute().plot.bar(figsize=(24, 3))","metadata":{"execution":{"iopub.status.busy":"2022-06-08T11:42:20.19015Z","iopub.execute_input":"2022-06-08T11:42:20.190785Z","iopub.status.idle":"2022-06-08T11:50:55.602855Z","shell.execute_reply.started":"2022-06-08T11:42:20.190745Z","shell.execute_reply":"2022-06-08T11:50:55.600686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Limit thre training dataset","metadata":{}},{"cell_type":"code","source":"# ToDo: take only fraction of the training data dues to HW limitations\nprint(f\"table size: {len(df_train)}\")\n# display(df_train.head())\ndf_train = df_train.sample(0.4).compute()\nprint(f\"table size: {len(df_train)}\")","metadata":{"execution":{"iopub.status.busy":"2022-06-08T11:50:55.607904Z","iopub.execute_input":"2022-06-08T11:50:55.60849Z","iopub.status.idle":"2022-06-08T12:07:19.979808Z","shell.execute_reply.started":"2022-06-08T11:50:55.608431Z","shell.execute_reply":"2022-06-08T12:07:19.974645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Compute some stat parameters\n\nTurned out that mean nor STD canot be succesfully computed with float16","metadata":{}},{"cell_type":"code","source":"# params = {\n#     'mean': {c: np.nanmean(df_train[c], dtype=np.float32) for c in useful_cols},\n#     'std': {c: np.nanstd(df_train[c], dtype=np.float32) for c in useful_cols},\n#     # 'codes': {},\n#     # 'numerical_fields': useful_cols,\n#     # 'categorical_fields': [],\n# }\n# print(params)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-06-08T12:07:19.985207Z","iopub.execute_input":"2022-06-08T12:07:19.987627Z","iopub.status.idle":"2022-06-08T12:07:19.995686Z","shell.execute_reply.started":"2022-06-08T12:07:19.987576Z","shell.execute_reply":"2022-06-08T12:07:19.994603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training the task with Lightning⚡Flash\n\n## 1. Create the DataModule","metadata":{"papermill":{"duration":0.018697,"end_time":"2022-06-03T09:51:48.117373","exception":false,"start_time":"2022-06-03T09:51:48.098676","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# this is given by organizers\ncategorical_cols = ['B_30', 'B_38', 'D_114', 'D_116', 'D_117', 'D_120', 'D_126', 'D_63', 'D_64', 'D_66', 'D_68']\n\n# set the remaining as nuerical\nnumerical_cols = [c for c in col_dtypes if c in df_train.columns and c not in categorical_cols]","metadata":{"execution":{"iopub.status.busy":"2022-06-08T12:07:19.997043Z","iopub.execute_input":"2022-06-08T12:07:19.998045Z","iopub.status.idle":"2022-06-08T12:07:20.037649Z","shell.execute_reply.started":"2022-06-08T12:07:19.998007Z","shell.execute_reply":"2022-06-08T12:07:20.036549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"datamodule = TabularClassificationData.from_data_frame(\n    categorical_fields=categorical_cols,\n    numerical_fields=numerical_cols,\n    target_fields=\"target\",\n    train_data_frame=df_train.fillna(0),\n    val_split=0.1,\n    batch_size=512,\n    # predict_data_frame=df_test.fillna(0),\n)\n\npprint(datamodule.parameters)","metadata":{"papermill":{"duration":0.086844,"end_time":"2022-06-03T09:51:48.625078","exception":false,"start_time":"2022-06-03T09:51:48.538234","status":"completed"},"tags":[],"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-06-08T12:07:20.039588Z","iopub.execute_input":"2022-06-08T12:07:20.040222Z","iopub.status.idle":"2022-06-08T12:09:11.459078Z","shell.execute_reply.started":"2022-06-08T12:07:20.040138Z","shell.execute_reply":"2022-06-08T12:09:11.457969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2. Build the task","metadata":{"papermill":{"duration":0.019957,"end_time":"2022-06-03T09:51:48.666541","exception":false,"start_time":"2022-06-03T09:51:48.646584","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# from torchmetrics import F1\n\nmodel = TabularClassifier.from_data(\n    datamodule,\n    # backbone=\"tabnet\",\n    backbone=\"tabtransformer\",\n#     metrics=F1(),\n    optimizer=\"Adamax\",\n    learning_rate=0.1,\n    lr_scheduler=(\"StepLR\", {\"step_size\": 7500}),\n)","metadata":{"papermill":{"duration":0.353557,"end_time":"2022-06-03T09:51:49.042174","exception":false,"start_time":"2022-06-03T09:51:48.688617","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-08T13:28:32.013135Z","iopub.execute_input":"2022-06-08T13:28:32.014793Z","iopub.status.idle":"2022-06-08T13:28:32.412695Z","shell.execute_reply.started":"2022-06-08T13:28:32.014681Z","shell.execute_reply":"2022-06-08T13:28:32.409429Z"}},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3. Create the trainer and train the model","metadata":{"papermill":{"duration":0.026283,"end_time":"2022-06-03T09:51:49.100961","exception":false,"start_time":"2022-06-03T09:51:49.074678","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from pytorch_lightning.loggers import CSVLogger\n# from pytorch_lightning.callbacks import StochasticWeightAveraging\nfrom pytorch_lightning import seed_everything\n\nseed_everything(7)\n# swa = StochasticWeightAveraging(swa_epoch_start=0.6)\ntrainer = flash.Trainer(\n    max_epochs=15,\n    #callbacks=[swa],\n    gpus=torch.cuda.device_count(),\n    logger=CSVLogger(save_dir='logs/'),\n    accumulate_grad_batches=24,\n    # gradient_clip_val=0.1,\n    val_check_interval=0.25\n)","metadata":{"_kg_hide-output":true,"papermill":{"duration":0.061127,"end_time":"2022-06-03T09:51:49.193656","exception":false,"start_time":"2022-06-03T09:51:49.132529","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-08T13:28:32.420357Z","iopub.execute_input":"2022-06-08T13:28:32.422452Z","iopub.status.idle":"2022-06-08T13:28:32.477682Z","shell.execute_reply.started":"2022-06-08T13:28:32.422399Z","shell.execute_reply":"2022-06-08T13:28:32.473994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer.fit(model, datamodule=datamodule)","metadata":{"_kg_hide-output":true,"papermill":{"duration":108.869732,"end_time":"2022-06-03T09:53:38.095168","exception":false,"start_time":"2022-06-03T09:51:49.225436","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-08T13:28:32.481986Z","iopub.execute_input":"2022-06-08T13:28:32.48535Z","iopub.status.idle":"2022-06-08T14:25:57.301851Z","shell.execute_reply.started":"2022-06-08T13:28:32.485292Z","shell.execute_reply":"2022-06-08T14:25:57.300594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set()\n\nmetrics = pd.read_csv(f'{trainer.logger.log_dir}/metrics.csv')\n# display(metrics.head())\nmetrics.set_index(\"step\", inplace=True)\ndel metrics[\"epoch\"]\nsns.relplot(data=metrics, kind=\"line\")\nplt.gca().set_ylim([0, 1.25])\nplt.gcf().set_size_inches(10, 5)","metadata":{"papermill":{"duration":0.815914,"end_time":"2022-06-03T09:53:38.970966","exception":false,"start_time":"2022-06-03T09:53:38.155052","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-08T14:25:57.306658Z","iopub.execute_input":"2022-06-08T14:25:57.307102Z","iopub.status.idle":"2022-06-08T14:25:58.19751Z","shell.execute_reply.started":"2022-06-08T14:25:57.307065Z","shell.execute_reply":"2022-06-08T14:25:58.196563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del df_train\n\nparams = dict(datamodule.parameters)\ndel datamodule","metadata":{"execution":{"iopub.status.busy":"2022-06-08T14:25:58.198844Z","iopub.execute_input":"2022-06-08T14:25:58.19921Z","iopub.status.idle":"2022-06-08T14:25:59.327616Z","shell.execute_reply.started":"2022-06-08T14:25:58.199178Z","shell.execute_reply":"2022-06-08T14:25:59.326608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 4. Generate predictions from a CSV","metadata":{"papermill":{"duration":0.060885,"end_time":"2022-06-03T09:53:39.090663","exception":false,"start_time":"2022-06-03T09:53:39.029778","status":"completed"},"tags":[]}},{"cell_type":"code","source":"! head ../input/amex-default-prediction/sample_submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-06-08T17:18:01.200476Z","iopub.execute_input":"2022-06-08T17:18:01.201139Z","iopub.status.idle":"2022-06-08T17:18:02.112712Z","shell.execute_reply.started":"2022-06-08T17:18:01.201094Z","shell.execute_reply":"2022-06-08T17:18:02.111362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir -p /kaggle/temp\n\nfrom tqdm.auto import tqdm\n\nhead = None\nlines = []\ncounter = 0\npbar = tqdm(desc=\"Exported CSV tables\")\nwith open(\"../input/amex-default-prediction/test_data.csv\") as fp:\n    for line in fp:\n        if not head:\n            head = line\n        else:\n            lines.append(line)\n        if len(lines) < 100_000:\n            continue\n        with open(f\"/kaggle/temp/test_data_{counter}.csv\", \"w\") as fpp:\n            fpp.writelines([head] + lines)\n        lines = []\n        counter += 1\n        pbar.update()\n\nwith open(f\"/kaggle/temp/test_data_{counter}.csv\", \"w\") as fpp:\n    fpp.writelines([head] + lines)\n\n!ls -l /kaggle/temp/test_data_*.csv","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-06-08T14:25:59.328805Z","iopub.execute_input":"2022-06-08T14:25:59.32976Z","iopub.status.idle":"2022-06-08T14:36:39.787923Z","shell.execute_reply.started":"2022-06-08T14:25:59.329719Z","shell.execute_reply":"2022-06-08T14:36:39.785986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import glob\n\ntest_files = sorted(glob.glob(\"/kaggle/temp/test_data_*.csv\"))\n\nindexes, predictions = [], []\nfor tfile in tqdm(test_files, desc=\"Iterate over Test fractions\"):\n    df_test = pd.read_csv(tfile, dtype=col_dtypes).set_index('customer_ID')\n    indexes += list(df_test.index)\n    datamodule = TabularClassificationData.from_data_frame(\n        parameters=params,\n        batch_size=64,\n        predict_data_frame=df_test.fillna(0),\n    )\n    predictions += trainer.predict(model, datamodule=datamodule, output=\"classes\")","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-06-08T14:36:39.790722Z","iopub.execute_input":"2022-06-08T14:36:39.791187Z","iopub.status.idle":"2022-06-08T15:54:35.169604Z","shell.execute_reply.started":"2022-06-08T14:36:39.79115Z","shell.execute_reply":"2022-06-08T15:54:35.168366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from itertools import chain\n\ndf_preds = pd.DataFrame({\n    \"customer_ID\": indexes,\n    \"prediction\": list(chain(*predictions))\n})\ndf_preds[\"prediction\"].value_counts().plot.pie()","metadata":{"papermill":{"duration":0.370996,"end_time":"2022-06-03T09:53:39.99933","exception":false,"start_time":"2022-06-03T09:53:39.628334","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-08T17:11:15.097917Z","iopub.execute_input":"2022-06-08T17:11:15.098531Z","iopub.status.idle":"2022-06-08T17:11:21.96211Z","shell.execute_reply.started":"2022-06-08T17:11:15.098486Z","shell.execute_reply":"2022-06-08T17:11:21.960578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_preds_short = df_preds.groupby(\"customer_ID\").median()\ndisplay(df_preds_short.head())\ndf_preds_short[\"prediction\"].value_counts().plot.pie()\n\nprint(len(df_preds_short))\ndf_preds_short[[\"prediction\"]].to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-06-08T17:16:50.380193Z","iopub.execute_input":"2022-06-08T17:16:50.381427Z","iopub.status.idle":"2022-06-08T17:16:57.757718Z","shell.execute_reply.started":"2022-06-08T17:16:50.381377Z","shell.execute_reply":"2022-06-08T17:16:57.756593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! head submission.csv","metadata":{"papermill":{"duration":0.869427,"end_time":"2022-06-03T09:53:40.928218","exception":false,"start_time":"2022-06-03T09:53:40.058791","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-06-08T17:16:57.759834Z","iopub.execute_input":"2022-06-08T17:16:57.760596Z","iopub.status.idle":"2022-06-08T17:16:58.748725Z","shell.execute_reply.started":"2022-06-08T17:16:57.760545Z","shell.execute_reply":"2022-06-08T17:16:58.746782Z"},"trusted":true},"execution_count":null,"outputs":[]}]}