{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# 🙊Toxic comments with Lightning⚡Flash\n\n[Flash](https://lightning-flash.readthedocs.io/en/stable) makes complex AI recipes for over 15 tasks across 7 data domains accessible to all.\n\nIn a nutshell, Flash is the production grade research framework you always dreamed of but didn't have time to build.","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"# ! pip install -q lightning-flash[text]\n# ! pip install -q 'https://github.com/PyTorchLightning/lightning-flash/archive/refs/heads/master.zip#egg=lightning-flash[text]'\n! pip install -q 'https://github.com/PyTorchLightning/lightning-flash/archive/refs/heads/fix/serialize_tokenizer.zip#egg=lightning-flash[text]'\n! pip install -q mplfinance\n! pip install -q --upgrade pandas --force-reinstall\n! pip list | grep -E \"lightning|torch\"","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-12-02T12:45:44.250305Z","iopub.execute_input":"2021-12-02T12:45:44.250657Z","iopub.status.idle":"2021-12-02T12:47:27.736326Z","shell.execute_reply.started":"2021-12-02T12:45:44.250572Z","shell.execute_reply":"2021-12-02T12:47:27.734832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ! pip download -q lightning-flash[text] --prefer-binary --dest frozen_packages\n! pip wheel -q 'https://github.com/PyTorchLightning/lightning-flash/archive/refs/heads/fix/serialize_tokenizer.zip#egg=lightning-flash[text]' --wheel-dir frozen_packages\n! rm frozen_packages/torch-*\n! ls frozen_packages","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-12-02T12:47:27.739222Z","iopub.execute_input":"2021-12-02T12:47:27.739908Z","iopub.status.idle":"2021-12-02T12:48:58.804851Z","shell.execute_reply.started":"2021-12-02T12:47:27.739859Z","shell.execute_reply":"2021-12-02T12:48:58.803710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! ls /kaggle/input/jigsaw-multilingual-toxic-comment-classification","metadata":{"execution":{"iopub.status.busy":"2021-12-02T12:48:58.808270Z","iopub.execute_input":"2021-12-02T12:48:58.808940Z","iopub.status.idle":"2021-12-02T12:48:59.562637Z","shell.execute_reply.started":"2021-12-02T12:48:58.808891Z","shell.execute_reply":"2021-12-02T12:48:59.561309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data exolorations & preparation\n\nChecking the input data and pairing with Crypto names","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ncsv_train = \"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\"\ndf_train = pd.read_csv(csv_train)\ndisplay(df_train.head())\n\n_= df_train.plot.hist(bins=2, grid=True, sharex=True, logy=True)","metadata":{"execution":{"iopub.status.busy":"2021-12-02T12:48:59.566130Z","iopub.execute_input":"2021-12-02T12:48:59.566716Z","iopub.status.idle":"2021-12-02T12:49:03.366858Z","shell.execute_reply.started":"2021-12-02T12:48:59.566667Z","shell.execute_reply":"2021-12-02T12:49:03.365831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"csv_comemnts = \"/kaggle/input/jigsaw-toxic-severity-rating/comments_to_score.csv\"\ndf_comments = pd.read_csv(csv_comemnts)\ndisplay(df_comments.head())","metadata":{"execution":{"iopub.status.busy":"2021-12-02T12:49:03.368530Z","iopub.execute_input":"2021-12-02T12:49:03.368842Z","iopub.status.idle":"2021-12-02T12:49:03.464062Z","shell.execute_reply.started":"2021-12-02T12:49:03.368798Z","shell.execute_reply":"2021-12-02T12:49:03.462882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### ToDo\n\nConsider some label aggregation for this competition...","metadata":{}},{"cell_type":"code","source":"df_train[\"sum\"] = df_train[[\"toxic\", \"severe_toxic\", \"obscene\", \"threat\", \"insult\", \"identity_hate\"]].sum(axis=1)\ndf_train[\"any\"] = df_train[\"sum\"].gt(0).astype(int)\n_= df_train[\"any\"].plot.hist(bins=2, grid=True)","metadata":{"execution":{"iopub.status.busy":"2021-12-02T12:49:03.465563Z","iopub.execute_input":"2021-12-02T12:49:03.467127Z","iopub.status.idle":"2021-12-02T12:49:03.762327Z","shell.execute_reply.started":"2021-12-02T12:49:03.467060Z","shell.execute_reply":"2021-12-02T12:49:03.761331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training with Flash Lightning\n\nSee the classification docs: https://lightning-flash.readthedocs.io/en/stable/reference/text_classification.html","metadata":{}},{"cell_type":"code","source":"import torch\n\nimport flash\nfrom flash.text import TextClassificationData, TextClassifier","metadata":{"execution":{"iopub.status.busy":"2021-12-02T12:49:03.763894Z","iopub.execute_input":"2021-12-02T12:49:03.765087Z","iopub.status.idle":"2021-12-02T12:49:14.272401Z","shell.execute_reply.started":"2021-12-02T12:49:03.765034Z","shell.execute_reply":"2021-12-02T12:49:14.271020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 1. Create the DataModule","metadata":{}},{"cell_type":"code","source":"datamodule = TextClassificationData.from_data_frame(\n    input_field=\"comment_text\",\n    target_fields=\"any\",  # \"toxic\",\n    train_data_frame=df_train,\n    val_data_frame=df_train,\n    backbone=\"xlm-roberta-base\",\n    batch_size=64,\n    num_workers=0,\n)","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-12-02T12:49:14.274810Z","iopub.execute_input":"2021-12-02T12:49:14.275207Z","iopub.status.idle":"2021-12-02T12:54:06.014961Z","shell.execute_reply.started":"2021-12-02T12:49:14.275163Z","shell.execute_reply":"2021-12-02T12:54:06.014038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 2. Build the task","metadata":{}},{"cell_type":"code","source":"from torchmetrics import F1, Precision\n\nmodel = TextClassifier(\n    backbone=datamodule.backbone,\n    num_classes=datamodule.num_classes,\n    metrics=[Precision(), F1()],\n)\nmodel.model.save_pretrained(\"./used-HF-model\")\n! ls -l ./used-HF-model","metadata":{"execution":{"iopub.status.busy":"2021-12-02T12:54:06.017182Z","iopub.execute_input":"2021-12-02T12:54:06.017745Z","iopub.status.idle":"2021-12-02T12:55:00.121757Z","shell.execute_reply.started":"2021-12-02T12:54:06.017695Z","shell.execute_reply":"2021-12-02T12:55:00.120579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 3. Create the trainer and finetune the model","metadata":{}},{"cell_type":"code","source":"import torch\nfrom pytorch_lightning.loggers import CSVLogger\n# from pytorch_lightning.callbacks import StochasticWeightAveraging\n\n# swa = StochasticWeightAveraging(swa_epoch_start=0.6)\nlogger = CSVLogger(save_dir='logs/')\ntrainer = flash.Trainer(\n    max_epochs=10,\n    logger=logger,\n    gpus=torch.cuda.device_count(),\n    # callbacks=[swa],\n    accumulate_grad_batches=12,\n    gradient_clip_val=0.1,\n    precision=16,\n    # enable_ort=True,  # if you have PT>=1.5\n    auto_lr_find=True,\n)\n\ntrainer.tune(model, datamodule=datamodule, lr_find_kwargs=dict(min_lr=1e-5, max_lr=0.1, num_training=65),)\nprint(f\"Learning Rate: {model.learning_rate}\")\n\ntrainer.finetune(model, datamodule=datamodule, strategy=\"freeze\")\n\n# Save the model!\ntrainer.save_checkpoint(\"text_classification_model.pt\")","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-12-02T12:55:00.127013Z","iopub.execute_input":"2021-12-02T12:55:00.127333Z","iopub.status.idle":"2021-12-02T14:59:02.654839Z","shell.execute_reply.started":"2021-12-02T12:55:00.127299Z","shell.execute_reply":"2021-12-02T14:59:02.652571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set()\n\nmetrics = pd.read_csv(f'{trainer.logger.log_dir}/metrics.csv')\ndel metrics[\"step\"]\nmetrics.set_index(\"epoch\", inplace=True)\ndisplay(metrics.dropna(axis=1, how=\"all\").head())\ng = sns.relplot(data=metrics, kind=\"line\")\nplt.gcf().set_size_inches(15, 5)","metadata":{"execution":{"iopub.status.busy":"2021-12-02T14:59:02.658059Z","iopub.execute_input":"2021-12-02T14:59:02.659088Z","iopub.status.idle":"2021-12-02T14:59:05.619933Z","shell.execute_reply.started":"2021-12-02T14:59:02.659036Z","shell.execute_reply":"2021-12-02T14:59:05.618967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ! ls -l ~/.cache/huggingface/\n# ! mkdir -p cache/huggingface\n# ! rsync -ahv ~/.cache/huggingface/ cache/huggingface --exclude=\"*.lock\"\n# ! ls -l cache/huggingface","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-12-02T14:59:05.621730Z","iopub.execute_input":"2021-12-02T14:59:05.622394Z","iopub.status.idle":"2021-12-02T14:59:05.627253Z","shell.execute_reply.started":"2021-12-02T14:59:05.622342Z","shell.execute_reply":"2021-12-02T14:59:05.626160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### 4. Classify new comments","metadata":{}},{"cell_type":"code","source":"import math\nfrom flash.core.classification import Logits, Probabilities\nfrom tqdm.auto import tqdm\n\nmodel.output = Logits()\n# predictions = model.predict(df_comments[\"text\"])\n\npredictions = []\nfor i in tqdm(range(math.ceil(len(df_comments) / datamodule.batch_size))):\n    batch = df_comments[\"text\"][i * datamodule.batch_size:(i + 1) * datamodule.batch_size]\n    predictions += model.predict(batch)\n\nprint(f\"inputs={len(df_comments)} ; preds={len(predictions)}\")\nprint(predictions[0])","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-12-02T14:59:05.628798Z","iopub.execute_input":"2021-12-02T14:59:05.629432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\n\npredictions = np.array(predictions)[:, -1]\n_= plt.hist(predictions, bins=25)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submit = pd.DataFrame(zip(df_comments[\"comment_id\"], predictions), columns=(\"comment_id\", \"score\"))\ndf_submit.set_index(\"comment_id\", inplace=True)\ndf_submit.to_csv(\"submission.csv\")\n\n! head submission.csv","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}