{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom transformers import AutoTokenizer, AutoModelForSequenceClassification\nfrom fastai.text.all import *\nfrom sklearn.model_selection import train_test_split\n\nfrom torch.utils.data import Dataset","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:19:22.968156Z","iopub.execute_input":"2021-10-06T08:19:22.968588Z","iopub.status.idle":"2021-10-06T08:19:27.53071Z","shell.execute_reply.started":"2021-10-06T08:19:22.968502Z","shell.execute_reply":"2021-10-06T08:19:27.529885Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/quora-insincere-questions-classification/\"\ntrain_df = pd.read_csv(path + \"train.csv\")\ntest_df = pd.read_csv(path + \"test.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:19:27.532199Z","iopub.execute_input":"2021-10-06T08:19:27.532552Z","iopub.status.idle":"2021-10-06T08:19:32.073817Z","shell.execute_reply.started":"2021-10-06T08:19:27.532513Z","shell.execute_reply":"2021-10-06T08:19:32.072979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:19:32.075735Z","iopub.execute_input":"2021-10-06T08:19:32.076096Z","iopub.status.idle":"2021-10-06T08:19:32.097135Z","shell.execute_reply.started":"2021-10-06T08:19:32.076059Z","shell.execute_reply":"2021-10-06T08:19:32.096069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape, test_df.shape","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:19:32.098968Z","iopub.execute_input":"2021-10-06T08:19:32.099338Z","iopub.status.idle":"2021-10-06T08:19:32.105138Z","shell.execute_reply.started":"2021-10-06T08:19:32.099277Z","shell.execute_reply":"2021-10-06T08:19:32.104166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"target\"].value_counts()/train_df.shape[0]","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:19:32.106831Z","iopub.execute_input":"2021-10-06T08:19:32.107268Z","iopub.status.idle":"2021-10-06T08:19:32.130781Z","shell.execute_reply.started":"2021-10-06T08:19:32.107232Z","shell.execute_reply":"2021-10-06T08:19:32.129811Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_text = pd.concat([train_df[\"question_text\"], test_df[\"question_text\"]], axis=0)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:19:32.132244Z","iopub.execute_input":"2021-10-06T08:19:32.132641Z","iopub.status.idle":"2021-10-06T08:19:32.179699Z","shell.execute_reply.started":"2021-10-06T08:19:32.132606Z","shell.execute_reply":"2021-10-06T08:19:32.178879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sample_text(n=10):\n    sample = all_text.sample(n)\n    print(\" | \".join(sample))","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:19:32.181002Z","iopub.execute_input":"2021-10-06T08:19:32.181427Z","iopub.status.idle":"2021-10-06T08:19:32.185902Z","shell.execute_reply.started":"2021-10-06T08:19:32.181386Z","shell.execute_reply":"2021-10-06T08:19:32.185092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_text()","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:19:32.187381Z","iopub.execute_input":"2021-10-06T08:19:32.188071Z","iopub.status.idle":"2021-10-06T08:19:32.234107Z","shell.execute_reply.started":"2021-10-06T08:19:32.188029Z","shell.execute_reply":"2021-10-06T08:19:32.233158Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"question_text\"].apply(lambda x:len(x.split())).plot(kind=\"hist\");","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:19:32.23702Z","iopub.execute_input":"2021-10-06T08:19:32.237331Z","iopub.status.idle":"2021-10-06T08:19:34.154255Z","shell.execute_reply.started":"2021-10-06T08:19:32.237299Z","shell.execute_reply":"2021-10-06T08:19:34.153495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained('bert-base-cased')","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:19:34.156021Z","iopub.execute_input":"2021-10-06T08:19:34.156371Z","iopub.status.idle":"2021-10-06T08:19:38.324645Z","shell.execute_reply.started":"2021-10-06T08:19:34.156337Z","shell.execute_reply":"2021-10-06T08:19:38.323883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class QuestionDataset(Dataset):\n    def __init__(self, X, y, tokenizer):\n        self.text = X.reset_index(drop=True)\n        self.targets = y.reset_index(drop=True)\n        self.tok = tokenizer\n    \n    def __len__(self):\n        return len(self.text)\n    \n    def __getitem__(self, idx):\n        \n        text = self.text[idx]\n        targ = self.targets[idx]\n        \n        return self.tok(text, padding='max_length', \n                        truncation=True,\n                        max_length=30,\n                        return_tensors=\"pt\")[\"input_ids\"][0], tensor(targ)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:19:38.325885Z","iopub.execute_input":"2021-10-06T08:19:38.326243Z","iopub.status.idle":"2021-10-06T08:19:38.333143Z","shell.execute_reply.started":"2021-10-06T08:19:38.326206Z","shell.execute_reply":"2021-10-06T08:19:38.332113Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = train_df\nX_train, X_valid, y_train, y_valid = train_test_split(df[\"question_text\"], df[\"target\"], \n                                                      stratify=df[\"target\"],  test_size=0.01)\n\ntrain_ds = QuestionDataset(X_train, y_train, tokenizer)\nvalid_ds = QuestionDataset(X_valid, y_valid, tokenizer)\n\ntrain_dl = DataLoader(train_ds, bs=256)\nvalid_dl = DataLoader(valid_ds, bs=512)\ndls = DataLoaders(train_dl, valid_dl).to(\"cuda\")","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:19:38.334737Z","iopub.execute_input":"2021-10-06T08:19:38.335169Z","iopub.status.idle":"2021-10-06T08:19:39.598265Z","shell.execute_reply.started":"2021-10-06T08:19:38.335134Z","shell.execute_reply":"2021-10-06T08:19:39.597366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds.__getitem__(56)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:19:39.599648Z","iopub.execute_input":"2021-10-06T08:19:39.599981Z","iopub.status.idle":"2021-10-06T08:19:39.636961Z","shell.execute_reply.started":"2021-10-06T08:19:39.599947Z","shell.execute_reply":"2021-10-06T08:19:39.636187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bert = AutoModelForSequenceClassification.from_pretrained('bert-base-cased').train()\n\nclassifier = nn.Sequential(\n    nn.Linear(768, 1024),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(1024, 2)\n)\n\nbert.classifier = classifier\n\nclass BertClassifier(Module):\n    def __init__(self, bert):\n        self.bert = bert\n    def forward(self, x):\n        x = self.bert(x)\n        return x.logits\n\nmodel = BertClassifier(bert).to(\"cuda\")","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:19:39.638327Z","iopub.execute_input":"2021-10-06T08:19:39.638651Z","iopub.status.idle":"2021-10-06T08:20:10.044068Z","shell.execute_reply.started":"2021-10-06T08:19:39.638618Z","shell.execute_reply":"2021-10-06T08:20:10.043217Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"done\")","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:20:10.045437Z","iopub.execute_input":"2021-10-06T08:20:10.045775Z","iopub.status.idle":"2021-10-06T08:20:10.052222Z","shell.execute_reply.started":"2021-10-06T08:20:10.045741Z","shell.execute_reply":"2021-10-06T08:20:10.049918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_0 = (train_df[\"target\"] == 0).sum()\nn_1 = (train_df[\"target\"] == 1).sum()\nn = n_0 + n_1","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:20:10.053658Z","iopub.execute_input":"2021-10-06T08:20:10.054023Z","iopub.status.idle":"2021-10-06T08:20:10.071951Z","shell.execute_reply.started":"2021-10-06T08:20:10.053988Z","shell.execute_reply":"2021-10-06T08:20:10.07124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_weights = tensor([n / (n+n_0), n / (n+n_1)]).to('cuda')\nlearn = Learner(dls, model, \n                loss_func=nn.CrossEntropyLoss(weight=class_weights), \n                metrics=[accuracy, F1Score()]).to_fp16()\nlearn.lr_find()","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:20:10.0731Z","iopub.execute_input":"2021-10-06T08:20:10.073462Z","iopub.status.idle":"2021-10-06T08:21:06.731226Z","shell.execute_reply.started":"2021-10-06T08:20:10.073428Z","shell.execute_reply":"2021-10-06T08:21:06.730468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fit_one_cycle(2, lr_max=0.001318)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T08:21:06.732522Z","iopub.execute_input":"2021-10-06T08:21:06.732848Z","iopub.status.idle":"2021-10-06T10:34:06.855423Z","shell.execute_reply.started":"2021-10-06T08:21:06.732811Z","shell.execute_reply":"2021-10-06T10:34:06.85449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score","metadata":{"execution":{"iopub.status.busy":"2021-10-06T10:34:06.856743Z","iopub.execute_input":"2021-10-06T10:34:06.857087Z","iopub.status.idle":"2021-10-06T10:34:06.862905Z","shell.execute_reply.started":"2021-10-06T10:34:06.857051Z","shell.execute_reply":"2021-10-06T10:34:06.862174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds, targs = learn.get_preds()","metadata":{"execution":{"iopub.status.busy":"2021-10-06T10:34:06.864135Z","iopub.execute_input":"2021-10-06T10:34:06.864516Z","iopub.status.idle":"2021-10-06T10:34:22.20042Z","shell.execute_reply.started":"2021-10-06T10:34:06.86448Z","shell.execute_reply":"2021-10-06T10:34:22.19965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"thresholds = np.linspace(0.3, 0.7, 50)\nfor threshold in thresholds:\n    f1 = f1_score(targs, F.softmax(preds, dim=1)[:, 1]>threshold)\n    print(f\"threshold:{threshold:.4f} - f1:{f1:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2021-10-06T10:34:22.201681Z","iopub.execute_input":"2021-10-06T10:34:22.202036Z","iopub.status.idle":"2021-10-06T10:34:22.46715Z","shell.execute_reply.started":"2021-10-06T10:34:22.201996Z","shell.execute_reply":"2021-10-06T10:34:22.466413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tensor = tokenizer(list(test_df[\"question_text\"]),\n                        padding=\"max_length\",\n                        truncation=True,\n                        max_length=30,\n                        return_tensors=\"pt\")[\"input_ids\"]","metadata":{"execution":{"iopub.status.busy":"2021-10-06T10:34:22.469817Z","iopub.execute_input":"2021-10-06T10:34:22.470074Z","iopub.status.idle":"2021-10-06T10:34:50.791741Z","shell.execute_reply.started":"2021-10-06T10:34:22.470048Z","shell.execute_reply":"2021-10-06T10:34:50.790665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TestDS:\n    def __init__(self, tensors):\n        self.tensors = tensors\n    \n    def __len__(self):\n        return len(self.tensors)\n    \n    def __getitem__(self, idx):\n        t = self.tensors[idx]\n        return t, tensor(0)\n\ntest_dl = DataLoader(TestDS(test_tensor), bs=128)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T10:34:50.793581Z","iopub.execute_input":"2021-10-06T10:34:50.793952Z","iopub.status.idle":"2021-10-06T10:34:50.800592Z","shell.execute_reply.started":"2021-10-06T10:34:50.793904Z","shell.execute_reply":"2021-10-06T10:34:50.798858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds = learn.get_preds(dl=test_dl)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T10:34:50.801992Z","iopub.execute_input":"2021-10-06T10:34:50.80241Z","iopub.status.idle":"2021-10-06T10:41:25.85198Z","shell.execute_reply.started":"2021-10-06T10:34:50.802368Z","shell.execute_reply":"2021-10-06T10:41:25.851179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = (F.softmax(test_preds[0], dim=1)[:, 1]>0.48).int()\nsub = pd.read_csv(path + \"sample_submission.csv\")\nsub[\"prediction\"] = prediction\nsub.to_csv(\"./submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-06T10:58:16.712641Z","iopub.execute_input":"2021-10-06T10:58:16.712971Z","iopub.status.idle":"2021-10-06T10:58:18.744029Z","shell.execute_reply.started":"2021-10-06T10:58:16.71294Z","shell.execute_reply":"2021-10-06T10:58:18.743177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}