{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nfrom transformers import AutoTokenizer, AutoModelForSequenceClassification\nfrom fastai.text.all import *\nfrom sklearn.model_selection import train_test_split\n\nfrom torch.utils.data import Dataset","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:34:21.225477Z","iopub.execute_input":"2021-06-09T07:34:21.225869Z","iopub.status.idle":"2021-06-09T07:34:21.231383Z","shell.execute_reply.started":"2021-06-09T07:34:21.225822Z","shell.execute_reply":"2021-06-09T07:34:21.23036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"/kaggle/input/quora-insincere-questions-classification/\"\ntrain_df = pd.read_csv(path + \"train.csv\")\ntest_df = pd.read_csv(path + \"test.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:34:21.233189Z","iopub.execute_input":"2021-06-09T07:34:21.23358Z","iopub.status.idle":"2021-06-09T07:34:23.95417Z","shell.execute_reply.started":"2021-06-09T07:34:21.233546Z","shell.execute_reply":"2021-06-09T07:34:23.953342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape, test_df.shape","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:34:23.957052Z","iopub.execute_input":"2021-06-09T07:34:23.95768Z","iopub.status.idle":"2021-06-09T07:34:23.963232Z","shell.execute_reply.started":"2021-06-09T07:34:23.95764Z","shell.execute_reply":"2021-06-09T07:34:23.962321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"target\"].value_counts()/train_df.shape[0]","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:34:23.964895Z","iopub.execute_input":"2021-06-09T07:34:23.965224Z","iopub.status.idle":"2021-06-09T07:34:23.985972Z","shell.execute_reply.started":"2021-06-09T07:34:23.965192Z","shell.execute_reply":"2021-06-09T07:34:23.985088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"all_text = pd.concat([train_df[\"question_text\"], test_df[\"question_text\"]], axis=0)","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:34:23.987154Z","iopub.execute_input":"2021-06-09T07:34:23.987514Z","iopub.status.idle":"2021-06-09T07:34:24.072233Z","shell.execute_reply.started":"2021-06-09T07:34:23.987478Z","shell.execute_reply":"2021-06-09T07:34:24.071408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sample_text(n=10):\n    sample = all_text.sample(n)\n    print(\" | \".join(sample))","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:34:24.073485Z","iopub.execute_input":"2021-06-09T07:34:24.073996Z","iopub.status.idle":"2021-06-09T07:34:24.078887Z","shell.execute_reply.started":"2021-06-09T07:34:24.073955Z","shell.execute_reply":"2021-06-09T07:34:24.077964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_text()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:34:24.080204Z","iopub.execute_input":"2021-06-09T07:34:24.080551Z","iopub.status.idle":"2021-06-09T07:34:24.13216Z","shell.execute_reply.started":"2021-06-09T07:34:24.080516Z","shell.execute_reply":"2021-06-09T07:34:24.131172Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"question_text\"].apply(lambda x:len(x.split())).plot(kind=\"hist\");","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:34:24.133522Z","iopub.execute_input":"2021-06-09T07:34:24.133882Z","iopub.status.idle":"2021-06-09T07:34:25.959314Z","shell.execute_reply.started":"2021-06-09T07:34:24.133845Z","shell.execute_reply":"2021-06-09T07:34:25.958261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained('bert-base-cased')","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:34:25.962323Z","iopub.execute_input":"2021-06-09T07:34:25.962798Z","iopub.status.idle":"2021-06-09T07:34:27.297881Z","shell.execute_reply.started":"2021-06-09T07:34:25.962759Z","shell.execute_reply":"2021-06-09T07:34:27.297078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class QuestionDataset(Dataset):\n    def __init__(self, X, y, tokenizer):\n        self.text = X.reset_index(drop=True)\n        self.targets = y.reset_index(drop=True)\n        self.tok = tokenizer\n    \n    def __len__(self):\n        return len(self.text)\n    \n    def __getitem__(self, idx):\n        \n        text = self.text[idx]\n        targ = self.targets[idx]\n        \n        return self.tok(text, padding='max_length', \n                        truncation=True,\n                        max_length=30,\n                        return_tensors=\"pt\")[\"input_ids\"][0], tensor(targ)","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:34:27.299615Z","iopub.execute_input":"2021-06-09T07:34:27.299966Z","iopub.status.idle":"2021-06-09T07:34:27.306506Z","shell.execute_reply.started":"2021-06-09T07:34:27.29993Z","shell.execute_reply":"2021-06-09T07:34:27.305394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = train_df\nX_train, X_valid, y_train, y_valid = train_test_split(df[\"question_text\"], df[\"target\"], \n                                                      stratify=df[\"target\"],  test_size=0.01)\n\ntrain_ds = QuestionDataset(X_train, y_train, tokenizer)\nvalid_ds = QuestionDataset(X_valid, y_valid, tokenizer)\n\ntrain_dl = DataLoader(train_ds, bs=256)\nvalid_dl = DataLoader(valid_ds, bs=512)\ndls = DataLoaders(train_dl, valid_dl).to(\"cuda\")","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:34:27.308088Z","iopub.execute_input":"2021-06-09T07:34:27.308502Z","iopub.status.idle":"2021-06-09T07:34:28.424675Z","shell.execute_reply.started":"2021-06-09T07:34:27.308466Z","shell.execute_reply":"2021-06-09T07:34:28.42383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bert = AutoModelForSequenceClassification.from_pretrained('bert-base-cased').train()\n\nclassifier = nn.Sequential(\n    nn.Linear(768, 1024),\n    nn.ReLU(),\n    nn.Dropout(0.5),\n    nn.Linear(1024, 2)\n)\n\nbert.classifier = classifier\n\nclass BertClassifier(Module):\n    def __init__(self, bert):\n        self.bert = bert\n    def forward(self, x):\n        x = self.bert(x)\n        return x.logits\n\nmodel = BertClassifier(bert).to(\"cuda\")","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:34:28.425972Z","iopub.execute_input":"2021-06-09T07:34:28.426338Z","iopub.status.idle":"2021-06-09T07:34:50.775444Z","shell.execute_reply.started":"2021-06-09T07:34:28.426302Z","shell.execute_reply":"2021-06-09T07:34:50.774516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_0 = (train_df[\"target\"] == 0).sum()\nn_1 = (train_df[\"target\"] == 1).sum()\nn = n_0 + n_1","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:34:50.776823Z","iopub.execute_input":"2021-06-09T07:34:50.777189Z","iopub.status.idle":"2021-06-09T07:34:50.791059Z","shell.execute_reply.started":"2021-06-09T07:34:50.77715Z","shell.execute_reply":"2021-06-09T07:34:50.790202Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_weights = tensor([n / (n+n_0), n / (n+n_1)]).to('cuda')\nlearn = Learner(dls, model, \n                loss_func=nn.CrossEntropyLoss(weight=class_weights), \n                metrics=[accuracy, F1Score()]).to_fp16()\nlearn.lr_find()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:34:50.792307Z","iopub.execute_input":"2021-06-09T07:34:50.792758Z","iopub.status.idle":"2021-06-09T07:35:49.761156Z","shell.execute_reply.started":"2021-06-09T07:34:50.792627Z","shell.execute_reply":"2021-06-09T07:35:49.760357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fit_one_cycle(2, lr_max=5e-5)","metadata":{"execution":{"iopub.status.busy":"2021-06-09T07:35:49.762486Z","iopub.execute_input":"2021-06-09T07:35:49.762824Z","iopub.status.idle":"2021-06-09T09:48:08.619557Z","shell.execute_reply.started":"2021-06-09T07:35:49.762785Z","shell.execute_reply":"2021-06-09T09:48:08.618784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score","metadata":{"execution":{"iopub.status.busy":"2021-06-09T09:48:08.620871Z","iopub.execute_input":"2021-06-09T09:48:08.621162Z","iopub.status.idle":"2021-06-09T09:48:08.625798Z","shell.execute_reply.started":"2021-06-09T09:48:08.621125Z","shell.execute_reply":"2021-06-09T09:48:08.625066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds, targs = learn.get_preds()","metadata":{"execution":{"iopub.status.busy":"2021-06-09T09:48:08.626896Z","iopub.execute_input":"2021-06-09T09:48:08.627367Z","iopub.status.idle":"2021-06-09T09:48:23.820528Z","shell.execute_reply.started":"2021-06-09T09:48:08.627328Z","shell.execute_reply":"2021-06-09T09:48:23.819781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"thresholds = np.linspace(0.3, 0.7, 50)\nfor threshold in thresholds:\n    f1 = f1_score(targs, F.softmax(preds, dim=1)[:, 1]>threshold)\n    print(f\"threshold:{threshold:.4f} - f1:{f1:.4f}\")","metadata":{"execution":{"iopub.status.busy":"2021-06-09T09:48:23.821732Z","iopub.execute_input":"2021-06-09T09:48:23.822055Z","iopub.status.idle":"2021-06-09T09:48:24.059034Z","shell.execute_reply.started":"2021-06-09T09:48:23.822028Z","shell.execute_reply":"2021-06-09T09:48:24.058325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_tensor = tokenizer(list(test_df[\"question_text\"]),\n                        padding=\"max_length\",\n                        truncation=True,\n                        max_length=30,\n                        return_tensors=\"pt\")[\"input_ids\"]","metadata":{"execution":{"iopub.status.busy":"2021-06-09T11:28:34.43399Z","iopub.execute_input":"2021-06-09T11:28:34.434349Z","iopub.status.idle":"2021-06-09T11:29:01.651188Z","shell.execute_reply.started":"2021-06-09T11:28:34.434312Z","shell.execute_reply":"2021-06-09T11:29:01.650339Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TestDS:\n    def __init__(self, tensors):\n        self.tensors = tensors\n    \n    def __len__(self):\n        return len(self.tensors)\n    \n    def __getitem__(self, idx):\n        t = self.tensors[idx]\n        return t, tensor(0)\n\ntest_dl = DataLoader(TestDS(test_tensor), bs=128)","metadata":{"execution":{"iopub.status.busy":"2021-06-09T11:29:01.653234Z","iopub.execute_input":"2021-06-09T11:29:01.65362Z","iopub.status.idle":"2021-06-09T11:29:01.659519Z","shell.execute_reply.started":"2021-06-09T11:29:01.653581Z","shell.execute_reply":"2021-06-09T11:29:01.658642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preds = learn.get_preds(dl=test_dl)","metadata":{"execution":{"iopub.status.busy":"2021-06-09T11:29:01.661453Z","iopub.execute_input":"2021-06-09T11:29:01.661919Z","iopub.status.idle":"2021-06-09T11:35:36.046999Z","shell.execute_reply.started":"2021-06-09T11:29:01.661883Z","shell.execute_reply":"2021-06-09T11:35:36.046205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"prediction = (F.softmax(test_preds[0], dim=1)[:, 1]>0.48).int()\nsub = pd.read_csv(path + \"sample_submission.csv\")\nsub[\"prediction\"] = prediction\nsub.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2021-06-09T11:35:36.048458Z","iopub.execute_input":"2021-06-09T11:35:36.048807Z","iopub.status.idle":"2021-06-09T11:35:37.777613Z","shell.execute_reply.started":"2021-06-09T11:35:36.048769Z","shell.execute_reply":"2021-06-09T11:35:37.776718Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}