{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\npd.options.display.max_colwidth = None\npd.options.display.max_columns = 10\n\nfrom IPython.core.interactiveshell import InteractiveShell\nInteractiveShell.ast_node_interactivity = \"all\"","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-09-08T16:07:28.496163Z","iopub.execute_input":"2021-09-08T16:07:28.496552Z","iopub.status.idle":"2021-09-08T16:07:28.517147Z","shell.execute_reply.started":"2021-09-08T16:07:28.496471Z","shell.execute_reply":"2021-09-08T16:07:28.5161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Quora insincere questions classification\n\n## Objective\n\n* Predict whether a question asked on Quora is sincere or not\n* Binary classification","metadata":{"execution":{"iopub.status.busy":"2021-05-24T01:10:49.742859Z","iopub.execute_input":"2021-05-24T01:10:49.743278Z","iopub.status.idle":"2021-05-24T01:10:49.747603Z","shell.execute_reply.started":"2021-05-24T01:10:49.743246Z","shell.execute_reply":"2021-05-24T01:10:49.746746Z"}}},{"cell_type":"code","source":"train_data = pd.read_csv(\"../input/quora-insincere-questions-classification/train.csv\")\ntest_data = pd.read_csv(\"../input/quora-insincere-questions-classification/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:07:36.041715Z","iopub.execute_input":"2021-09-08T16:07:36.042067Z","iopub.status.idle":"2021-09-08T16:07:40.238797Z","shell.execute_reply.started":"2021-09-08T16:07:36.042036Z","shell.execute_reply":"2021-09-08T16:07:40.237953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hand_picked_positives = [0, 8, 12, 16, 41]\npositive_data = train_data.loc[train_data['target'] == 1].iloc[hand_picked_positives].copy()\npositive_data","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:07:42.152829Z","iopub.execute_input":"2021-09-08T16:07:42.153161Z","iopub.status.idle":"2021-09-08T16:07:42.21512Z","shell.execute_reply.started":"2021-09-08T16:07:42.153131Z","shell.execute_reply":"2021-09-08T16:07:42.214253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hand_picked_negatives = [2, 7, 11, 17, 28]\nnegative_data = train_data.loc[train_data['target'] == 0].iloc[hand_picked_negatives].copy()\nnegative_data","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:07:42.856171Z","iopub.execute_input":"2021-09-08T16:07:42.856503Z","iopub.status.idle":"2021-09-08T16:07:42.958429Z","shell.execute_reply.started":"2021-09-08T16:07:42.856473Z","shell.execute_reply":"2021-09-08T16:07:42.957349Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Size of training set: {len(train_data)}')\nprint(f'Size of testing set: {len(test_data)}')\nprint('Distribution of labels in training set:')\nprint(train_data['target'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:07:45.333594Z","iopub.execute_input":"2021-09-08T16:07:45.333925Z","iopub.status.idle":"2021-09-08T16:07:45.353721Z","shell.execute_reply.started":"2021-09-08T16:07:45.333895Z","shell.execute_reply":"2021-09-08T16:07:45.352647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.concat([train_data.loc[train_data['target']==1].tail(2500), train_data.loc[train_data['target']==0].tail(2500)], axis=0)\n\nprint(f'Updated size of training set: {len(train_data)}')\nprint(f'Updated size of testing set: {len(test_data)}')\nprint('Updated distribution of labels in training set:')\ntrain_data['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:07:46.450859Z","iopub.execute_input":"2021-09-08T16:07:46.451222Z","iopub.status.idle":"2021-09-08T16:07:46.651581Z","shell.execute_reply.started":"2021-09-08T16:07:46.451192Z","shell.execute_reply":"2021-09-08T16:07:46.650595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Following code is mostly copied from: https://www.thepythoncode.com/article/finetuning-bert-using-huggingface-transformers-python (thank you!)\n\n# the model we gonna train, base uncased BERT\n# check text classification models here: https://huggingface.co/models?filter=text-classification\nmodel_name = \"bert-base-uncased\"\n# max sequence length for each document/sentence sample\nmax_length = 128","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:07:47.289262Z","iopub.execute_input":"2021-09-08T16:07:47.289823Z","iopub.status.idle":"2021-09-08T16:07:47.294821Z","shell.execute_reply.started":"2021-09-08T16:07:47.289745Z","shell.execute_reply":"2021-09-08T16:07:47.29351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import BertTokenizer, BertTokenizerFast, BertForSequenceClassification\n\n# load the tokenizer\ntokenizer = BertTokenizerFast.from_pretrained(model_name, do_lower_case=True)\ntokenizer_not_fast = BertTokenizer.from_pretrained(model_name, do_lower_case=True)\n","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:07:48.059966Z","iopub.execute_input":"2021-09-08T16:07:48.060336Z","iopub.status.idle":"2021-09-08T16:07:54.907572Z","shell.execute_reply.started":"2021-09-08T16:07:48.060305Z","shell.execute_reply":"2021-09-08T16:07:54.906106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\ntrain_texts, val_texts, train_labels, val_labels = train_test_split(train_data['question_text'].apply(str).tolist(),\n                                                                    train_data['target'].apply(int).tolist(), train_size=0.8)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:07:54.908941Z","iopub.execute_input":"2021-09-08T16:07:54.909288Z","iopub.status.idle":"2021-09-08T16:07:55.802433Z","shell.execute_reply.started":"2021-09-08T16:07:54.909261Z","shell.execute_reply":"2021-09-08T16:07:55.801396Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_encodings = tokenizer(train_texts, truncation=True, padding=True, max_length=max_length)\nval_encodings = tokenizer(val_texts, truncation=True, padding=True, max_length=max_length)\n\n\ntrain_encodings_not_fast = tokenizer_not_fast(train_texts, truncation=True, padding=True, max_length=max_length)\nval_encodings_not_fast = tokenizer_not_fast(val_texts, truncation=True, padding=True, max_length=max_length)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:07:55.80434Z","iopub.execute_input":"2021-09-08T16:07:55.804676Z","iopub.status.idle":"2021-09-08T16:07:59.250103Z","shell.execute_reply.started":"2021-09-08T16:07:55.80464Z","shell.execute_reply":"2021-09-08T16:07:59.249255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# when using Rust-based tokenizers (\"*Fast\" tokenizers), the outputs of the tokenizers are objects of Encoding class \nencoding1 = train_encodings[0]\nencoding1\n\n# when using Python-based tokenizers (\"not fast\" tokenizers), the outputs of the tokenizers are python dicts \n# encoding_not_fast1 = train_encodings_not_fast[0]  # won't work: \"Indexing with integers (to access backend Encoding for a given batch index) is not available when using Python based tokenizers\"\ntrain_encodings_not_fast.keys()\n# train_encodings_not_fast['input_ids'][0]","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:07:59.251566Z","iopub.execute_input":"2021-09-08T16:07:59.251868Z","iopub.status.idle":"2021-09-08T16:07:59.262741Z","shell.execute_reply.started":"2021-09-08T16:07:59.251833Z","shell.execute_reply":"2021-09-08T16:07:59.261721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\n\nclass CustomDataset(torch.utils.data.Dataset):\n    def __init__(self, encodings, labels):\n        self.encodings = encodings\n        self.labels = labels\n\n    def __getitem__(self, idx):\n        item = {key: torch.tensor(val[idx]) for key, val in self.encodings.items()}\n        item['labels'] = torch.tensor(self.labels[idx])\n        return item\n\n    def __len__(self):\n        return len(self.labels)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:07:59.264332Z","iopub.execute_input":"2021-09-08T16:07:59.264727Z","iopub.status.idle":"2021-09-08T16:07:59.27168Z","shell.execute_reply.started":"2021-09-08T16:07:59.264691Z","shell.execute_reply":"2021-09-08T16:07:59.270544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# convert our tokenized data into a torch Dataset\ntrain_dataset = CustomDataset(train_encodings, train_labels)\nvalid_dataset = CustomDataset(val_encodings, val_labels)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:07:59.272922Z","iopub.execute_input":"2021-09-08T16:07:59.273526Z","iopub.status.idle":"2021-09-08T16:07:59.281378Z","shell.execute_reply.started":"2021-09-08T16:07:59.27349Z","shell.execute_reply":"2021-09-08T16:07:59.280566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load the model and pass to CUDA\nmodel = BertForSequenceClassification.from_pretrained(model_name, num_labels=2).to(\"cuda\")","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:07:59.282625Z","iopub.execute_input":"2021-09-08T16:07:59.283012Z","iopub.status.idle":"2021-09-08T16:08:20.258151Z","shell.execute_reply.started":"2021-09-08T16:07:59.282957Z","shell.execute_reply":"2021-09-08T16:08:20.257228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score\n\ndef compute_metrics(pred):\n  labels = pred.label_ids\n  preds = pred.predictions.argmax(-1)\n  # calculate accuracy using sklearn's function\n  acc = accuracy_score(labels, preds)\n  return {\n      'accuracy': acc,\n  }","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:08:20.26024Z","iopub.execute_input":"2021-09-08T16:08:20.260527Z","iopub.status.idle":"2021-09-08T16:08:20.270256Z","shell.execute_reply.started":"2021-09-08T16:08:20.260501Z","shell.execute_reply":"2021-09-08T16:08:20.269221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training ","metadata":{}},{"cell_type":"code","source":"from transformers import Trainer, TrainingArguments\n\ntraining_args = TrainingArguments(\n    output_dir='./results',          # output directory\n    num_train_epochs=1,              # total number of training epochs\n    per_device_train_batch_size=16,  # batch size per device during training\n    per_device_eval_batch_size=20,   # batch size for evaluation\n    warmup_steps=50,                 # number of warmup steps for learning rate scheduler\n    weight_decay=0.01,               # strength of weight decay\n    logging_dir='./logs',            # directory for storing logs\n    load_best_model_at_end=True,     # load the best model when finished training (default metric is loss)\n    # but you can specify `metric_for_best_model` argument to change to accuracy or other metric\n    logging_steps=50,               # log & save weights each logging_steps\n    evaluation_strategy=\"steps\",     # evaluate each `logging_steps`\n)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:08:20.272052Z","iopub.execute_input":"2021-09-08T16:08:20.272421Z","iopub.status.idle":"2021-09-08T16:08:26.362867Z","shell.execute_reply.started":"2021-09-08T16:08:20.272384Z","shell.execute_reply":"2021-09-08T16:08:26.361999Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer = Trainer(\n    model=model,                         # the instantiated Transformers model to be trained\n    args=training_args,                  # training arguments, defined above\n    train_dataset=train_dataset,         # training dataset\n    eval_dataset=valid_dataset,          # evaluation dataset\n    compute_metrics=compute_metrics,     # the callback that computes metrics of interest\n)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:08:26.364088Z","iopub.execute_input":"2021-09-08T16:08:26.364464Z","iopub.status.idle":"2021-09-08T16:08:27.450909Z","shell.execute_reply.started":"2021-09-08T16:08:26.364427Z","shell.execute_reply":"2021-09-08T16:08:27.450013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train the model\ntrainer.train()","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:08:27.452336Z","iopub.execute_input":"2021-09-08T16:08:27.452681Z","iopub.status.idle":"2021-09-08T16:11:04.363285Z","shell.execute_reply.started":"2021-09-08T16:08:27.452643Z","shell.execute_reply":"2021-09-08T16:11:04.361004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation\n## Evaluate using trainer.evaluate()","metadata":{}},{"cell_type":"code","source":"# evaluate the current model after training\ntrainer.evaluate()\naccuracy_from_evaluate_method = trainer.evaluate()['eval_accuracy']\nprint(f'> Accuracy from .evaluate() method: {accuracy_from_evaluate_method}')","metadata":{"execution":{"iopub.status.busy":"2021-09-08T17:03:50.931714Z","iopub.execute_input":"2021-09-08T17:03:50.932058Z","iopub.status.idle":"2021-09-08T17:03:57.762038Z","shell.execute_reply.started":"2021-09-08T17:03:50.932027Z","shell.execute_reply":"2021-09-08T17:03:57.761164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluate by manually passing texts throught the model ","metadata":{}},{"cell_type":"code","source":"def get_prediction_proba(text):\n    # prepare our text into tokenized sequence\n    inputs = tokenizer(text, padding=True, truncation=True, return_tensors=\"pt\").to(\"cuda\")\n    # perform inference to our model\n    outputs = model(**inputs)\n    # get output probabilities by doing softmax\n    probs = outputs[0].softmax(1)\n    # executing argmax function to get the candidate label\n    return probs\n\ndef get_prediction(text):\n    return get_prediction_proba(text).argmax().item()","metadata":{"execution":{"iopub.status.busy":"2021-09-08T17:09:24.17776Z","iopub.execute_input":"2021-09-08T17:09:24.178124Z","iopub.status.idle":"2021-09-08T17:09:24.18542Z","shell.execute_reply.started":"2021-09-08T17:09:24.178091Z","shell.execute_reply":"2021-09-08T17:09:24.184509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# manual verification of accuracy \ndf_val_texts = pd.DataFrame({'text': val_texts, 'label': val_labels})\ndf_val_texts['prediction'] = df_val_texts['text'].apply(get_prediction)\ndf_val_texts['correct_prediction'] = np.where(df_val_texts['prediction'] == df_val_texts['label'], 1, 0)\n\nnum_correct = df_val_texts['correct_prediction'].sum()\nnum_total_predictions = len(df_val_texts)\naccuracy_manual = num_correct/num_total_predictions\n\nprint(f'accuracy_manual: {accuracy_manual}')\naccuracy_matches = accuracy_from_evaluate_method == accuracy_manual \nprint(f'Match accuracy in manual and .evaluate(): {accuracy_matches}')","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:36:16.411927Z","iopub.execute_input":"2021-09-08T16:36:16.412304Z","iopub.status.idle":"2021-09-08T16:36:30.270995Z","shell.execute_reply.started":"2021-09-08T16:36:16.41227Z","shell.execute_reply":"2021-09-08T16:36:30.270204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluate using trainer.predict()","metadata":{}},{"cell_type":"code","source":"# verification of accuracy using trainer.predict(): \nprediction_output = trainer.predict(valid_dataset)\n\n# we get predictions as a numpy array of logits: \npredictions = prediction_output.predictions  # note: do not use predicted_outputs.label_ids -- that will just give back the true label_ids, not the predictions \nprint(f'prediction shape: {predictions.shape}')  # 1000 predictions; each with two outputs, one for each class \nprint(f'num preds = len(val_texts): {predictions.shape[0] == len(val_texts)}')\n\n# example: \nfirst_prediction = predictions[0]\nprint(f'first pred: {first_prediction}')\n","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:44:32.116372Z","iopub.execute_input":"2021-09-08T16:44:32.116707Z","iopub.status.idle":"2021-09-08T16:44:35.525631Z","shell.execute_reply.started":"2021-09-08T16:44:32.116678Z","shell.execute_reply":"2021-09-08T16:44:35.524762Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.special import softmax \n\nsoftmax_output = softmax(predictions, axis=1)\nprint(f'> softmax_output: \\n{softmax_output}')\nprint(f'> softmax_output.sum(axis=1).shape: \\n {softmax_output.sum(axis=1).shape}')\n\npredicted_labels = softmax_output.argmax(axis=1)\ncorrect_predictions  = (predicted_labels == prediction_output.label_ids)\nnum_correct_predictions = correct_predictions.sum()\nprint(f'> num correct preds: \\n {num_correct_predictions}')\n\naccuracy_from_predict_method = num_correct_predictions/len(df_val_texts)\nprint(f'> accuracy from predict method: {accuracy_from_predict_method}')\naccuracy_matches = accuracy_from_predict_method == accuracy_manual \nprint(f'> Match accuracy in manual and .predict(): {accuracy_matches}')\n","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:45:07.526537Z","iopub.execute_input":"2021-09-08T16:45:07.526889Z","iopub.status.idle":"2021-09-08T16:45:07.547158Z","shell.execute_reply.started":"2021-09-08T16:45:07.526857Z","shell.execute_reply":"2021-09-08T16:45:07.546035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"accuracy_manual\naccuracy_from_evaluate_method\naccuracy_from_predict_method","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:45:09.309865Z","iopub.execute_input":"2021-09-08T16:45:09.310243Z","iopub.status.idle":"2021-09-08T16:45:09.324643Z","shell.execute_reply.started":"2021-09-08T16:45:09.31021Z","shell.execute_reply":"2021-09-08T16:45:09.32092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# getting predicted labels in one line: \npredicted_labels = softmax(prediction_output.predictions, axis=1).argmax(axis=1)\n\ndf_preds = pd.DataFrame({'text': val_texts, 'pred_label': predicted_labels, 'label': prediction_output.label_ids})\ndf_preds.head()\ndf_preds[df_preds['pred_label'] != df_preds['label']].sample(20)","metadata":{"execution":{"iopub.status.busy":"2021-09-08T16:45:45.681394Z","iopub.execute_input":"2021-09-08T16:45:45.681722Z","iopub.status.idle":"2021-09-08T16:45:45.714957Z","shell.execute_reply.started":"2021-09-08T16:45:45.681694Z","shell.execute_reply":"2021-09-08T16:45:45.714092Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Random Tests","metadata":{}},{"cell_type":"code","source":"print(get_prediction_proba(\"Is this sincere question?\"))","metadata":{"execution":{"iopub.status.busy":"2021-05-27T22:22:49.832744Z","iopub.execute_input":"2021-05-27T22:22:49.83308Z","iopub.status.idle":"2021-05-27T22:22:49.86593Z","shell.execute_reply.started":"2021-05-27T22:22:49.833049Z","shell.execute_reply":"2021-05-27T22:22:49.865078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"positive_data['pred'] = positive_data['question_text'].apply(get_prediction))\nnegative_data['pred'] = negative_data['question_text'].apply(lambda x: get_prediction.argmax().item())","metadata":{"execution":{"iopub.status.busy":"2021-05-27T22:19:16.938133Z","iopub.execute_input":"2021-05-27T22:19:16.938614Z","iopub.status.idle":"2021-05-27T22:19:17.127315Z","shell.execute_reply.started":"2021-05-27T22:19:16.938562Z","shell.execute_reply":"2021-05-27T22:19:17.126513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"positive_data","metadata":{"execution":{"iopub.status.busy":"2021-05-27T22:19:19.282478Z","iopub.execute_input":"2021-05-27T22:19:19.282931Z","iopub.status.idle":"2021-05-27T22:19:19.30124Z","shell.execute_reply.started":"2021-05-27T22:19:19.282891Z","shell.execute_reply":"2021-05-27T22:19:19.300042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"negative_data","metadata":{"execution":{"iopub.status.busy":"2021-05-27T22:19:22.448958Z","iopub.execute_input":"2021-05-27T22:19:22.449377Z","iopub.status.idle":"2021-05-27T22:19:22.472098Z","shell.execute_reply.started":"2021-05-27T22:19:22.44933Z","shell.execute_reply":"2021-05-27T22:19:22.471159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}