{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-20T09:33:58.216188Z","iopub.execute_input":"2022-07-20T09:33:58.216806Z","iopub.status.idle":"2022-07-20T09:33:58.246019Z","shell.execute_reply.started":"2022-07-20T09:33:58.216718Z","shell.execute_reply":"2022-07-20T09:33:58.245025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport re\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nsns.set_style(\"darkgrid\")\n\nimport string\nfrom wordcloud import WordCloud\nfrom nltk import word_tokenize\nfrom nltk.corpus import stopwords\nfrom nltk.stem import WordNetLemmatizer\n\nfrom sklearn.metrics import confusion_matrix, f1_score\nfrom tqdm import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:34:00.642339Z","iopub.execute_input":"2022-07-20T09:34:00.642818Z","iopub.status.idle":"2022-07-20T09:34:02.251918Z","shell.execute_reply.started":"2022-07-20T09:34:00.642762Z","shell.execute_reply":"2022-07-20T09:34:02.250825Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/nlp-getting-started/train.csv\")\nprint (df_train.shape)\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:34:04.879540Z","iopub.execute_input":"2022-07-20T09:34:04.879893Z","iopub.status.idle":"2022-07-20T09:34:04.936956Z","shell.execute_reply.started":"2022-07-20T09:34:04.879862Z","shell.execute_reply":"2022-07-20T09:34:04.936010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(\"/kaggle/input/nlp-getting-started/test.csv\")\nprint (df_test.shape)\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:34:06.698803Z","iopub.execute_input":"2022-07-20T09:34:06.699364Z","iopub.status.idle":"2022-07-20T09:34:06.727904Z","shell.execute_reply.started":"2022-07-20T09:34:06.699329Z","shell.execute_reply":"2022-07-20T09:34:06.726919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\n\nif torch.cuda.is_available():       \n    device = torch.device(\"cuda\")\n    print(f'There are {torch.cuda.device_count()} GPU(s) available.')\n    print('Device name:', torch.cuda.get_device_name(0))\n\nelse:\n    print('No GPU available, using the CPU instead.')\n    device = torch.device(\"cpu\")","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:34:08.353016Z","iopub.execute_input":"2022-07-20T09:34:08.353578Z","iopub.status.idle":"2022-07-20T09:34:10.122152Z","shell.execute_reply.started":"2022-07-20T09:34:08.353541Z","shell.execute_reply":"2022-07-20T09:34:10.120956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# データの前処理","metadata":{}},{"cell_type":"code","source":"!pip install contractions","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:34:14.038588Z","iopub.execute_input":"2022-07-20T09:34:14.039097Z","iopub.status.idle":"2022-07-20T09:34:26.220599Z","shell.execute_reply.started":"2022-07-20T09:34:14.039065Z","shell.execute_reply":"2022-07-20T09:34:26.219474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# データの前処理　短縮形を元の形に統一する．\nimport contractions\n\ndf_train['text']=df_train['text'].apply(lambda x : contractions.fix(x))\ndf_test['text']=df_test['text'].apply(lambda x : contractions.fix(x))\n\ndf_train","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:34:27.934167Z","iopub.execute_input":"2022-07-20T09:34:27.934808Z","iopub.status.idle":"2022-07-20T09:34:28.094929Z","shell.execute_reply.started":"2022-07-20T09:34:27.934773Z","shell.execute_reply":"2022-07-20T09:34:28.093915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# URLを削除する\ndf_train['text']=df_train['text'].apply(lambda x : re.compile(r'(https|http)?:\\/\\/(\\w|\\.|\\/|\\?|\\=|\\&|\\%)*\\b').sub(r'',x))\ndf_test['text']=df_test['text'].apply(lambda x : re.compile(r'(https|http)?:\\/\\/(\\w|\\.|\\/|\\?|\\=|\\&|\\%)*\\b').sub(r'',x))\n\ndf_train","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:34:30.458124Z","iopub.execute_input":"2022-07-20T09:34:30.459234Z","iopub.status.idle":"2022-07-20T09:34:30.629954Z","shell.execute_reply.started":"2022-07-20T09:34:30.459140Z","shell.execute_reply":"2022-07-20T09:34:30.629050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ハッシュタグなどの記号の削除\ndf_train['text']=df_train['text'].apply(lambda x : x.translate(str.maketrans('','',string.punctuation)))\ndf_test['text']=df_test['text'].apply(lambda x : x.translate(str.maketrans('','',string.punctuation)))\n\ndf_train","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:34:32.965480Z","iopub.execute_input":"2022-07-20T09:34:32.966389Z","iopub.status.idle":"2022-07-20T09:34:33.038598Z","shell.execute_reply.started":"2022-07-20T09:34:32.966350Z","shell.execute_reply":"2022-07-20T09:34:33.037562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# BERTのFine tuning","metadata":{}},{"cell_type":"code","source":"!pip install transformers","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:35:29.834454Z","iopub.execute_input":"2022-07-20T09:35:29.835273Z","iopub.status.idle":"2022-07-20T09:35:39.085814Z","shell.execute_reply.started":"2022-07-20T09:35:29.835234Z","shell.execute_reply":"2022-07-20T09:35:39.084682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import BertTokenizer\n\n# Load the BERT tokenizer\ntokenizer = BertTokenizer.from_pretrained('bert-base-uncased', do_lower_case=True)\n\n# Create a function to tokenize a set of texts\ndef preprocessing_for_bert(data):\n    # Create empty lists to store outputs\n    input_ids = []\n    attention_masks = []\n    \n    # For every sentence...\n    for sent in data:\n        encoded_sent = tokenizer.encode_plus(\n            text=sent,\n            add_special_tokens=True,        # Add `[CLS]` and `[SEP]`\n            max_length=MAX_LEN,             # Max length to truncate/pad\n            pad_to_max_length=True,         # Pad sentence to max length\n            return_attention_mask=True      # Return attention mask\n            )\n        \n        # Add the outputs to the lists\n        input_ids.append(encoded_sent.get('input_ids'))\n        attention_masks.append(encoded_sent.get('attention_mask'))\n        \n    # Convert lists to tensors\n    input_ids = torch.tensor(input_ids)\n    attention_masks = torch.tensor(attention_masks)\n\n    return input_ids, attention_masks","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:35:39.089997Z","iopub.execute_input":"2022-07-20T09:35:39.090461Z","iopub.status.idle":"2022-07-20T09:35:46.991386Z","shell.execute_reply.started":"2022-07-20T09:35:39.090414Z","shell.execute_reply":"2022-07-20T09:35:46.990506Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concatenate train data and test data\nall_tweets = np.concatenate([df_train['text'], df_test['text']])\n\n# Encode our concatenated data\nencoded_tweets = [tokenizer.encode(sent, add_special_tokens=True) for sent in all_tweets]\n\n# Find the maximum length\nmax_len = max([len(sent) for sent in encoded_tweets])\nprint('Max length: ', max_len)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:35:50.554365Z","iopub.execute_input":"2022-07-20T09:35:50.555216Z","iopub.status.idle":"2022-07-20T09:35:57.461224Z","shell.execute_reply.started":"2022-07-20T09:35:50.555144Z","shell.execute_reply":"2022-07-20T09:35:57.459395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nX = df_train['text']\ny = df_train.target.values\n\nX_train, X_val, y_train, y_val = train_test_split(X, y, test_size=0.1, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:40:25.209343Z","iopub.execute_input":"2022-07-20T09:40:25.209816Z","iopub.status.idle":"2022-07-20T09:40:25.225063Z","shell.execute_reply.started":"2022-07-20T09:40:25.209772Z","shell.execute_reply":"2022-07-20T09:40:25.223862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_LEN = 64\n\n# Print sentence 0 and its encoded token ids\ntoken_ids = list(preprocessing_for_bert([X[0]])[0].squeeze().numpy())\nprint('Original: ', X[0])\nprint('Token IDs: ', token_ids)\n\n# Run function `preprocessing_for_bert` on the train set and the validation set\nprint('Tokenizing data...')\ntrain_inputs, train_masks = preprocessing_for_bert(X_train)\nval_inputs, val_masks = preprocessing_for_bert(X_val)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:36:00.719354Z","iopub.execute_input":"2022-07-20T09:36:00.720261Z","iopub.status.idle":"2022-07-20T09:36:05.941021Z","shell.execute_reply.started":"2022-07-20T09:36:00.720210Z","shell.execute_reply":"2022-07-20T09:36:05.940067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:40:29.269980Z","iopub.execute_input":"2022-07-20T09:40:29.270363Z","iopub.status.idle":"2022-07-20T09:40:29.277293Z","shell.execute_reply.started":"2022-07-20T09:40:29.270330Z","shell.execute_reply":"2022-07-20T09:40:29.276305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import TensorDataset, DataLoader, RandomSampler, SequentialSampler\n\n# Convert other data types to torch.Tensor\ntrain_labels = torch.tensor(y_train)\nval_labels = torch.tensor(y_val)\n\n# For fine-tuning BERT, the authors recommend a batch size of 16 or 32.\nbatch_size = 32\n\n# Create the DataLoader for our training set\ntrain_data = TensorDataset(train_inputs, train_masks, train_labels)\ntrain_sampler = RandomSampler(train_data)\ntrain_dataloader = DataLoader(train_data, sampler=train_sampler, batch_size=batch_size)\n\n# Create the DataLoader for our validation set\nval_data = TensorDataset(val_inputs, val_masks, val_labels)\nval_sampler = SequentialSampler(val_data)\nval_dataloader = DataLoader(val_data, sampler=val_sampler, batch_size=batch_size)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:40:40.279225Z","iopub.execute_input":"2022-07-20T09:40:40.280105Z","iopub.status.idle":"2022-07-20T09:40:40.288531Z","shell.execute_reply.started":"2022-07-20T09:40:40.280056Z","shell.execute_reply":"2022-07-20T09:40:40.287461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nimport torch\nimport torch.nn as nn\nfrom transformers import BertModel\n\n# Create the BertClassfier class\nclass BertClassifier(nn.Module):\n    \"\"\"Bert Model for Classification Tasks.\n    \"\"\"\n    def __init__(self, freeze_bert=False):\n        \"\"\"\n        @param    bert: a BertModel object\n        @param    classifier: a torch.nn.Module classifier\n        @param    freeze_bert (bool): Set `False` to fine-tune the BERT model\n        \"\"\"\n        super(BertClassifier, self).__init__()\n        # Specify hidden size of BERT, hidden size of our classifier, and number of labels\n        D_in, H, D_out = 768, 50, 2\n\n        # Instantiate BERT model\n        self.bert = BertModel.from_pretrained('bert-base-uncased')\n\n        # Instantiate an one-layer feed-forward classifier\n        self.classifier = nn.Sequential(\n            nn.Linear(D_in, H),\n            nn.ReLU(),\n            #nn.Dropout(0.5),\n            nn.Linear(H, D_out)\n        )\n        \n        # Freeze the BERT model\n        if freeze_bert:\n            for param in self.bert.parameters():\n                param.requires_grad = False\n        \n    def forward(self, input_ids, attention_mask):\n        \"\"\"\n        Feed input to BERT and the classifier to compute logits.\n        @param    input_ids (torch.Tensor): an input tensor with shape (batch_size,\n                      max_length)\n        @param    attention_mask (torch.Tensor): a tensor that hold attention mask\n                      information with shape (batch_size, max_length)\n        @return   logits (torch.Tensor): an output tensor with shape (batch_size,\n                      num_labels)\n        \"\"\"\n        # Feed input to BERT\n        outputs = self.bert(input_ids=input_ids,\n                            attention_mask=attention_mask)\n        \n        # Extract the last hidden state of the token `[CLS]` for classification task\n        last_hidden_state_cls = outputs[0][:, 0, :]\n\n        # Feed input to classifier to compute logits\n        logits = self.classifier(last_hidden_state_cls)\n\n        return logits","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:40:41.265554Z","iopub.execute_input":"2022-07-20T09:40:41.265913Z","iopub.status.idle":"2022-07-20T09:40:41.276279Z","shell.execute_reply.started":"2022-07-20T09:40:41.265883Z","shell.execute_reply":"2022-07-20T09:40:41.273558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AdamW, get_linear_schedule_with_warmup\n\ndef initialize_model(epochs=4):\n    \"\"\"Initialize the Bert Classifier, the optimizer and the learning rate scheduler.\n    \"\"\"\n    # Instantiate Bert Classifier\n    bert_classifier = BertClassifier(freeze_bert=False)\n\n    # Tell PyTorch to run the model on GPU\n    bert_classifier.to(device)\n\n    # Create the optimizer\n    optimizer = AdamW(bert_classifier.parameters(),\n                      lr=5e-5,    # Default learning rate\n                      eps=1e-8    # Default epsilon value\n                      )\n\n    # Total number of training steps\n    total_steps = len(train_dataloader) * epochs\n\n    # Set up the learning rate scheduler\n    scheduler = get_linear_schedule_with_warmup(optimizer,\n                                                num_warmup_steps=0, # Default value\n                                                num_training_steps=total_steps)\n    return bert_classifier, optimizer, scheduler","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:40:42.896712Z","iopub.execute_input":"2022-07-20T09:40:42.897062Z","iopub.status.idle":"2022-07-20T09:40:42.904278Z","shell.execute_reply.started":"2022-07-20T09:40:42.897032Z","shell.execute_reply":"2022-07-20T09:40:42.902876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import random\nimport time\n\n# Specify loss function\nloss_fn = nn.CrossEntropyLoss()\n\ndef set_seed(seed_value=42):\n    \"\"\"Set seed for reproducibility.\n    \"\"\"\n    random.seed(seed_value)\n    np.random.seed(seed_value)\n    torch.manual_seed(seed_value)\n    torch.cuda.manual_seed_all(seed_value)\n\ndef train(model, train_dataloader, val_dataloader=None, epochs=4, evaluation=False):\n    \"\"\"Train the BertClassifier model.\n    \"\"\"\n    # Start training loop\n    print(\"Start training...\\n\")\n    for epoch_i in range(epochs):\n        # =======================================\n        #               Training\n        # =======================================\n        # Print the header of the result table\n        print(f\"{'Epoch':^7} | {'Batch':^7} | {'Train Loss':^12} | {'Val Loss':^10} | {'Val Acc':^9} | {'Elapsed':^9}\")\n        print(\"-\"*70)\n\n        # Measure the elapsed time of each epoch\n        t0_epoch, t0_batch = time.time(), time.time()\n\n        # Reset tracking variables at the beginning of each epoch\n        total_loss, batch_loss, batch_counts = 0, 0, 0\n\n        # Put the model into the training mode\n        model.train()\n\n        # For each batch of training data...\n        for step, batch in enumerate(train_dataloader):\n            batch_counts +=1\n            # Load batch to GPU\n            b_input_ids, b_attn_mask, b_labels = tuple(t.to(device) for t in batch)\n\n            # Zero out any previously calculated gradients\n            model.zero_grad()\n\n            # Perform a forward pass. This will return logits.\n            logits = model(b_input_ids, b_attn_mask)\n\n            # Compute loss and accumulate the loss values\n            loss = loss_fn(logits, b_labels)\n            batch_loss += loss.item()\n            total_loss += loss.item()\n            \n            # Perform a backward pass to calculate gradients\n            loss.backward()\n\n            # Clip the norm of the gradients to 1.0 to prevent \"exploding gradients\"\n            torch.nn.utils.clip_grad_norm_(model.parameters(), 1.0)\n\n            # Update parameters and the learning rate\n            optimizer.step()\n            scheduler.step()\n\n            # Print the loss values and time elapsed for every 20 batches\n            if (step % 20 == 0 and step != 0) or (step == len(train_dataloader) - 1):\n                # Calculate time elapsed for 20 batches\n                time_elapsed = time.time() - t0_batch\n\n                # Print training results\n                print(f\"{epoch_i + 1:^7} | {step:^7} | {batch_loss / batch_counts:^12.6f} | {'-':^10} | {'-':^9} | {time_elapsed:^9.2f}\")\n                \n                # Reset batch tracking variables\n                batch_loss, batch_counts = 0, 0\n                t0_batch = time.time()\n\n        # Calculate the average loss over the entire training data\n        avg_train_loss = total_loss / len(train_dataloader)\n\n        print(\"-\"*70)\n        # =======================================\n        #               Evaluation\n        # =======================================\n        if evaluation == True:\n            # After the completion of each training epoch, measure the model's performance\n            # on our validation set.\n            val_loss, val_accuracy = evaluate(model, val_dataloader)\n\n            # Print performance over the entire training data\n            time_elapsed = time.time() - t0_epoch\n            \n            print(f\"{epoch_i + 1:^7} | {'-':^7} | {avg_train_loss:^12.6f} | {val_loss:^10.6f} | {val_accuracy:^9.2f} | {time_elapsed:^9.2f}\")\n            print(\"-\"*70)\n        print(\"\\n\")\n    \n    print(\"Training complete!\")\n    \n    \ndef evaluate(model, val_dataloader):\n    \"\"\"After the completion of each training epoch, measure the model's performance\n    on our validation set.\n    \"\"\"\n    # Put the model into the evaluation mode. The dropout layers are disabled during\n    # the test time.\n    model.eval()\n\n    # Tracking variables\n    val_accuracy = []\n    val_loss = []\n\n    # For each batch in our validation set...\n    for batch in val_dataloader:\n        # Load batch to GPU\n        b_input_ids, b_attn_mask, b_labels = tuple(t.to(device) for t in batch)\n\n        # Compute logits\n        with torch.no_grad():\n            logits = model(b_input_ids, b_attn_mask)\n\n        # Compute loss\n        loss = loss_fn(logits, b_labels)\n        val_loss.append(loss.item())\n\n        # Get the predictions\n        preds = torch.argmax(logits, dim=1).flatten()\n        \n        # Calculate the accuracy rate\n        accuracy = (preds == b_labels).cpu().numpy().mean() * 100\n        val_accuracy.append(accuracy)\n\n    # Compute the average accuracy and loss over the validation set.\n    val_loss = np.mean(val_loss)\n    val_accuracy = np.mean(val_accuracy)\n\n    return val_loss, val_accuracy","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:40:43.781758Z","iopub.execute_input":"2022-07-20T09:40:43.782103Z","iopub.status.idle":"2022-07-20T09:40:43.815116Z","shell.execute_reply.started":"2022-07-20T09:40:43.782073Z","shell.execute_reply":"2022-07-20T09:40:43.814089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# set_seed(42)    # Set seed for reproducibility\n# bert_classifier, optimizer, scheduler = initialize_model(epochs=2)\n# train(bert_classifier, train_dataloader, val_dataloader, epochs=2, evaluation=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:40:46.240413Z","iopub.execute_input":"2022-07-20T09:40:46.240764Z","iopub.status.idle":"2022-07-20T09:42:17.299674Z","shell.execute_reply.started":"2022-07-20T09:40:46.240734Z","shell.execute_reply":"2022-07-20T09:42:17.298557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn.functional as F\n\ndef bert_predict(model, test_dataloader):\n    \"\"\"Perform a forward pass on the trained BERT model to predict probabilities\n    on the test set.\n    \"\"\"\n    # Put the model into the evaluation mode. The dropout layers are disabled during\n    # the test time.\n    model.eval()\n\n    all_logits = []\n\n    # For each batch in our test set...\n    for batch in test_dataloader:\n        # Load batch to GPU\n        b_input_ids, b_attn_mask = tuple(t.to(device) for t in batch)[:2]\n\n        # Compute logits\n        with torch.no_grad():\n            logits = model(b_input_ids, b_attn_mask)\n        all_logits.append(logits)\n    \n    # Concatenate logits from each batch\n    all_logits = torch.cat(all_logits, dim=0)\n\n    # Apply softmax to calculate probabilities\n    probs = F.softmax(all_logits, dim=1).cpu().numpy()\n\n    return probs","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:42:21.050128Z","iopub.execute_input":"2022-07-20T09:42:21.050848Z","iopub.status.idle":"2022-07-20T09:42:21.058434Z","shell.execute_reply.started":"2022-07-20T09:42:21.050811Z","shell.execute_reply":"2022-07-20T09:42:21.057256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, roc_curve, auc\n\ndef evaluate_roc(probs, y_true):\n    \"\"\"\n    - Print AUC and accuracy on the test set\n    - Plot ROC\n    @params    probs (np.array): an array of predicted probabilities with shape (len(y_true), 2)\n    @params    y_true (np.array): an array of the true values with shape (len(y_true),)\n    \"\"\"\n    preds = probs[:, 1]\n    fpr, tpr, threshold = roc_curve(y_true, preds)\n    roc_auc = auc(fpr, tpr)\n    print(f'AUC: {roc_auc:.4f}')\n       \n    # Get accuracy over the test set\n    y_pred = np.where(preds >= 0.4, 1, 0)\n    accuracy = accuracy_score(y_true, y_pred)\n    print(f'Accuracy: {accuracy*100:.2f}%')\n    \n    # Plot ROC AUC\n    plt.title('Receiver Operating Characteristic')\n    plt.plot(fpr, tpr, 'b', label = 'AUC = %0.2f' % roc_auc)\n    plt.legend(loc = 'lower right')\n    plt.plot([0, 1], [0, 1],'r--')\n    plt.xlim([0, 1])\n    plt.ylim([0, 1])\n    plt.ylabel('True Positive Rate')\n    plt.xlabel('False Positive Rate')\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:47:10.194659Z","iopub.execute_input":"2022-07-20T09:47:10.195014Z","iopub.status.idle":"2022-07-20T09:47:10.207422Z","shell.execute_reply.started":"2022-07-20T09:47:10.194983Z","shell.execute_reply":"2022-07-20T09:47:10.206436Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Compute predicted probabilities on the test set\n# probs = bert_predict(bert_classifier, val_dataloader)\n\n# # Evaluate the Bert classifier\n# evaluate_roc(probs, y_val)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:47:12.613445Z","iopub.execute_input":"2022-07-20T09:47:12.613791Z","iopub.status.idle":"2022-07-20T09:47:14.208049Z","shell.execute_reply.started":"2022-07-20T09:47:12.613761Z","shell.execute_reply":"2022-07-20T09:47:14.207066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Concatenate the train set and the validation set\nfull_train_data = torch.utils.data.ConcatDataset([train_data, val_data])\nfull_train_sampler = RandomSampler(full_train_data)\nfull_train_dataloader = DataLoader(full_train_data, sampler=full_train_sampler, batch_size=32)\n\n# Train the Bert Classifier on the entire training data\nset_seed(42)\nbert_classifier, optimizer, scheduler = initialize_model(epochs=2)\ntrain(bert_classifier, full_train_dataloader, epochs=2)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:47:17.551009Z","iopub.execute_input":"2022-07-20T09:47:17.551677Z","iopub.status.idle":"2022-07-20T09:48:54.065194Z","shell.execute_reply.started":"2022-07-20T09:47:17.551639Z","shell.execute_reply":"2022-07-20T09:48:54.064184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Run `preprocessing_for_bert` on the test set\nprint('Tokenizing data...')\ntest_inputs, test_masks = preprocessing_for_bert(df_test['text'])\n\n# Create the DataLoader for our test set\ntest_dataset = TensorDataset(test_inputs, test_masks)\ntest_sampler = SequentialSampler(test_dataset)\ntest_dataloader = DataLoader(test_dataset, sampler=test_sampler, batch_size=32)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:48:57.805829Z","iopub.execute_input":"2022-07-20T09:48:57.806197Z","iopub.status.idle":"2022-07-20T09:48:59.934478Z","shell.execute_reply.started":"2022-07-20T09:48:57.806144Z","shell.execute_reply":"2022-07-20T09:48:59.933512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compute predicted probabilities on the test set\nprobs = bert_predict(bert_classifier, test_dataloader)\n\n# Get predictions from the probabilities\nthreshold = 0.6 # best is 0.7\npreds = np.where(probs[:, 1] > threshold, 1, 0)\n\n# Number of tweets predicted non-negative\nprint(\"Number of tweets predicted non-negative: \", preds.sum())","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:49:03.560658Z","iopub.execute_input":"2022-07-20T09:49:03.561271Z","iopub.status.idle":"2022-07-20T09:49:09.409889Z","shell.execute_reply.started":"2022-07-20T09:49:03.561235Z","shell.execute_reply":"2022-07-20T09:49:09.408806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# output = test_data[preds==1]\n# list(output.sample(20).text)\n# np.set_printoptions(threshold=np.inf)\n# print(preds)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:49:15.104622Z","iopub.execute_input":"2022-07-20T09:49:15.104971Z","iopub.status.idle":"2022-07-20T09:49:15.124760Z","shell.execute_reply.started":"2022-07-20T09:49:15.104941Z","shell.execute_reply":"2022-07-20T09:49:15.123836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/nlp-getting-started/sample_submission.csv')\nsub['target'] = list(map(int, preds))\nsub.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-20T09:49:20.197591Z","iopub.execute_input":"2022-07-20T09:49:20.197948Z","iopub.status.idle":"2022-07-20T09:49:20.219067Z","shell.execute_reply.started":"2022-07-20T09:49:20.197918Z","shell.execute_reply":"2022-07-20T09:49:20.218163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}