{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# install needed libraries","metadata":{"id":"O8QMjPON0GxI"}},{"cell_type":"code","source":"!pip install camel-tools","metadata":{"id":"v3pC02r8tbLs","outputId":"3e590e3e-7b3e-4eb4-94d8-00cc50fc3d87"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from google.colab import drive\ndrive.mount('/content/drive')","metadata":{"id":"9qDtg43N0ds2","outputId":"8d990de6-53c8-4755-e52a-68d0fb635aac"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Import Important Packages and Sentiment Analysis Libraries\n","metadata":{"id":"7koor0Pnxgu_"}},{"cell_type":"code","source":"# import libraries\nimport numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nfrom sklearn.metrics import accuracy_score, classification_report, accuracy_score , confusion_matrix\n","metadata":{"id":"gULAk0o_LT_H"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load data into csv\ndf = pd.read_csv(\"/Kaggle/train.csv\", encoding =\"utf-8\")","metadata":{"id":"Tp9mnO5J5B-J"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df ","metadata":{"id":"K4MqkYH6Lv9v","outputId":"f07d01de-861e-4a55-cd0f-09ccdd79141f"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load data into csv\ndf_test_kaggle = pd.read_csv(\"/Kaggle/test.csv\", encoding =\"utf-8\")","metadata":{"id":"gc6aWlUvmGQe"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_kaggle ","metadata":{"id":"e0daj0GL5ODk","outputId":"aa4200d6-bb89-44da-8c69-4e02ad32d18a"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re \nimport string\nfrom camel_tools.utils.dediac import dediac_ar\nfrom camel_tools.utils.normalize import normalize_unicode\nfrom camel_tools.utils.normalize import normalize_alef_maksura_ar\nfrom camel_tools.utils.normalize import normalize_alef_ar\nfrom camel_tools.utils.normalize import normalize_teh_marbuta_ar\nimport arabicstopwords.arabicstopwords as stp","metadata":{"id":"l4nAzv0P_IU3"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"arabic_punctuations = '''`÷×؛<>_()*&^%][ـ،/:\"؟.,'{}~¦+|!”…“–ـ'''\ndef clean(text):\n\n  '''\n  Clean input text form urls, handles, tabs, line jumps, and extra white spaces\n  '''\n  \n  text = re.sub(r\"[0-9]+\", \" \", text)#remove numbers\n  # text = re.sub(r\"[\\.\\,\\#_\\|\\:\\?\\?\\/\\=]\", \" \", text)# remove special characters\n  text = re.sub('['+string.punctuation+']', ' ', text)# remove english punctionation \n  #remove Arabic question mark\n  text = text.replace(\"؟\",\"\")\n  text = re.sub(r\"\\t\", \" \", text)  # remove tabs\n  text = re.sub(r\"\\n\", \" \", text)  # remove line jump\n  text = re.sub(r\"\\s+\", \" \", text)  # remove extra white space\n  text = text.strip()\n  return text\n\n\ndef normalize_camel(text):\n  # Normalize alef variants to 'ا'\n    text = normalize_alef_ar(text)\n  # Normalize alef maksura 'ى' to yeh 'ي'\n    text = normalize_alef_maksura_ar(text)\n  # Normalize teh marbuta 'ة' to heh 'ه'\n    text = normalize_teh_marbuta_ar(text)\n    return(text)\n# remove arabic punctioations\ndef remove_punctuations(text):\n    translator = str.maketrans('', '', arabic_punctuations)\n    return text.translate(translator)\n\n# perform all preprocessing steps\ndef preprocess(sentence):\n  # apply preprocessing steps on the given sentence\n  sentence = clean(sentence)\n  sentence = dediac_ar(sentence)\n  sentence = normalize_unicode(sentence)\n  sentence = normalize_camel(sentence)\n  sentence = remove_punctuations(sentence)\n  return sentence","metadata":{"id":"EfO4xNLX-jnh"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"text_cleaned_1\"]= df[\"GroundTruthText\"].apply(lambda x:   preprocess(x))","metadata":{"id":"lgLxn9H1tu8d"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"id":"LZVH4-a0uSRa","outputId":"c44c68c8-fb46-406b-d2d2-f48552558aac"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install transformers","metadata":{"id":"YRGLo7EmxHKD","outputId":"e263b5a3-0dc1-4710-ccfd-5d3ca898ab6a"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#An IPython magic extension for printing date and time stamps, version numbers, and hardware information.\n!pip install -q -U watermark","metadata":{"id":"KxNtZBUhxQIX","outputId":"a7de0ec2-e23f-4f1e-e619-01b0e844b580"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%reload_ext watermark\n%watermark -v -p numpy,pandas,torch,transformers","metadata":{"id":"xBEN2HDVxWba","outputId":"06936981-eefd-4ad0-847e-e8053587dca9"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Setup & Config\nimport transformers\nfrom transformers import BertModel, BertTokenizer, AdamW, get_linear_schedule_with_warmup\nfrom transformers import AutoTokenizer, AutoModel\nimport torch\n\n\nimport numpy as np\nimport pandas as pd\nimport seaborn as sns\nimport re\nimport seaborn as sns\nimport time\nfrom pylab import rcParams\nimport matplotlib.pyplot as plt\nfrom matplotlib import rc\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import (confusion_matrix, classification_report, accuracy_score,\n                             f1_score, make_scorer, roc_auc_score)\nfrom collections import defaultdict\nfrom textwrap import wrap\n\nfrom torch import nn, optim\nfrom torch.utils.data import Dataset, DataLoader\nimport torch.nn.functional as F\n\nfrom tqdm import tqdm\n\n%matplotlib inline\n%config InlineBackend.figure_format='retina'\n\nsns.set(style='whitegrid', palette='muted', font_scale=1.2)\n\nHAPPY_COLORS_PALETTE = [\"#01BEFE\", \"#FFDD00\", \"#FF7D00\", \"#FF006D\", \"#ADFF02\", \"#8F00FF\"]\n\nsns.set_palette(sns.color_palette(HAPPY_COLORS_PALETTE))\n\nrcParams['figure.figsize'] = 12, 8\n\nRANDOM_SEED = 42\nnp.random.seed(RANDOM_SEED)\ntorch.manual_seed(RANDOM_SEED)\n\ndevice = torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\")\ndevice","metadata":{"id":"MtOyfZSaxXZP","outputId":"7bfeb8ec-aada-45ed-c84f-86f61209168c"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(df.SpeakerDialect)\nplt.xlabel('labels');","metadata":{"id":"OBZht2tkxl37","outputId":"aea66dae-c2d9-4aff-c817-dc00a900ca78"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"{\n'Najdi':1\n'Hijazi':2\n'Khaliji':3\n'ModernStandardArabic':4\n}","metadata":{"id":"Dh7svboWFAMY"}},{"cell_type":"code","source":"#(Najdi, Hijazi, Khaliji, ModernStandardArabic)\ndef get_dialect(label):\n    if label== \"Najdi\":\n        return 0\n    elif label== \"Hijazi\":\n        return 1\n    elif label== \"Khaliji\":\n        return 2 \n    elif label== \"ModernStandardArabic\":\n        return 3 \n    \n\ndf['label'] = df.SpeakerDialect.apply(get_dialect)\ndf.head(2)","metadata":{"id":"e1Q2flxlHhYR","outputId":"c8a57b39-dab5-4a1f-f3b4-b0b4157f7800"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(df.label)\nplt.xlabel('labels');","metadata":{"id":"vqzIblyJ7OS2","outputId":"81cd04fe-b3c3-49dd-b685-b5b752be93cc"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Preprocessing\n\nYou might already know that Machine Learning models don't work with raw text. You need to convert text to numbers (of some sort). BERT requires even more attention (good one, right?). Here are the requirements: \n\n- Add special tokens to separate sentences and do classification\n- Pass sequences of constant length (introduce padding)\n- Create array of 0s (pad token) and 1s (real token) called *attention mask*\n\nThe Transformers library provides (you've guessed it) a wide variety of Transformer models (including BERT). It works with TensorFlow and PyTorch! It also includes prebuild tokenizers that do the heavy lifting for us!\n","metadata":{"id":"rDeZLsA3sga5"}},{"cell_type":"code","source":"#PRE_TRAINED_MODEL_NAME = \"UBC-NLP/MARBERT\"\nPRE_TRAINED_MODEL_NAME = 'Ammar-alhaj-ali/arabic-MARBERT-dialect-identification-city'\n#PRE_TRAINED_MODEL_NAME = \"bashar-talafha/multi-dialect-bert-base-arabi\"\n#MARBERT = \"UBC-NLP/MARBERT\"\n#ARBERT = \"UBC-NLP/ARBERT\"","metadata":{"id":"ZtOnnCPjp11_"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"### Train , val and test data \nfrom sklearn.model_selection import train_test_split\nRANDOM_SEED = 42\ntrain_df , test_df  = train_test_split(df, test_size = 0.3, random_state = RANDOM_SEED, shuffle = True)\nval_df , test_df  = train_test_split(df, test_size = 0.5, random_state = RANDOM_SEED, shuffle = True)\n\nprint(train_df.shape, test_df.shape , val_df.shape)","metadata":{"id":"bxtehkETtiEX","outputId":"a7814123-2c78-49af-ea9e-71ca9c1729d5"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = df_test_kaggle\nprint (test_df.shape)","metadata":{"id":"QbG5xTpCi02z"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's load a pre-trained [BertTokenizer](https://huggingface.co/transformers/model_doc/bert.html#berttokenizer):","metadata":{"id":"QI1I9UbPsoCx"}},{"cell_type":"code","source":"tokenizer =AutoTokenizer.from_pretrained(PRE_TRAINED_MODEL_NAME)","metadata":{"id":"glSrUGH59VCU","outputId":"43ed981d-3d12-4742-ecea-81aae86ff1cb"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We'll use this text to understand the tokenization process:\nSome basic operations can convert the text to tokens and tokens to unique integers (ids):","metadata":{"id":"wmRsMQnZsuAq"}},{"cell_type":"code","source":"sample_text = \"لا أحب أكل التفاح وأحب أكل المانجو والبطيخ\"\n               \ntokens = tokenizer.tokenize(sample_text)\nids = tokenizer.convert_tokens_to_ids(tokens)\nprint(f'{sample_text}')\nprint('='*60)\nprint(tokens)\nprint('='*60)\nprint(ids)","metadata":{"id":"O9-KKh_pCglS","outputId":"98793dcb-330c-4957-fc0d-e88f3b97cf81"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# special tokens\nprint(tokenizer.sep_token , tokenizer.sep_token_id)\nprint(tokenizer.cls_token,tokenizer.cls_token_id)\n\nprint(tokenizer.pad_token,tokenizer.pad_token_id)","metadata":{"id":"k1uqusF5CuNq","outputId":"b2176a3e-3e7c-49c5-f397-cf34592f9ed6"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"All of that work can be done using the [`encode_plus()`](https://huggingface.co/transformers/main_classes/tokenizer.html#transformers.PreTrainedTokenizer.encode_plus) method:","metadata":{"id":"ZHhRxEVhs8ic"}},{"cell_type":"code","source":"\n\nencoding = tokenizer.encode_plus(\n      sample_text,\n      add_special_tokens=True,\n      max_length=128,\n      truncation=True,\n      return_token_type_ids=False,\n      pad_to_max_length=True,\n      return_attention_mask=True,\n      return_tensors='pt')\n\n","metadata":{"id":"J8F9nqp2Cz58","outputId":"1f52ab75-6f66-4fdc-8121-0834c5d31e46"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The token ids are now stored in a Tensor and padded to a length of 128:","metadata":{"id":"2ig1dRGztB0h"}},{"cell_type":"code","source":"print(encoding['input_ids'].shape)\nencoding['input_ids']","metadata":{"id":"JcLspOyXC23b","outputId":"c4710e38-0151-480d-fac0-c47128833203"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(encoding['attention_mask'].shape)\nencoding['attention_mask']","metadata":{"id":"PMFdjRfPC6Ep","outputId":"5089a65a-ce7d-46a0-f61e-357540cfe227"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Choosing Sequence Length\n\nBERT works with fixed-length sequences. We'll use a simple strategy to choose the max length. Let's store the token length of each review:","metadata":{"id":"J1LQaGwEtKme"}},{"cell_type":"code","source":"token_lens = []\n\nfor txt in df.text_cleaned_1:\n  tokens = tokenizer.encode(txt, truncation=True , max_length=512)\n  token_lens.append(len(tokens))","metadata":{"id":"Xnopwt-p9sBi"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"and plot the distribution:","metadata":{"id":"_Opu7MDXtQDT"}},{"cell_type":"code","source":"sns.distplot(token_lens)\nplt.xlim([0, 256]);\nplt.xlabel('Token count');","metadata":{"id":"9DVbLwxZCGyp","outputId":"e82d1369-c621-4d96-e80b-4e67dedadc60"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Most of the tweets seem to contain less than 55 tokens, but we'll be on the safe side and choose a maximum length of 55.","metadata":{"id":"nFcXhefMtUQs"}},{"cell_type":"code","source":"MAX_LEN = 50","metadata":{"id":"uGzwvLX7COGc"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have all building blocks required to create a PyTorch dataset. Let's do it:","metadata":{"id":"y6PM78ZTttZn"}},{"cell_type":"code","source":"class DialectDataset(Dataset):\n\n  def __init__(self, sentences, targets, tokenizer, max_len):\n    self.sentences = sentences\n    self.targets = targets\n    self.tokenizer = tokenizer\n    self.max_len = max_len\n  \n  def __len__(self):\n    return len(self.sentences)\n  \n  def __getitem__(self, item):\n    sentence = str(self.sentences[item])\n    target = self.targets[item]\n\n    encoding = self.tokenizer.encode_plus(\n      sentence,\n      add_special_tokens=True,\n      max_length=self.max_len,\n      return_token_type_ids=False,\n      pad_to_max_length=True,\n      return_attention_mask=True,\n      return_tensors='pt',\n      padding='max_length',\n      truncation=True,\n      \n\n    )\n\n    return {\n      'sentence_text': sentence,\n      'input_ids': encoding['input_ids'].flatten(),\n      'attention_mask': encoding['attention_mask'].flatten(),\n      'targets': torch.tensor(target, dtype=torch.long)\n    }","metadata":{"id":"T14pYLUDCTyG"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_data_loader(df, tokenizer, max_len, batch_size):\n  ds = DialectDataset(\n    sentences=df.text_cleaned_1.to_numpy(),\n    targets=df.label.to_numpy(),\n    tokenizer=tokenizer,\n    max_len=max_len\n  )\n\n  return DataLoader(\n    ds,\n    batch_size=batch_size,\n    num_workers=2 \n  )","metadata":{"id":"I9mgWqK4DPa1"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 16\n\ntrain_data_loader = create_data_loader(train_df, tokenizer, MAX_LEN, BATCH_SIZE)\nval_data_loader = create_data_loader(val_df, tokenizer, MAX_LEN, BATCH_SIZE)\ntest_data_loader = create_data_loader(test_df , tokenizer, MAX_LEN, BATCH_SIZE)","metadata":{"id":"k20cveQjDVcd"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_kaggle","metadata":{"id":"WuxS_KinjLx3"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_loader_ = create_data_loader(df_test_kaggle , tokenizer, MAX_LEN, BATCH_SIZE)","metadata":{"id":"qSlaYSDqkkXz"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.text_cleaned_1.to_numpy()","metadata":{"id":"Xn-vCczvGqk8","outputId":"bc0f3a74-99c5-4a48-edc3-886b7ef1ab1c"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = next(iter(train_data_loader))\ndata.keys()","metadata":{"id":"u9s_zXbZGwrI","outputId":"8e60b9d7-0e0d-4618-c8cb-96e4423dbcd1"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bert_model = AutoModel.from_pretrained(PRE_TRAINED_MODEL_NAME, return_dict=False)","metadata":{"id":"4KF-Aun-HART","outputId":"bb32816a-ebef-4066-dbf2-3911c78e7c85"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dialect Classification with BERT and Hugging Face","metadata":{"id":"0aASJybPt28Z"}},{"cell_type":"code","source":"class DialectClassifier(nn.Module):\n\n  def __init__(self, n_classes):\n    super(DialectClassifier, self).__init__()\n    self.bert = AutoModel.from_pretrained(PRE_TRAINED_MODEL_NAME)\n    self.relu = nn.ReLU()\n    self.drop = nn.Dropout(p=0.3)\n    self.out = nn.Linear(self.bert.config.hidden_size, n_classes)\n    #self.softmax = nn.Softmax(dim = 1)\n  \n  def forward(self, input_ids, attention_mask):\n    _, pooled_output = self.bert(\n      input_ids=input_ids,\n      attention_mask=attention_mask,\n      return_dict=False\n    )\n    output = self.drop(pooled_output)\n    return self.out(output)","metadata":{"id":"Xorrfe7eI1Rz"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Najdi, Hijazi, Khaliji, ModernStandardArabic\nclass_names = [\"Najdi\",'Hijazi', 'Khaliji', 'ModernStandardArabic']\nprint(len(class_names))\nmodel = DialectClassifier(len(class_names))\nmodel = model.to(device)","metadata":{"id":"-DywXSS1JzxL","outputId":"df77f057-d1f8-4e37-aa28-f27d8460ed18"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"input_ids = data['input_ids'].to(device)\nattention_mask = data['attention_mask'].to(device)\n\nprint(input_ids.shape) # batch size x seq length\nprint(attention_mask.shape) # batch size x seq length","metadata":{"id":"-xjVc9xGK232","outputId":"b51f56a6-4064-4495-df2a-e7af44b1bf86"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To get the predicted probabilities from our trained model, we'll apply the softmax function to the outputs:","metadata":{"id":"aKciDIWv6Uij"}},{"cell_type":"code","source":"F.softmax(model(input_ids, attention_mask), dim=1)","metadata":{"id":"1xuqfY7bLGny","outputId":"ee231ef2-8b54-4cb4-fda3-11c75b0e4c6a"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training","metadata":{"id":"qzcqh4VOuFOe"}},{"cell_type":"markdown","source":"To reproduce the training procedure from the BERT paper, we'll use the [AdamW](https://huggingface.co/transformers/main_classes/optimizer_schedules.html#adamw) optimizer provided by Hugging Face. It corrects weight decay, so it's similar to the original paper. We'll also use a linear scheduler with no warmup steps:","metadata":{"id":"HpwkHB4PuDlB"}},{"cell_type":"markdown","source":"How do we come up with all hyperparameters? The BERT authors have some recommendations for fine-tuning:\n\n- Batch size: 16, 32\n- Learning rate (Adam): 5e-5, 3e-5, 2e-5\n- Number of epochs: 2, 3, 4\n\n\nWe're going to ignore the number of epochs recommendation but stick with the rest. Note that increasing the batch size reduces the training time significantly, but gives you lower accuracy.\n\n","metadata":{"id":"_BvWW5Wfx0nM"}},{"cell_type":"code","source":"#Compute class weights for the labels in the dataset and then pass these weights to the loss function so that it takes care of the class imbalance. In PyTorch it can be done as shown below:\nfrom sklearn.utils.class_weight import compute_class_weight\n\n#compute the class weights\nclass_weights = compute_class_weight(class_weight = 'balanced', classes= np.unique(train_df.label.values), y= train_df.label.values)\n\nprint(\"Class Weights:\",class_weights)\n\n# converting list of class weights to a tensor\nweights= torch.tensor(class_weights,dtype=torch.float)\nprint(weights)","metadata":{"id":"AFi4QaaN87h3","outputId":"a67e0f3d-2bae-4777-ce37-922174168559"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS = 8\n# implementation of Adam algorithm with weight K fix , the same as the original Bert paper\noptimizer = AdamW(model.parameters(), lr=2e-5, correct_bias=False, no_deprecation_warning=True)\ntotal_steps = len(train_data_loader) * EPOCHS\n# linear schedule which is going to decay the warning rate, no warmup steps just as Bert paper recommend\nscheduler = get_linear_schedule_with_warmup(\n  optimizer,\n  num_warmup_steps=0,\n  num_training_steps=total_steps\n)\n#define the loss function which cross entropy loss since it is a classification and putting it into GPU\nloss_fn = nn.CrossEntropyLoss(weight=weights).to(device)","metadata":{"id":"U-qZ9ovA9Xsu"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's continue with writing a helper/custom function for training our model for one epoch:","metadata":{"id":"7W6Hl7WOCIQK"}},{"cell_type":"code","source":"# \ndef train_epoch( model, data_loader, loss_fn, optimizer,  device,  scheduler,  n_examples):\n  model = model.train()\n\n  losses = []\n  correct_predictions = 0\n  \n  for d in data_loader:\n    input_ids = d[\"input_ids\"].to(device)\n    attention_mask = d[\"attention_mask\"].to(device)\n    targets = d[\"targets\"].to(device)\n\n    outputs = model(\n      input_ids=input_ids,\n      attention_mask=attention_mask\n    )\n\n    _, preds = torch.max(outputs, dim=1)\n    loss = loss_fn(outputs, targets) \n\n    correct_predictions += torch.sum(preds == targets)\n    losses.append(loss.item())\n    loss.backward()\n    nn.utils.clip_grad_norm_(model.parameters(), max_norm=1.0)\n    \n    optimizer.step()\n    # update for wights \n    scheduler.step()\n    # for gradiant to be zero to prevent update on the previous grad\n    optimizer.zero_grad()\n\n  return correct_predictions.double() / n_examples, np.mean(losses)","metadata":{"id":"FxN6_Hu2OTDP"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Training the model should look familiar, except for two things. The scheduler gets called every time a batch is fed to the model. We're avoiding exploding gradients by clipping the gradients of the model using [clip_grad_norm_](https://pytorch.org/docs/stable/nn.html#clip-grad-norm).\n\nLet's write another one that helps us evaluate the model on a given data loader:","metadata":{"id":"N_5O-rKiyK4A"}},{"cell_type":"code","source":"def eval_model(model, data_loader, loss_fn, device, n_examples):\n  model = model.eval()\n\n  losses = []\n  correct_predictions = 0\n\n  with torch.no_grad():# don't update gradients\n    for d in data_loader:\n      input_ids = d[\"input_ids\"].to(device)\n      attention_mask = d[\"attention_mask\"].to(device)\n      targets = d[\"targets\"].to(device)\n\n      outputs = model(\n        input_ids=input_ids,\n        attention_mask=attention_mask\n      )\n      _, preds = torch.max(outputs, dim=1)\n\n      loss = loss_fn(outputs, targets)\n\n      correct_predictions += torch.sum(preds == targets)\n      losses.append(loss.item())\n\n  return correct_predictions.double() / n_examples, np.mean(losses)","metadata":{"id":"q8Xocok5OVnv"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\")\nmodel.cuda()","metadata":{"id":"PCYIWjr_hu4x","outputId":"efa95ab3-4529-4f47-ac5e-cf5b7650671b"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Using those two, we can write our training loop:","metadata":{"id":"9ZVf9S2jyaWR"}},{"cell_type":"code","source":"%%time\n\nhistory = defaultdict(list)\nbest_accuracy = 0\n\nfor epoch in range(EPOCHS):\n\n  print(f'Epoch {epoch + 1}/{EPOCHS}')\n  print('-' * 2)\n\n  train_acc, train_loss = train_epoch(model, train_data_loader, loss_fn, optimizer, device, scheduler, len(train_df))\n\n  print(f'Train loss {train_loss} accuracy {train_acc}')\n\n  val_acc, val_loss = eval_model( model, val_data_loader, loss_fn, device,  len(val_df))\n\n  print(f'Val   loss {val_loss} accuracy {val_acc}')\n  print()\n\n  history['train_acc'].append(train_acc)\n  history['train_loss'].append(train_loss)\n  history['val_acc'].append(val_acc)\n  history['val_loss'].append(val_loss)\n\n  if val_acc > best_accuracy:\n    torch.save(model.state_dict(), 'best_model_state_8_epoch.bin')\n    best_accuracy = val_acc","metadata":{"id":"EwM1FleIOe01","outputId":"20532fa4-40a3-4083-f97d-166087db5718"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Uncomment the next cell to download my pre-trained model:","metadata":{"id":"9wR58WloyrXw"}},{"cell_type":"markdown","source":"## Evaluation\n\nSo how good is our model on predicting sentiment? Let's start by calculating the accuracy on the test data:","metadata":{"id":"8JV9NVexyxNO"}},{"cell_type":"code","source":"test_acc, _ = eval_model(model, test_data_loader, loss_fn, device, len(test_df))\n\ntest_acc.item()","metadata":{"id":"jGrQ_myMOoI0","outputId":"6c951e2e-d577-45a9-f25d-8019c040f40c"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We'll define a helper function to get the predictions from our model:","metadata":{"id":"iPYex5sxy34t"}},{"cell_type":"code","source":"def get_predictions(model, data_loader):\n  model = model.eval()\n  \n  texts = []\n  predictions = []\n  prediction_probs = []\n  real_values = []\n\n  with torch.no_grad():\n    for d in data_loader:\n\n      texts = d[\"sentence_text\"]\n      input_ids = d[\"input_ids\"].to(device)\n      attention_mask = d[\"attention_mask\"].to(device)\n      targets = d[\"targets\"].to(device)\n\n      outputs = model(\n        input_ids=input_ids,\n        attention_mask=attention_mask\n      )\n      _, preds = torch.max(outputs, dim=1)\n\n      probs = F.softmax(outputs, dim=1)\n\n      texts.extend(texts)\n      predictions.extend(preds)\n      prediction_probs.extend(probs)\n      real_values.extend(targets)\n\n  predictions = torch.stack(predictions).cpu()\n  prediction_probs = torch.stack(prediction_probs).cpu()\n  real_values = torch.stack(real_values).cpu()\n  return texts, predictions, prediction_probs, real_values","metadata":{"id":"AIP-0tipVSdk"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This is similar to the evaluation function, except that we're storing the text of the reviews and the predicted probabilities (by applying the softmax on the model outputs):","metadata":{"id":"9adlFhOTzBUY"}},{"cell_type":"code","source":"y_texts, y_pred, y_pred_probs, y_test = get_predictions(\n  model,\n  test_data_loader\n)","metadata":{"id":"5YMEvooHzCwR"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_kaggle[\"label_2\"]= y_pred_.tolist()","metadata":{"id":"7NAlkXOZoHW2"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#(Najdi, Hijazi, Khaliji, ModernStandardArabic)\ndef get_dialect(label):\n    if label== 0:\n        return \"Najdi\" \n    elif label== 1:\n        return \"Hijazi\" \n    elif label== 2 :\n        return \"Khaliji\"\n    elif label== 3 :\n        return \"ModernStandardArabic\" \n    \n\ndf_test_kaggle['label_pre'] = df_test_kaggle.label_2.apply(get_dialect)\ndf_test_kaggle.head(2)","metadata":{"id":"6bQ4GVVulBxE","outputId":"ed722019-a649-4909-8b11-d964f5ce61b7"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#(Najdi, Hijazi, Khaliji, ModernStandardArabic)\ndef get_dialect_2(label):\n    if label== \"Najdi\":\n        return 1 \n    elif label== \"Hijazi\":\n        return 2 \n    elif label== \"Khaliji\":\n        return 3 \n    elif label== \"ModernStandardArabic\":\n        return 4\n    \n\ndf_test_kaggle['SpeakerDialect'] = df_test_kaggle.label_pre.apply(get_dialect_2)\ndf_test_kaggle.head(2)","metadata":{"id":"xxpGQnt6lmju","outputId":"793f133c-836d-4382-84d1-bf9cec255377"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_kaggle[[\"SegmentID\",\"SpeakerDialect\"]].to_csv(\"/Kaggle compition/result_2_test.csv\",encoding=\"utf-8\", index=False)","metadata":{"id":"TZhnTzQamTbC","outputId":"d052648b-8b64-466c-9b95-0b28908b31bc"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_kaggle[[\"SegmentID\",\"SpeakerDialect\"]].to_csv(\"/Kaggle/result_2_test.csv\",encoding=\"utf-8\", index=False)","metadata":{"id":"uqnpky2qmvyf"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_kaggle.to_csv(\"/Kaggle/result_1_test.csv\", encoding=\"utf-8\", index+False)","metadata":{"id":"wO0j-tVSotMB","outputId":"d8d87c5f-d88c-4ebb-f517-5fca98ca964f"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_kaggle.to_csv(\"/Kaggle/result_1_test.csv\", encoding=\"utf-8\", index=False)","metadata":{"id":"redb1grKo_Jv"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's have a look at the classification report","metadata":{"id":"xiX9rbdRzJ6p"}},{"cell_type":"code","source":"print(classification_report(y_test, y_pred, target_names=class_names))","metadata":{"id":"TiUpPNx3jhc3","outputId":"4a40affe-ab3c-4e62-c25a-6d1cd41cc443"},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred","metadata":{"id":"AyZ2uspnl_XO","outputId":"7a6db576-d62b-4863-b601-5135a0aae07a"},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **References**\n\n\n\n*   [CAMeL Tools.](https://github.com/CAMeL-Lab/camel_tools)\n\n\n*   [Vader](https://pypi.org/project/vaderSentiment/)\n\n\n*   [what is training warmup steps](https://datascience.stackexchange.com/questions/55991/in-the-context-of-deep-learning-what-is-training-warmup-steps)\n\n- [BERT: Pre-training of Deep Bidirectional Transformers for Language Understanding](https://arxiv.org/abs/1810.04805)\n- [L11 Language Models - Alec Radford (OpenAI)](https://www.youtube.com/watch?v=BnpB3GrpsfM)\n- [The Illustrated BERT, ELMo, and co.](https://jalammar.github.io/illustrated-bert/)\n- [BERT Fine-Tuning Tutorial with PyTorch](https://mccormickml.com/2019/07/22/BERT-fine-tuning/)\n- [How to Fine-Tune BERT for Text Classification?](https://arxiv.org/pdf/1905.05583.pdf)\n- [Huggingface Transformers](https://huggingface.co/transformers/)\n- [BERT Explained: State of the art language model for NLP](https://towardsdatascience.com/bert-explained-state-of-the-art-language-model-for-nlp-f8b21a9b6270)","metadata":{"id":"Jf8L3Knu8EL5"}},{"cell_type":"code","source":"","metadata":{"id":"P7efQm2Zy1Vn"},"execution_count":null,"outputs":[]}]}