{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"This is a Starter notebook to train a BERT model on this competition, a lot of work needed to tune the notebook to better scores! But I hope this will help othrers \"get thier feet wet\" and try to learn and have fun :-)","metadata":{}},{"cell_type":"markdown","source":"Part Zero: GPU Or CPU ?","metadata":{}},{"cell_type":"code","source":"import torch\n\n# If there's a GPU available...\nif torch.cuda.is_available():    \n\n    # Tell PyTorch to use the GPU.    \n    device = torch.device(\"cuda\")\n\n    print('There are %d GPU(s) available.' % torch.cuda.device_count())\n\n    print('We will use the GPU:', torch.cuda.get_device_name(0))\n    !nvidia-smi\n\n# If not...\nelse:\n    print('No GPU available, using the CPU instead.')\n    device = torch.device(\"cpu\")\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-02-25T02:41:10.048264Z","iopub.execute_input":"2023-02-25T02:41:10.048720Z","iopub.status.idle":"2023-02-25T02:41:13.874689Z","shell.execute_reply.started":"2023-02-25T02:41:10.048679Z","shell.execute_reply":"2023-02-25T02:41:13.873072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Part One: Installing needed packages","metadata":{}},{"cell_type":"code","source":"#!pip install pyarabic\n#!pip install optuna==2.3.0\n#!pip install transformers==4.2.1\n#!pip install tokenizers==0.9.4","metadata":{"execution":{"iopub.status.busy":"2023-02-25T02:41:13.877903Z","iopub.execute_input":"2023-02-25T02:41:13.878851Z","iopub.status.idle":"2023-02-25T02:41:25.418002Z","shell.execute_reply.started":"2023-02-25T02:41:13.878803Z","shell.execute_reply":"2023-02-25T02:41:25.416681Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Part Two: Import needed packages","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pyarabic.araby as ar\n\nimport re, functools, operator, string\nimport torch , optuna, gc, random, os\n\nfrom tqdm import tqdm_notebook as tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, accuracy_score, f1_score, confusion_matrix, precision_score , recall_score\nfrom transformers import AutoConfig, AutoModelForSequenceClassification, AutoTokenizer\nfrom transformers.data.processors import SingleSentenceClassificationProcessor\nfrom transformers import Trainer , TrainingArguments\nfrom transformers.trainer_utils import EvaluationStrategy\nfrom transformers.data.processors.utils import InputFeatures\nfrom torch.utils.data import Dataset\nfrom torch.utils.data import DataLoader\nfrom sklearn.utils import resample\n\nimport logging\n\nlogging.basicConfig(level=logging.WARNING)\nlogger = logging.getLogger(__name__)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-25T02:41:25.420170Z","iopub.execute_input":"2023-02-25T02:41:25.420825Z","iopub.status.idle":"2023-02-25T02:41:34.657028Z","shell.execute_reply.started":"2023-02-25T02:41:25.420776Z","shell.execute_reply":"2023-02-25T02:41:34.655919Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Part Three: Define cleaning method","metadata":{}},{"cell_type":"code","source":"def data_cleaning (text):\n  try:\n    text = re.sub(r'^https?:\\/\\/.*[\\r\\n]*', '', text, flags=re.MULTILINE)\n    text = re.sub(r'^http?:\\/\\/.*[\\r\\n]*', '', text, flags=re.MULTILINE)\n    text = re.sub(r\"http\\S+\", \"\", text)\n    text = re.sub(r\"https\\S+\", \"\", text)\n    text = re.sub(r'\\s+', ' ', text)\n    text = re.sub(\"(\\s\\d+)\",\"\",text) \n    text = re.sub(r\"$\\d+\\W+|\\b\\d+\\b|\\W+\\d+$\", \"\", text)\n    text = re.sub(\"\\d+\", \" \", text)\n    text = ar.strip_tashkeel(text)\n    text = ar.strip_tatweel(text)\n    text = text.replace(\"#\", \" \");\n    text = text.replace(\"@\", \" \");\n    text = text.replace(\"_\", \" \");\n    translator = str.maketrans('', '', string.punctuation)\n    text = text.translate(translator)\n    text = text.replace(\"آ\", \"ا\")\n    text = text.replace(\"إ\", \"ا\")\n    text = text.replace(\"أ\", \"ا\")\n    text = text.replace(\"ؤ\", \"و\")\n    text = text.replace(\"ئ\", \"ي\")\n  except:\n    return text\n   \n  return text\n","metadata":{"execution":{"iopub.status.busy":"2023-02-25T02:42:58.787778Z","iopub.execute_input":"2023-02-25T02:42:58.788592Z","iopub.status.idle":"2023-02-25T02:42:58.798684Z","shell.execute_reply.started":"2023-02-25T02:42:58.788552Z","shell.execute_reply":"2023-02-25T02:42:58.797409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Part Four: Prepare Training Set","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(\"/kaggle/input/ml-olympiad-dialectrecognition/train.csv\", sep=\",\")\ntrain_data","metadata":{"execution":{"iopub.status.busy":"2023-02-25T02:43:59.282481Z","iopub.execute_input":"2023-02-25T02:43:59.282941Z","iopub.status.idle":"2023-02-25T02:44:01.297760Z","shell.execute_reply.started":"2023-02-25T02:43:59.282899Z","shell.execute_reply":"2023-02-25T02:44:01.296653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_dict = {\"ProcessedText\": train_data['ProcessedText'], \"SpeakerDialect\":train_data['SpeakerDialect'] }\ntrain_data_new = pd.DataFrame.from_dict(train_data_dict)\ntrain_data_new","metadata":{"execution":{"iopub.status.busy":"2023-02-25T02:44:25.400889Z","iopub.execute_input":"2023-02-25T02:44:25.401269Z","iopub.status.idle":"2023-02-25T02:44:25.422727Z","shell.execute_reply.started":"2023-02-25T02:44:25.401236Z","shell.execute_reply":"2023-02-25T02:44:25.421007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data_new['SpeakerDialect'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-02-25T02:44:40.254124Z","iopub.execute_input":"2023-02-25T02:44:40.254974Z","iopub.status.idle":"2023-02-25T02:44:40.271830Z","shell.execute_reply.started":"2023-02-25T02:44:40.254892Z","shell.execute_reply":"2023-02-25T02:44:40.270706Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_new[\"ProcessedText\"] = train_data_new[\"ProcessedText\"].apply(lambda x:   data_cleaning(x))\ntrain_data_new","metadata":{"execution":{"iopub.status.busy":"2023-02-25T02:44:54.482311Z","iopub.execute_input":"2023-02-25T02:44:54.482697Z","iopub.status.idle":"2023-02-25T02:45:00.760896Z","shell.execute_reply.started":"2023-02-25T02:44:54.482663Z","shell.execute_reply":"2023-02-25T02:45:00.759392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Part Five: Spliting Training Data (Train , Evaluation)","metadata":{}},{"cell_type":"code","source":"# First setting the max_len , will be useful later for BERT Model\nExtra_Len = 6 # an extra padding in length , found to be useful for increasing F-score\nMax_Len = train_data[\"ProcessedText\"].str.split().str.len().max() + Extra_Len\n\nprint(Max_Len)\n\n#Spliting the Training data\nTest_Size = 0.15\nRand_Seed = 42 \n\ntrain_set, evaluation_set = train_test_split( train_data_new, test_size= Test_Size, random_state= Rand_Seed)\n\nprint(\"Train set: \")\nprint(train_set[\"SpeakerDialect\"].value_counts())\nprint(\"---------------------------\")\nprint (\"Evaluation set: \")\nprint (evaluation_set[\"SpeakerDialect\"].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-02-25T02:46:19.627501Z","iopub.execute_input":"2023-02-25T02:46:19.628196Z","iopub.status.idle":"2023-02-25T02:46:20.554928Z","shell.execute_reply.started":"2023-02-25T02:46:19.628153Z","shell.execute_reply":"2023-02-25T02:46:20.553675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Part Six: Preparing BERTModel Classes","metadata":{}},{"cell_type":"code","source":"Model_Used= \"aubmindlab/bert-base-arabertv02\"\nTask_Name = \"classification\"\n\nclass Dataset:\n    def __init__(\n        self,\n        name,\n        train,\n        test,\n        label_list,\n    ):\n        self.name = name\n        self.train = train\n        self.test = test\n        self.label_list = label_list\n        \nclass BERTModelDataset(Dataset):\n    def __init__(self, text, target, model_name, max_len, label_map):\n      super(BERTModelDataset).__init__()\n      self.text = text\n      self.target = target\n      self.tokenizer_name = model_name\n      self.tokenizer = AutoTokenizer.from_pretrained(model_name)\n      self.max_len = max_len\n      self.label_map = label_map\n  \n    def __len__(self):\n      return len(self.text)\n\n    def __getitem__(self,item):\n      text = str(self.text[item])\n      text = \" \".join(text.split())\n    \n      encoded_review = self.tokenizer.encode_plus(\n      text,\n      max_length= self.max_len,\n      add_special_tokens= True,\n      return_token_type_ids=False,\n      pad_to_max_length=True,\n      truncation='longest_first',\n      return_attention_mask=True,\n      return_tensors='pt'\n    )\n      input_ids = encoded_review['input_ids'].to(device)\n      attention_mask = encoded_review['attention_mask'].to(device)\n\n      return InputFeatures(input_ids=input_ids.flatten(), attention_mask=attention_mask.flatten(), label=self.label_map[self.target[item]])\n","metadata":{"execution":{"iopub.status.busy":"2023-02-25T02:50:02.776798Z","iopub.execute_input":"2023-02-25T02:50:02.777713Z","iopub.status.idle":"2023-02-25T02:50:02.788144Z","shell.execute_reply.started":"2023-02-25T02:50:02.777675Z","shell.execute_reply":"2023-02-25T02:50:02.786911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Part Seven: Defining Needed Methods for training and evaluation","metadata":{}},{"cell_type":"code","source":"def model_init():\n  return AutoModelForSequenceClassification.from_pretrained(Model_Used, return_dict=True, num_labels=len(label_map))\n\ndef compute_metrics(p): #p should be of type EvalPrediction\n  preds = np.argmax(p.predictions, axis=1)\n  assert len(preds) == len(p.label_ids)\n  print(classification_report(p.label_ids,preds))\n  #print(confusion_matrix(p.label_ids,preds))\n  macro_f1 = f1_score(p.label_ids,preds,average='macro')\n  macro_precision = precision_score(p.label_ids,preds,average='macro')\n  macro_recall = recall_score(p.label_ids,preds,average='macro')\n  acc = accuracy_score(p.label_ids,preds)\n  return {\n      'macro_f1' : macro_f1, \n      'macro_precision': macro_precision,\n      'macro_recall': macro_recall,\n      'accuracy': acc\n  }\n\ndef set_seed(seed):\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    np.random.seed(seed)\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-25T02:50:30.703652Z","iopub.execute_input":"2023-02-25T02:50:30.704408Z","iopub.status.idle":"2023-02-25T02:50:30.713218Z","shell.execute_reply.started":"2023-02-25T02:50:30.704367Z","shell.execute_reply":"2023-02-25T02:50:30.712007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Part Eight: Build Train and Evaluation Data Sets","metadata":{}},{"cell_type":"code","source":"label_list = list(train_set[\"SpeakerDialect\"].unique())\n\nprint(label_list)\nprint(train_set[\"SpeakerDialect\"].value_counts())\n\ndata_set = Dataset( \"OLY\", train_set, evaluation_set, label_list )\n\nlabel_map = { v:index for index, v in enumerate(label_list) }\nprint(label_map)\n\ntrain_dataset = BERTModelDataset(train_set[\"ProcessedText\"].to_list(),\n                                 train_set[\"SpeakerDialect\"].to_list(),Model_Used,int(Max_Len),label_map)\n\nevaluation_dataset = BERTModelDataset(evaluation_set[\"ProcessedText\"].to_list(),\n                                      evaluation_set[\"SpeakerDialect\"].to_list(),Model_Used,int(Max_Len),label_map)","metadata":{"execution":{"iopub.status.busy":"2023-02-25T02:51:00.403581Z","iopub.execute_input":"2023-02-25T02:51:00.403971Z","iopub.status.idle":"2023-02-25T02:51:14.631718Z","shell.execute_reply.started":"2023-02-25T02:51:00.403920Z","shell.execute_reply":"2023-02-25T02:51:14.630679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Part Nine: Definge Training Arguments","metadata":{}},{"cell_type":"code","source":"#define training arguments\ntraining_args = TrainingArguments(\"./train\")\ntraining_args.lr_scheduler_type = 'cosine'\ntraining_args.evaluate_during_training = True\ntraining_args.adam_epsilon =1e-8 \ntraining_args.learning_rate = 2e-05\ntraining_args.fp16 = True\ntraining_args.per_device_train_batch_size = 64\ntraining_args.per_device_eval_batch_size = 32\ntraining_args.gradient_accumulation_steps = 2\ntraining_args.num_train_epochs= 2\ntraining_args.warmup_steps = 0 \ntraining_args.evaluation_strategy = EvaluationStrategy.EPOCH\ntraining_args.seed = 42 \ntraining_args.disable_tqdm = False","metadata":{"execution":{"iopub.status.busy":"2023-02-25T02:58:04.976591Z","iopub.execute_input":"2023-02-25T02:58:04.977300Z","iopub.status.idle":"2023-02-25T02:58:04.988696Z","shell.execute_reply.started":"2023-02-25T02:58:04.977252Z","shell.execute_reply":"2023-02-25T02:58:04.987439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Part 10: Build Trainer","metadata":{}},{"cell_type":"code","source":"training_args.dataloader_pin_memory = False\ngc.collect()\ntorch.cuda.empty_cache()\nset_seed(Rand_Seed) \n\ntrainer = Trainer(\n    model = model_init(),\n    args = training_args,\n    train_dataset = train_dataset,\n    eval_dataset= evaluation_dataset,\n    compute_metrics=compute_metrics\n)\n\nprint(training_args.seed)","metadata":{"execution":{"iopub.status.busy":"2023-02-25T02:58:08.595392Z","iopub.execute_input":"2023-02-25T02:58:08.595936Z","iopub.status.idle":"2023-02-25T02:58:11.772090Z","shell.execute_reply.started":"2023-02-25T02:58:08.595893Z","shell.execute_reply":"2023-02-25T02:58:11.771004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Part 11: Train !","metadata":{}},{"cell_type":"code","source":"import os\nos.environ[\"WANDB_DISABLED\"] = \"true\"\ntrainer.train()","metadata":{"execution":{"iopub.status.busy":"2023-02-25T02:58:14.914075Z","iopub.execute_input":"2023-02-25T02:58:14.915003Z","iopub.status.idle":"2023-02-25T03:18:52.039577Z","shell.execute_reply.started":"2023-02-25T02:58:14.914922Z","shell.execute_reply":"2023-02-25T03:18:52.038679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Step 11: Prepare Test Data","metadata":{}},{"cell_type":"code","source":"test_data = pd.read_csv(\"/kaggle/input/ml-olympiad-dialectrecognition/test.csv\", sep=\",\")\ntest_data","metadata":{"execution":{"iopub.status.busy":"2023-02-25T03:20:47.492094Z","iopub.execute_input":"2023-02-25T03:20:47.492787Z","iopub.status.idle":"2023-02-25T03:20:47.545668Z","shell.execute_reply.started":"2023-02-25T03:20:47.492750Z","shell.execute_reply":"2023-02-25T03:20:47.544480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_dict = {\"SegmentID\":test_data['SegmentID'],\"ProcessedText\": test_data['ProcessedText']}\ntest_data_new = pd.DataFrame.from_dict(test_data_dict)\ntest_data_new","metadata":{"execution":{"iopub.status.busy":"2023-02-25T03:21:34.657782Z","iopub.execute_input":"2023-02-25T03:21:34.658173Z","iopub.status.idle":"2023-02-25T03:21:34.671970Z","shell.execute_reply.started":"2023-02-25T03:21:34.658139Z","shell.execute_reply":"2023-02-25T03:21:34.670802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_new[\"ProcessedText\"] = test_data_new[\"ProcessedText\"].apply(lambda x:   data_cleaning(x))\ntest_data_new","metadata":{"execution":{"iopub.status.busy":"2023-02-25T03:22:08.238940Z","iopub.execute_input":"2023-02-25T03:22:08.239639Z","iopub.status.idle":"2023-02-25T03:22:08.330896Z","shell.execute_reply.started":"2023-02-25T03:22:08.239601Z","shell.execute_reply":"2023-02-25T03:22:08.329908Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Part 11: Predict the values of the Testing Data","metadata":{}},{"cell_type":"code","source":"def predict(text, tokenizer):\n  try:\n  #print(text)\n    encoded_review = tokenizer.encode_plus(\n      text,\n      max_length=int(Max_Len),\n      add_special_tokens=True,\n      return_token_type_ids=False,\n      pad_to_max_length=True, #True,\n      truncation='longest_first',\n      return_attention_mask=True,\n      return_tensors='pt'\n    )\n\n    input_ids = encoded_review['input_ids'].to(device) #(input_ids + ([tokenizer.pad_token_id] * padding_length)).to(device)  \n    attention_mask = encoded_review['attention_mask'].to(device)\n    \n\n    output = trainer.model(input_ids, attention_mask)\n\n    _, prediction = torch.max(output[0], dim=1)\n    return prediction[0]\n  except:\n    return 0\n\ntokenizer = AutoTokenizer.from_pretrained(Model_Used)\n\nprediction_list = []\ni = 0\nfor element in test_data_new[\"ProcessedText\"]:\n    id = test_data[\"SegmentID\"][i]\n  \n    pre = predict(element,tokenizer)\n    pre_txt = label_list[pre]\n   \n    if pre_txt == 'Najdi': pre_txt = 1\n    if pre_txt == 'Hijazi': pre_txt = 2\n    if pre_txt == 'Khaliji': pre_txt = 3\n    if pre_txt == 'ModernStandardArabic': pre_txt = 4\n\n    \n    prediction_list.append(pre_txt)\n    \n    i = i + 1\n  \n#prediction_list","metadata":{"execution":{"iopub.status.busy":"2023-02-25T03:23:29.391876Z","iopub.execute_input":"2023-02-25T03:23:29.392587Z","iopub.status.idle":"2023-02-25T03:23:52.762297Z","shell.execute_reply.started":"2023-02-25T03:23:29.392550Z","shell.execute_reply":"2023-02-25T03:23:52.761145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Part 12: Submit","metadata":{}},{"cell_type":"code","source":"results = pd.DataFrame({'SegmentID' : test_data['SegmentID'].astype(str), 'SpeakerDialect' : prediction_list},\n                       columns = ['SegmentID', 'SpeakerDialect'])\nprint(results)\n\nresult_file = \"submission.csv\"\nresults.to_csv(result_file, sep= \",\", index = False)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-25T03:24:05.870486Z","iopub.execute_input":"2023-02-25T03:24:05.871447Z","iopub.status.idle":"2023-02-25T03:24:05.890132Z","shell.execute_reply.started":"2023-02-25T03:24:05.871393Z","shell.execute_reply":"2023-02-25T03:24:05.889014Z"},"trusted":true},"execution_count":null,"outputs":[]}]}