{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"pip install transformers plotly==5.8.0 pyyaml==5.4.1 datasets pytorch-lightning > /dev/null 2>&1","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:14:53.861356Z","iopub.execute_input":"2023-03-09T20:14:53.861952Z","iopub.status.idle":"2023-03-09T20:15:32.376592Z","shell.execute_reply.started":"2023-03-09T20:14:53.861907Z","shell.execute_reply":"2023-03-09T20:15:32.375197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\n\n# If there's a GPU available...\nif torch.cuda.is_available():    \n\n    # Tell PyTorch to use the GPU.    \n    device = torch.device(\"cuda\")\n\n    print('There are %d GPU(s) available.' % torch.cuda.device_count())\n\n    print('We will use the GPU:', torch.cuda.get_device_name(0))\n    !nvidia-smi\n\n# If not...\nelse:\n    print('No GPU available, using the CPU instead.')\n    device = torch.device(\"cpu\")","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:15:32.379378Z","iopub.execute_input":"2023-03-09T20:15:32.379784Z","iopub.status.idle":"2023-03-09T20:15:35.960249Z","shell.execute_reply.started":"2023-03-09T20:15:32.379739Z","shell.execute_reply":"2023-03-09T20:15:35.958987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!/usr/local/cuda/bin/nvcc --version\n\n!nvidia-smi","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:15:35.962487Z","iopub.execute_input":"2023-03-09T20:15:35.963780Z","iopub.status.idle":"2023-03-09T20:15:38.097424Z","shell.execute_reply.started":"2023-03-09T20:15:35.963731Z","shell.execute_reply":"2023-03-09T20:15:38.096156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport pyarabic.araby as ar\n\nimport re, functools, operator, string\nimport torch , optuna, gc, random, os\n\nfrom tqdm import tqdm_notebook as tqdm\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, accuracy_score, f1_score, confusion_matrix, precision_score , recall_score\nfrom transformers import AutoConfig, AutoModelForSequenceClassification, AutoTokenizer\nfrom transformers.data.processors import SingleSentenceClassificationProcessor\nfrom transformers import Trainer , TrainingArguments\nfrom transformers.trainer_utils import EvaluationStrategy\nfrom transformers.data.processors.utils import InputFeatures\nfrom torch.utils.data import Dataset\nfrom torch.utils.data import DataLoader\nfrom sklearn.utils import resample\n\nimport logging\n\nlogging.basicConfig(level=logging.WARNING)\nlogger = logging.getLogger(__name__)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:15:38.100806Z","iopub.execute_input":"2023-03-09T20:15:38.101465Z","iopub.status.idle":"2023-03-09T20:15:48.409484Z","shell.execute_reply.started":"2023-03-09T20:15:38.101422Z","shell.execute_reply":"2023-03-09T20:15:48.408411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"/kaggle/input/ml-olympiad-dialectrecognition/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/ml-olympiad-dialectrecognition/test.csv\")\nsub = pd.read_csv(\"/kaggle/input/ml-olympiad-dialectrecognition/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:15:48.411267Z","iopub.execute_input":"2023-03-09T20:15:48.412114Z","iopub.status.idle":"2023-03-09T20:15:50.046125Z","shell.execute_reply.started":"2023-03-09T20:15:48.412072Z","shell.execute_reply":"2023-03-09T20:15:50.045064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head(5)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:15:50.047778Z","iopub.execute_input":"2023-03-09T20:15:50.048139Z","iopub.status.idle":"2023-03-09T20:15:50.074914Z","shell.execute_reply.started":"2023-03-09T20:15:50.048095Z","shell.execute_reply":"2023-03-09T20:15:50.073827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:15:50.076350Z","iopub.execute_input":"2023-03-09T20:15:50.077343Z","iopub.status.idle":"2023-03-09T20:15:50.084799Z","shell.execute_reply.started":"2023-03-09T20:15:50.077300Z","shell.execute_reply":"2023-03-09T20:15:50.083384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Cleaning function \n\ndef data_cleaning (text):\n  try:\n    text = re.sub(r'^https?:\\/\\/.*[\\r\\n]*', '', text, flags=re.MULTILINE)\n    text = re.sub(r'^http?:\\/\\/.*[\\r\\n]*', '', text, flags=re.MULTILINE)\n    text = re.sub(r\"http\\S+\", \"\", text)\n    text = re.sub(r\"https\\S+\", \"\", text)\n    text = re.sub(r'\\s+', ' ', text)\n    text = re.sub(\"(\\s\\d+)\",\"\",text) \n    text = re.sub(r\"$\\d+\\W+|\\b\\d+\\b|\\W+\\d+$\", \"\", text)\n    text = re.sub(\"\\d+\", \" \", text)\n    text = ar.strip_tashkeel(text)\n    text = ar.strip_tatweel(text)\n    text = text.replace(\"#\", \" \");\n    text = text.replace(\"@\", \" \");\n    text = text.replace(\"_\", \" \");\n    translator = str.maketrans('', '', string.punctuation)\n    text = text.translate(translator)\n    text = text.replace(\"آ\", \"ا\")\n    text = text.replace(\"إ\", \"ا\")\n    text = text.replace(\"أ\", \"ا\")\n    text = text.replace(\"ؤ\", \"و\")\n    text = text.replace(\"ئ\", \"ي\")\n  except:\n    return text\n   \n  return text\n","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:15:50.086743Z","iopub.execute_input":"2023-03-09T20:15:50.087486Z","iopub.status.idle":"2023-03-09T20:15:50.098108Z","shell.execute_reply.started":"2023-03-09T20:15:50.087446Z","shell.execute_reply":"2023-03-09T20:15:50.096881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_new = train [['ProcessedText','SpeakerDialect']]\ntrain_data_new.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:15:50.099631Z","iopub.execute_input":"2023-03-09T20:15:50.099983Z","iopub.status.idle":"2023-03-09T20:15:50.128701Z","shell.execute_reply.started":"2023-03-09T20:15:50.099946Z","shell.execute_reply":"2023-03-09T20:15:50.127655Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_data_new['SpeakerDialect'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:15:50.133863Z","iopub.execute_input":"2023-03-09T20:15:50.134156Z","iopub.status.idle":"2023-03-09T20:15:50.151007Z","shell.execute_reply.started":"2023-03-09T20:15:50.134120Z","shell.execute_reply":"2023-03-09T20:15:50.149900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# cleaning data\n\n#train_data_new[\"ProcessedText\"] = train_data_new[\"ProcessedText\"].apply(lambda x:   data_cleaning(x))\ntrain_data_new.sample(3)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:15:50.152500Z","iopub.execute_input":"2023-03-09T20:15:50.153050Z","iopub.status.idle":"2023-03-09T20:15:50.168998Z","shell.execute_reply.started":"2023-03-09T20:15:50.153011Z","shell.execute_reply":"2023-03-09T20:15:50.168008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data_new.shape","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:15:50.170146Z","iopub.execute_input":"2023-03-09T20:15:50.170961Z","iopub.status.idle":"2023-03-09T20:15:50.178605Z","shell.execute_reply.started":"2023-03-09T20:15:50.170918Z","shell.execute_reply":"2023-03-09T20:15:50.177404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First setting the max_len , will be useful later for BERT Model\nExtra_Len = 7 # an extra padding in length , found to be useful for increasing F-score\nMax_Len = train_data_new[\"ProcessedText\"].str.split().str.len().max() + Extra_Len\n\nprint(Max_Len)\n\n#Spliting the Training data\nTest_Size = 0.2\nRand_Seed = 77 ","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:17:26.048360Z","iopub.execute_input":"2023-03-09T20:17:26.048837Z","iopub.status.idle":"2023-03-09T20:17:26.777848Z","shell.execute_reply.started":"2023-03-09T20:17:26.048793Z","shell.execute_reply":"2023-03-09T20:17:26.776544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split \n\ntrain_set, evaluation_set = train_test_split(train_data_new, test_size=Test_Size, stratify= train_data_new['SpeakerDialect'])","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:17:26.828470Z","iopub.execute_input":"2023-03-09T20:17:26.829066Z","iopub.status.idle":"2023-03-09T20:17:26.960294Z","shell.execute_reply.started":"2023-03-09T20:17:26.829032Z","shell.execute_reply":"2023-03-09T20:17:26.959289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Train set: \")\nprint(train_set[\"SpeakerDialect\"].value_counts())\nprint(\"---------------------------\")\nprint (\"Evaluation set: \")\nprint (evaluation_set[\"SpeakerDialect\"].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:17:26.961930Z","iopub.execute_input":"2023-03-09T20:17:26.962504Z","iopub.status.idle":"2023-03-09T20:17:26.979839Z","shell.execute_reply.started":"2023-03-09T20:17:26.962464Z","shell.execute_reply":"2023-03-09T20:17:26.978658Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Model_Used= \"aubmindlab/bert-base-arabertv02\"\nTask_Name = \"classification\"\n\nclass Dataset:\n    def __init__(\n        self,\n        name,\n        train,\n        test,\n        label_list,\n    ):\n        self.name = name\n        self.train = train\n        self.test = test\n        self.label_list = label_list\n        \nclass BERTModelDataset(Dataset):\n    def __init__(self, text, target, model_name, max_len, label_map):\n      super(BERTModelDataset).__init__()\n      self.text = text\n      self.target = target\n      self.tokenizer_name = model_name\n      self.tokenizer = AutoTokenizer.from_pretrained(model_name)\n      self.max_len = max_len\n      self.label_map = label_map\n  \n    def __len__(self):\n      return len(self.text)\n\n    def __getitem__(self,item):\n      text = str(self.text[item])\n      text = \" \".join(text.split())\n    \n      encoded_review = self.tokenizer.encode_plus(\n      text,\n      max_length= self.max_len,\n      add_special_tokens= True,\n      return_token_type_ids=False,\n      pad_to_max_length=True,\n      truncation='longest_first',\n      return_attention_mask=True,\n      return_tensors='pt'\n    )\n      input_ids = encoded_review['input_ids'].to(device)\n      attention_mask = encoded_review['attention_mask'].to(device)\n\n      return InputFeatures(input_ids=input_ids.flatten(), attention_mask=attention_mask.flatten(), label=self.label_map[self.target[item]])","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:17:26.983072Z","iopub.execute_input":"2023-03-09T20:17:26.985751Z","iopub.status.idle":"2023-03-09T20:17:26.996114Z","shell.execute_reply.started":"2023-03-09T20:17:26.985722Z","shell.execute_reply":"2023-03-09T20:17:26.995058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def model_init():\n  return AutoModelForSequenceClassification.from_pretrained(Model_Used, return_dict=True, num_labels=len(label_map))\n\ndef compute_metrics(p): #p should be of type EvalPrediction\n  preds = np.argmax(p.predictions, axis=1)\n  assert len(preds) == len(p.label_ids)\n  print(classification_report(p.label_ids,preds))\n  #print(confusion_matrix(p.label_ids,preds))\n  macro_f1 = f1_score(p.label_ids,preds,average='macro')\n  macro_precision = precision_score(p.label_ids,preds,average='macro')\n  macro_recall = recall_score(p.label_ids,preds,average='macro')\n  acc = accuracy_score(p.label_ids,preds)\n  return {\n      'macro_f1' : macro_f1, \n      'macro_precision': macro_precision,\n      'macro_recall': macro_recall,\n      'accuracy': acc\n  }\n\ndef set_seed(seed):\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed_all(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = False\n    np.random.seed(seed)\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:17:26.997795Z","iopub.execute_input":"2023-03-09T20:17:26.998153Z","iopub.status.idle":"2023-03-09T20:17:27.012397Z","shell.execute_reply.started":"2023-03-09T20:17:26.998112Z","shell.execute_reply":"2023-03-09T20:17:27.011105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_list = list(train_set[\"SpeakerDialect\"].unique())\n\nprint(label_list)\nprint(train_set[\"SpeakerDialect\"].value_counts())\n\ndata_set = Dataset( \"OLY\", train_set, evaluation_set, label_list )\n\nlabel_map = { v:index for index, v in enumerate(label_list) }\nprint(label_map)\n\ntrain_dataset = BERTModelDataset(train_set[\"ProcessedText\"].to_list(),\n                                 train_set[\"SpeakerDialect\"].to_list(),Model_Used,int(Max_Len),label_map)\n\nevaluation_dataset = BERTModelDataset(evaluation_set[\"ProcessedText\"].to_list(),\n                                      evaluation_set[\"SpeakerDialect\"].to_list(),Model_Used,int(Max_Len),label_map)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:17:27.014259Z","iopub.execute_input":"2023-03-09T20:17:27.014640Z","iopub.status.idle":"2023-03-09T20:17:27.506027Z","shell.execute_reply.started":"2023-03-09T20:17:27.014579Z","shell.execute_reply":"2023-03-09T20:17:27.504990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#define training arguments\ntraining_args = TrainingArguments(\"./train\")\ntraining_args.lr_scheduler_type = 'cosine'\ntraining_args.evaluate_during_training = True\ntraining_args.adam_epsilon =1e-8 \ntraining_args.learning_rate = 2e-05\ntraining_args.fp16 = True\ntraining_args.per_device_train_batch_size = 64\ntraining_args.per_device_eval_batch_size = 32\ntraining_args.gradient_accumulation_steps = 2\ntraining_args.num_train_epochs= 2\ntraining_args.warmup_steps = 0 \ntraining_args.evaluation_strategy = EvaluationStrategy.EPOCH\ntraining_args.seed = 77 \ntraining_args.disable_tqdm = False","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:17:27.507588Z","iopub.execute_input":"2023-03-09T20:17:27.508065Z","iopub.status.idle":"2023-03-09T20:17:27.518760Z","shell.execute_reply.started":"2023-03-09T20:17:27.508022Z","shell.execute_reply":"2023-03-09T20:17:27.517668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_args.dataloader_pin_memory = False\ngc.collect()\ntorch.cuda.empty_cache()\nset_seed(Rand_Seed) \n\ntrainer = Trainer(\n    model = model_init(),\n    args = training_args,\n    train_dataset = train_dataset,\n    eval_dataset= evaluation_dataset,\n    compute_metrics=compute_metrics\n)\n\nprint(training_args.seed)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:17:27.520346Z","iopub.execute_input":"2023-03-09T20:17:27.520832Z","iopub.status.idle":"2023-03-09T20:17:39.931973Z","shell.execute_reply.started":"2023-03-09T20:17:27.520794Z","shell.execute_reply":"2023-03-09T20:17:39.930839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nos.environ[\"WANDB_DISABLED\"] = \"true\"\ntrainer.train()","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:17:39.934580Z","iopub.execute_input":"2023-03-09T20:17:39.935916Z","iopub.status.idle":"2023-03-09T20:36:46.411395Z","shell.execute_reply.started":"2023-03-09T20:17:39.935872Z","shell.execute_reply":"2023-03-09T20:36:46.410207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:36:46.419340Z","iopub.execute_input":"2023-03-09T20:36:46.422113Z","iopub.status.idle":"2023-03-09T20:36:46.454601Z","shell.execute_reply.started":"2023-03-09T20:36:46.422065Z","shell.execute_reply":"2023-03-09T20:36:46.453162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data_new = test[['SegmentID','ProcessedText']]\ntest_data_new.head(3)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:36:46.456402Z","iopub.execute_input":"2023-03-09T20:36:46.457742Z","iopub.status.idle":"2023-03-09T20:36:46.479892Z","shell.execute_reply.started":"2023-03-09T20:36:46.457693Z","shell.execute_reply":"2023-03-09T20:36:46.478225Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_data_new[\"ProcessedText\"] = test_data_new[\"ProcessedText\"].apply(lambda x:   data_cleaning(x))\ntest_data_new","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:36:46.481698Z","iopub.execute_input":"2023-03-09T20:36:46.482498Z","iopub.status.idle":"2023-03-09T20:36:46.503037Z","shell.execute_reply.started":"2023-03-09T20:36:46.482454Z","shell.execute_reply":"2023-03-09T20:36:46.501810Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict(text, tokenizer):\n  try:\n  #print(text)\n    encoded_review = tokenizer.encode_plus(\n      text,\n      max_length=int(Max_Len),\n      add_special_tokens=True,\n      return_token_type_ids=False,\n      pad_to_max_length=True, #True,\n      truncation='longest_first',\n      return_attention_mask=True,\n      return_tensors='pt'\n    )\n\n    input_ids = encoded_review['input_ids'].to(device) #(input_ids + ([tokenizer.pad_token_id] * padding_length)).to(device)  \n    attention_mask = encoded_review['attention_mask'].to(device)\n    \n\n    output = trainer.model(input_ids, attention_mask)\n\n    _, prediction = torch.max(output[0], dim=1)\n    return prediction[0]\n  except:\n    return 0\n\ntokenizer = AutoTokenizer.from_pretrained(Model_Used)\n\nprediction_list = []\ni = 0\nfor element in test_data_new[\"ProcessedText\"]:\n    id = test[\"SegmentID\"][i]\n  \n    pre = predict(element,tokenizer)\n    pre_txt = label_list[pre]\n   \n    if pre_txt == 'Najdi': pre_txt = 1\n    if pre_txt == 'Hijazi': pre_txt = 2\n    if pre_txt == 'Khaliji': pre_txt = 3\n    if pre_txt == 'ModernStandardArabic': pre_txt = 4\n\n    \n    prediction_list.append(pre_txt)\n    \n    i = i + 1","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:36:46.504787Z","iopub.execute_input":"2023-03-09T20:36:46.505189Z","iopub.status.idle":"2023-03-09T20:37:17.695527Z","shell.execute_reply.started":"2023-03-09T20:36:46.505147Z","shell.execute_reply":"2023-03-09T20:37:17.694242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"results = pd.DataFrame({'SegmentID' : test['SegmentID'].astype(str), 'SpeakerDialect' : prediction_list},\n                       columns = ['SegmentID', 'SpeakerDialect'])\nprint(results)\n\nresult_file = \"submission.csv\"\nresults.to_csv(result_file, sep= \",\", index = False)","metadata":{"execution":{"iopub.status.busy":"2023-03-09T20:37:17.700012Z","iopub.execute_input":"2023-03-09T20:37:17.703935Z","iopub.status.idle":"2023-03-09T20:37:17.740115Z","shell.execute_reply.started":"2023-03-09T20:37:17.703876Z","shell.execute_reply":"2023-03-09T20:37:17.738948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}