{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11650,"sourceType":"datasetVersion","datasetId":8327},{"sourceId":7478837,"sourceType":"datasetVersion","datasetId":4353276},{"sourceId":7478962,"sourceType":"datasetVersion","datasetId":4353361}],"dockerImageVersionId":30299,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom keras.models import Sequential\nfrom keras.layers.recurrent import LSTM, GRU,SimpleRNN\nfrom keras.layers.core import Dense, Activation, Dropout\nfrom keras.layers.embeddings import Embedding\nfrom keras.utils import np_utils\nfrom sklearn import preprocessing, decomposition, model_selection, metrics, pipeline\nfrom keras.layers import BatchNormalization,GlobalMaxPooling1D, Conv1D, MaxPooling1D, Flatten, Bidirectional, SpatialDropout1D\nfrom keras.preprocessing import sequence, text\nfrom keras.callbacks import EarlyStopping\n\n\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n%matplotlib inline\nfrom plotly import graph_objs as go\nimport plotly.express as px\nimport plotly.figure_factory as ff\nimport matplotlib.pyplot as plt\nimport re\nimport pandas as pd\nimport string\nimport numpy as np \nimport random\nimport spacy\nimport transformers\nfrom tokenizers import BertWordPieceTokenizer\nimport tensorflow as tf\nimport nltk\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\nfrom nltk.tokenize import word_tokenize\nfrom nltk.corpus import stopwords\nfrom nltk.corpus import wordnet\nfrom nltk.stem import WordNetLemmatizer\n\nfrom keras.models import Sequential,Model, load_model\nfrom keras.layers import (LSTM, \n                          Embedding, \n                          BatchNormalization,\n                          Dense, \n                          Dropout, \n                          Bidirectional,\n                          GlobalAveragePooling1D,\n                          Concatenate,\n                          Attention,\n                          Input)\nfrom keras.losses import SparseCategoricalCrossentropy\nfrom keras.regularizers import l2\n#from keras.optimizers import Adam\nfrom keras.callbacks import ModelCheckpoint\n\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, accuracy_score, precision_score, recall_score, f1_score, roc_auc_score\nfrom sklearn.model_selection import StratifiedKFold\n#from random import shuffle\nfrom datetime import datetime\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-18T07:50:04.313854Z","iopub.execute_input":"2024-06-18T07:50:04.314529Z","iopub.status.idle":"2024-06-18T07:50:18.423106Z","shell.execute_reply.started":"2024-06-18T07:50:04.314438Z","shell.execute_reply":"2024-06-18T07:50:18.422218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.version.VERSION","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:50:18.424901Z","iopub.execute_input":"2024-06-18T07:50:18.425212Z","iopub.status.idle":"2024-06-18T07:50:18.433651Z","shell.execute_reply.started":"2024-06-18T07:50:18.425184Z","shell.execute_reply":"2024-06-18T07:50:18.432521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(transformers.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:50:18.435508Z","iopub.execute_input":"2024-06-18T07:50:18.435905Z","iopub.status.idle":"2024-06-18T07:50:18.449914Z","shell.execute_reply.started":"2024-06-18T07:50:18.435866Z","shell.execute_reply":"2024-06-18T07:50:18.448964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data1 = pd.read_csv('/kaggle/input/stackoverflowori/StackOverflow_Original.csv')\ndata2 = pd.read_csv('/kaggle/input/stackoverflownew/NewData.csv')\ndata1.head()","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2024-06-18T07:50:18.452086Z","iopub.execute_input":"2024-06-18T07:50:18.452425Z","iopub.status.idle":"2024-06-18T07:50:18.520685Z","shell.execute_reply.started":"2024-06-18T07:50:18.452395Z","shell.execute_reply":"2024-06-18T07:50:18.519880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"replacements = {-1: \"Negative\", 0: \"Neutral\", 1: \"Positive\"}\ndata1['oracle'].replace(replacements, inplace=True)\ndata1.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:50:18.521930Z","iopub.execute_input":"2024-06-18T07:50:18.522229Z","iopub.status.idle":"2024-06-18T07:50:18.536429Z","shell.execute_reply.started":"2024-06-18T07:50:18.522202Z","shell.execute_reply":"2024-06-18T07:50:18.535344Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data1.columns = ['id', 'text', 'target']\ndata2.columns = ['id', 'text', 'target']\ndata1 = data1.drop(['id'], axis=1)\ndata2 = data2.drop(['id'], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:50:18.537772Z","iopub.execute_input":"2024-06-18T07:50:18.538372Z","iopub.status.idle":"2024-06-18T07:50:18.549009Z","shell.execute_reply.started":"2024-06-18T07:50:18.538342Z","shell.execute_reply":"2024-06-18T07:50:18.548047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df =  pd.concat([data1, data2]).reset_index(drop=True)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:50:18.550179Z","iopub.execute_input":"2024-06-18T07:50:18.550487Z","iopub.status.idle":"2024-06-18T07:50:18.563906Z","shell.execute_reply.started":"2024-06-18T07:50:18.550460Z","shell.execute_reply":"2024-06-18T07:50:18.562856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_count = df.target.value_counts()\nval_count","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:50:18.565004Z","iopub.execute_input":"2024-06-18T07:50:18.565283Z","iopub.status.idle":"2024-06-18T07:50:18.574785Z","shell.execute_reply.started":"2024-06-18T07:50:18.565257Z","shell.execute_reply":"2024-06-18T07:50:18.573877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8,4))\nplt.bar(val_count.index, val_count.values)\nplt.title(\"Sentiment Data Distribution\")","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:50:18.576158Z","iopub.execute_input":"2024-06-18T07:50:18.576757Z","iopub.status.idle":"2024-06-18T07:50:18.811367Z","shell.execute_reply.started":"2024-06-18T07:50:18.576710Z","shell.execute_reply":"2024-06-18T07:50:18.810483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nimport string\nfrom sklearn.preprocessing import LabelEncoder","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:50:18.814040Z","iopub.execute_input":"2024-06-18T07:50:18.814315Z","iopub.status.idle":"2024-06-18T07:50:18.821188Z","shell.execute_reply.started":"2024-06-18T07:50:18.814289Z","shell.execute_reply":"2024-06-18T07:50:18.820304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_text(text):\n    '''Make text lowercase, remove text in square brackets,remove links,remove punctuation\n    and remove words containing numbers.'''\n    text = str(text).lower()\n    text = re.sub(r\"[\\[\\]]\", \"\", text)\n    text = re.sub('https?://\\S+|www\\.\\S+', '',  text)\n    text = re.sub('<.*?>+', '', text)\n    text = re.sub('[%s]' % re.escape(string.punctuation), '',  text)\n    text = re.sub('\\n', '', text)\n    text = re.sub('\\w*\\d\\w*', '',  text)\n    return text\ndf['text_clean'] = df['text'].apply(clean_text)\ndf","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:50:18.822226Z","iopub.execute_input":"2024-06-18T07:50:18.823279Z","iopub.status.idle":"2024-06-18T07:50:19.022423Z","shell.execute_reply.started":"2024-06-18T07:50:18.823237Z","shell.execute_reply":"2024-06-18T07:50:19.021443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"le = LabelEncoder()\nle.fit(df['target'])\n\ndf['target_encoded'] = le.transform(df['target'])\ndf","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:50:30.459071Z","iopub.execute_input":"2024-06-18T07:50:30.459483Z","iopub.status.idle":"2024-06-18T07:50:30.477297Z","shell.execute_reply.started":"2024-06-18T07:50:30.459449Z","shell.execute_reply":"2024-06-18T07:50:30.476275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['text_clean'].apply(lambda x:len(str(x).split())).max()","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:50:30.976231Z","iopub.execute_input":"2024-06-18T07:50:30.976967Z","iopub.status.idle":"2024-06-18T07:50:30.994926Z","shell.execute_reply.started":"2024-06-18T07:50:30.976925Z","shell.execute_reply":"2024-06-18T07:50:30.994031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Preparation","metadata":{}},{"cell_type":"code","source":"X_train, X_test, y_train, y_test = train_test_split(df['text_clean'],df['target_encoded'], test_size = 0.1, stratify=df['target_encoded'], random_state = 56)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:51:30.547725Z","iopub.execute_input":"2024-06-18T07:51:30.548146Z","iopub.status.idle":"2024-06-18T07:51:30.559095Z","shell.execute_reply.started":"2024-06-18T07:51:30.548110Z","shell.execute_reply":"2024-06-18T07:51:30.558171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nfiltered_data = pd.concat([X_test, y_test], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:51:31.362429Z","iopub.execute_input":"2024-06-18T07:51:31.363426Z","iopub.status.idle":"2024-06-18T07:51:31.369533Z","shell.execute_reply.started":"2024-06-18T07:51:31.363386Z","shell.execute_reply":"2024-06-18T07:51:31.368388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filtered_data_0 = filtered_data[filtered_data['text_clean'].str.contains('error', case=False, na=False) & filtered_data['target_encoded'].isin([0])]\nfilter_0 = filtered_data_0['text_clean']\ntarget_filter_0 = filtered_data_0['target_encoded']","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:53:02.047463Z","iopub.execute_input":"2024-06-18T07:53:02.048327Z","iopub.status.idle":"2024-06-18T07:53:02.058877Z","shell.execute_reply.started":"2024-06-18T07:53:02.048289Z","shell.execute_reply":"2024-06-18T07:53:02.057863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filtered_data_12 = filtered_data[filtered_data['text_clean'].str.contains('error', case=False, na=False) & filtered_data['target_encoded'].isin([1,2])]\nfilter_12 = filtered_data_12['text_clean']\ntarget_filter_12 = filtered_data_12['target_encoded']","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:53:02.723583Z","iopub.execute_input":"2024-06-18T07:53:02.724362Z","iopub.status.idle":"2024-06-18T07:53:02.732406Z","shell.execute_reply.started":"2024-06-18T07:53:02.724323Z","shell.execute_reply":"2024-06-18T07:53:02.731469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filtered_data_all = filtered_data[filtered_data['text_clean'].str.contains('error', case=False, na=False)]\nfilter_all = filtered_data_all['text_clean']\ntarget_filter_all = filtered_data_all['target_encoded']","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:53:03.450372Z","iopub.execute_input":"2024-06-18T07:53:03.450919Z","iopub.status.idle":"2024-06-18T07:53:03.458889Z","shell.execute_reply.started":"2024-06-18T07:53:03.450871Z","shell.execute_reply":"2024-06-18T07:53:03.457752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.concat([X_train, y_train], axis=1)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:53:07.237882Z","iopub.execute_input":"2024-06-18T07:53:07.238240Z","iopub.status.idle":"2024-06-18T07:53:07.244676Z","shell.execute_reply.started":"2024-06-18T07:53:07.238210Z","shell.execute_reply":"2024-06-18T07:53:07.243647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = train_data.reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:53:08.337939Z","iopub.execute_input":"2024-06-18T07:53:08.338901Z","iopub.status.idle":"2024-06-18T07:53:08.343823Z","shell.execute_reply.started":"2024-06-18T07:53:08.338848Z","shell.execute_reply":"2024-06-18T07:53:08.342736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_n, X_valid_n, y_train_n, y_valid_n = train_test_split(train_data['text_clean'], train_data['target_encoded'], \n                                                  stratify=train_data['target_encoded'], \n                                                  random_state=42, \n                                                  test_size=0.1)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:53:50.200896Z","iopub.execute_input":"2024-06-18T07:53:50.201303Z","iopub.status.idle":"2024-06-18T07:53:50.212542Z","shell.execute_reply.started":"2024-06-18T07:53:50.201266Z","shell.execute_reply":"2024-06-18T07:53:50.211366Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=512, maxlen=512):\n    \"\"\"\n    Encoder for encoding the text into sequence of integers for BERT Input\n    \"\"\"\n    tokenizer.enable_truncation(max_length=maxlen)\n    tokenizer.enable_padding(length=maxlen)\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    \n    return np.array(all_ids)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:53:55.016916Z","iopub.execute_input":"2024-06-18T07:53:55.017671Z","iopub.status.idle":"2024-06-18T07:53:55.025359Z","shell.execute_reply.started":"2024-06-18T07:53:55.017624Z","shell.execute_reply":"2024-06-18T07:53:55.024227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nBATCH_SIZE = 16\nMAX_LEN = 56","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:53:56.886959Z","iopub.execute_input":"2024-06-18T07:53:56.887873Z","iopub.status.idle":"2024-06-18T07:53:56.891900Z","shell.execute_reply.started":"2024-06-18T07:53:56.887835Z","shell.execute_reply":"2024-06-18T07:53:56.890921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#AUTO","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:53:58.584104Z","iopub.execute_input":"2024-06-18T07:53:58.584484Z","iopub.status.idle":"2024-06-18T07:53:58.589524Z","shell.execute_reply.started":"2024-06-18T07:53:58.584449Z","shell.execute_reply":"2024-06-18T07:53:58.588328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tokenization\n\nFor understanding please refer to hugging face documentation again","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# First load the real tokenizer\n\ntokenizer = transformers.BertTokenizer.from_pretrained('bert-base-uncased')\n# Save the loaded tokenizer locally\ntokenizer.save_pretrained('.')\n# Reload it with the huggingface tokenizers library\nfrom tokenizers import BertWordPieceTokenizer\nfast_tokenizer = BertWordPieceTokenizer('vocab.txt', lowercase=False)\nfast_tokenizer","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:53:59.357967Z","iopub.execute_input":"2024-06-18T07:53:59.358366Z","iopub.status.idle":"2024-06-18T07:54:00.848195Z","shell.execute_reply.started":"2024-06-18T07:53:59.358331Z","shell.execute_reply":"2024-06-18T07:54:00.847153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train_n = fast_encode(X_train_n.astype(str), fast_tokenizer, maxlen=MAX_LEN)\nX_valid_n = fast_encode(X_valid_n.astype(str), fast_tokenizer, maxlen=MAX_LEN)\nX_test = fast_encode(X_test.astype(str), fast_tokenizer, maxlen=MAX_LEN)\nfilter_0 = fast_encode(filter_0.astype(str), fast_tokenizer, maxlen=MAX_LEN)\nfilter_12 = fast_encode(filter_12.astype(str), fast_tokenizer, maxlen=MAX_LEN)\nfilter_all = fast_encode(filter_all.astype(str), fast_tokenizer, maxlen=MAX_LEN)\n","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:54:14.350488Z","iopub.execute_input":"2024-06-18T07:54:14.351377Z","iopub.status.idle":"2024-06-18T07:54:14.610715Z","shell.execute_reply.started":"2024-06-18T07:54:14.351338Z","shell.execute_reply":"2024-06-18T07:54:14.609619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_valid_n.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:54:36.981569Z","iopub.execute_input":"2024-06-18T07:54:36.982087Z","iopub.status.idle":"2024-06-18T07:54:36.988373Z","shell.execute_reply.started":"2024-06-18T07:54:36.982047Z","shell.execute_reply":"2024-06-18T07:54:36.987448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train.shape","metadata":{"execution":{"iopub.status.busy":"2024-06-05T05:40:06.840568Z","iopub.execute_input":"2024-06-05T05:40:06.840975Z","iopub.status.idle":"2024-06-05T05:40:06.849395Z","shell.execute_reply.started":"2024-06-05T05:40:06.840943Z","shell.execute_reply":"2024-06-05T05:40:06.848429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((X_train_n, y_train_n))\n    #.repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    #.prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((X_valid_n, y_valid_n))\n    .batch(BATCH_SIZE)\n    .cache()\n    #.prefetch(AUTO)\n)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:54:40.230025Z","iopub.execute_input":"2024-06-18T07:54:40.230396Z","iopub.status.idle":"2024-06-18T07:54:48.007831Z","shell.execute_reply.started":"2024-06-18T07:54:40.230363Z","shell.execute_reply":"2024-06-18T07:54:48.006976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_combined = np.concatenate([X_train_n, X_valid_n])\ny_combined = np.concatenate([y_train_n, y_valid_n])","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:54:53.389777Z","iopub.execute_input":"2024-06-18T07:54:53.390194Z","iopub.status.idle":"2024-06-18T07:54:53.397428Z","shell.execute_reply.started":"2024-06-18T07:54:53.390159Z","shell.execute_reply":"2024-06-18T07:54:53.396120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(transformer, max_len=512):\n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    out = Dense(3, activation='softmax')(cls_token)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='sparse_categorical_crossentropy', metrics=['accuracy'])\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:54:56.224199Z","iopub.execute_input":"2024-06-18T07:54:56.224944Z","iopub.status.idle":"2024-06-18T07:54:56.231364Z","shell.execute_reply.started":"2024-06-18T07:54:56.224905Z","shell.execute_reply":"2024-06-18T07:54:56.230434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:54:56.903208Z","iopub.execute_input":"2024-06-18T07:54:56.903978Z","iopub.status.idle":"2024-06-18T07:54:56.908855Z","shell.execute_reply.started":"2024-06-18T07:54:56.903938Z","shell.execute_reply":"2024-06-18T07:54:56.907859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strategy = tf.distribute.MirroredStrategy()\nwith strategy.scope():\n    transformer_layer = (\n        transformers.TFBertModel.from_pretrained('bert-base-uncased')\n    )\n    model = build_model(transformer_layer, max_len=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:54:57.599259Z","iopub.execute_input":"2024-06-18T07:54:57.600308Z","iopub.status.idle":"2024-06-18T07:55:21.478577Z","shell.execute_reply.started":"2024-06-18T07:54:57.600265Z","shell.execute_reply":"2024-06-18T07:55:21.477629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint = ModelCheckpoint(\n    'model_1.h5', \n    monitor = 'val_accuracy', \n    verbose = 1, \n    save_best_only = True\n)\nearly_stopping = tf.keras.callbacks.EarlyStopping(patience=5)\nstart = datetime.now()\nmodel_history = model.fit(\n    train_dataset, \n    epochs = 100,\n    batch_size = 16,\n    validation_data = (valid_dataset),\n    verbose = 1,\n    #class_weight=class_weights_dict,\n    callbacks = [checkpoint, early_stopping]\n)\nduration = datetime.now() - start\nprint(\"Training completed in time: \", duration)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:55:35.885563Z","iopub.execute_input":"2024-06-18T07:55:35.886404Z","iopub.status.idle":"2024-06-18T07:57:22.093429Z","shell.execute_reply.started":"2024-06-18T07:55:35.886362Z","shell.execute_reply":"2024-06-18T07:57:22.091541Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import TFBertModel","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:57:25.177004Z","iopub.execute_input":"2024-06-18T07:57:25.177442Z","iopub.status.idle":"2024-06-18T07:57:25.182517Z","shell.execute_reply.started":"2024-06-18T07:57:25.177406Z","shell.execute_reply":"2024-06-18T07:57:25.181508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_load = load_model('model_1.h5', custom_objects={'TFBertModel': TFBertModel})\ny_pred = model_load.predict(X_valid_n)\ny_pred_labels = np.argmax(y_pred, axis=1)\nprint(classification_report(y_valid_n, y_pred_labels))","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:57:26.573623Z","iopub.execute_input":"2024-06-18T07:57:26.574016Z","iopub.status.idle":"2024-06-18T07:57:33.826362Z","shell.execute_reply.started":"2024-06-18T07:57:26.573979Z","shell.execute_reply":"2024-06-18T07:57:33.825126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(\"Train Accuracy  : {:.2f} %\".format(model_history.history['accuracy'][-1]*100))\nprint(\"Test Accuracy   : {:.2f} %\".format(accuracy_score(y_valid_n, y_pred_labels) * 100))\nprint(\"Precision Score : {:.2f} %\".format(precision_score(y_valid_n, y_pred_labels, average='weighted') * 100))\nprint(\"Recall Score    : {:.2f} %\".format(recall_score(y_valid_n, y_pred_labels, average='weighted') * 100))\nprint(\"F1 Weighted Score : {:.2f} %\".format(f1_score(y_valid_n, y_pred_labels, average='weighted') * 100))\nprint(\"F1 Micro Score    : {:.2f} %\".format(f1_score(y_valid_n, y_pred_labels, average='micro') * 100))\nprint(\"F1 Macro Score    : {:.2f} %\".format(f1_score(y_valid_n, y_pred_labels, average='macro') * 100))","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:57:44.774488Z","iopub.execute_input":"2024-06-18T07:57:44.774897Z","iopub.status.idle":"2024-06-18T07:57:44.790720Z","shell.execute_reply.started":"2024-06-18T07:57:44.774860Z","shell.execute_reply":"2024-06-18T07:57:44.789751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_load = load_model('model_1.h5', custom_objects={'TFBertModel': TFBertModel})\ny_pred = model_load.predict(X_test)\ny_pred_labels = np.argmax(y_pred, axis=1)\nprint(classification_report(y_test, y_pred_labels))","metadata":{"execution":{"iopub.status.busy":"2024-06-18T07:57:48.554509Z","iopub.execute_input":"2024-06-18T07:57:48.554916Z","iopub.status.idle":"2024-06-18T07:57:55.730858Z","shell.execute_reply.started":"2024-06-18T07:57:48.554883Z","shell.execute_reply":"2024-06-18T07:57:55.729283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_sequences(encoded_sequences, tokenizer):\n    \"\"\"\n    Decodes sequences of integers back to text using the tokenizer\n    \"\"\"\n    texts = []\n    for seq in encoded_sequences:\n        text = tokenizer.decode(seq, skip_special_tokens=True)\n        texts.append(text)\n    return texts","metadata":{"execution":{"iopub.status.busy":"2024-06-18T08:00:34.358175Z","iopub.execute_input":"2024-06-18T08:00:34.358929Z","iopub.status.idle":"2024-06-18T08:00:34.364450Z","shell.execute_reply.started":"2024-06-18T08:00:34.358890Z","shell.execute_reply":"2024-06-18T08:00:34.363283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"original_texts = decode_sequences(X_test, fast_tokenizer)\ndata = pd.DataFrame(original_texts, columns=[\"Original_Text\"])\ntarget = pd.DataFrame(y_pred_labels, columns=[\"Prediction\"])\npredicted_data = pd.concat([data, target], axis=1)\npredicted_data = predicted_data.to_csv('predictedbert.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T08:00:56.298276Z","iopub.execute_input":"2024-06-18T08:00:56.298988Z","iopub.status.idle":"2024-06-18T08:00:56.330640Z","shell.execute_reply.started":"2024-06-18T08:00:56.298950Z","shell.execute_reply":"2024-06-18T08:00:56.329599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(\"Train Accuracy  : {:.2f} %\".format(model_history.history['accuracy'][-1]*100))\nprint(\"Test Accuracy   : {:.2f} %\".format(accuracy_score(y_test, y_pred_labels) * 100))\nprint(\"Precision Score : {:.2f} %\".format(precision_score(y_test, y_pred_labels, average='weighted') * 100))\nprint(\"Recall Score    : {:.2f} %\".format(recall_score(y_test, y_pred_labels, average='weighted') * 100))\nprint(\"F1 Weighted Score : {:.2f} %\".format(f1_score(y_test, y_pred_labels, average='weighted') * 100))\nprint(\"F1 Micro Score    : {:.2f} %\".format(f1_score(y_test, y_pred_labels, average='micro') * 100))\nprint(\"F1 Macro Score    : {:.2f} %\".format(f1_score(y_test, y_pred_labels, average='macro') * 100))","metadata":{"execution":{"iopub.status.busy":"2024-06-18T08:01:24.353657Z","iopub.execute_input":"2024-06-18T08:01:24.354600Z","iopub.status.idle":"2024-06-18T08:01:24.370388Z","shell.execute_reply.started":"2024-06-18T08:01:24.354557Z","shell.execute_reply":"2024-06-18T08:01:24.369187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_attention_scores(model, input_data):\n    pred = model.predict(input_data)\n    predictions = np.argmax(pred, axis=1)\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2024-06-18T08:01:25.859641Z","iopub.execute_input":"2024-06-18T08:01:25.860022Z","iopub.status.idle":"2024-06-18T08:01:25.865836Z","shell.execute_reply.started":"2024-06-18T08:01:25.859990Z","shell.execute_reply":"2024-06-18T08:01:25.864844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions_all = get_attention_scores(model_load, filter_all)\npredictions_0 = get_attention_scores(model_load, filter_0)\npredictions_12 = get_attention_scores(model_load, filter_12)\nprint(\"All Class\")\nprint(\"Precision Score : {:.2f} %\".format(precision_score(target_filter_all, predictions_all, average='weighted') * 100))\nprint(\"Recall Score    : {:.2f} %\".format(recall_score(target_filter_all, predictions_all, average='weighted') * 100))\nprint(\"F1 Weighted Score : {:.2f} %\".format(f1_score(target_filter_all, predictions_all, average='weighted') * 100))\nprint(\"F1 Micro Score    : {:.2f} %\".format(f1_score(target_filter_all, predictions_all, average='micro') * 100))\nprint(\"F1 Macro Score    : {:.2f} %\".format(f1_score(target_filter_all, predictions_all, average='macro') * 100))\nprint(\"0\")\nprint(\"Precision Score : {:.2f} %\".format(precision_score(target_filter_0, predictions_0, average='weighted') * 100))\nprint(\"Recall Score    : {:.2f} %\".format(recall_score(target_filter_0, predictions_0, average='weighted') * 100))\nprint(\"F1 Weighted Score : {:.2f} %\".format(f1_score(target_filter_0, predictions_0, average='weighted') * 100))\nprint(\"F1 Micro Score    : {:.2f} %\".format(f1_score(target_filter_0, predictions_0, average='micro') * 100))\nprint(\"F1 Macro Score    : {:.2f} %\".format(f1_score(target_filter_0, predictions_0, average='macro') * 100))\nprint(\"1-2\")\nprint(\"Precision Score : {:.2f} %\".format(precision_score(target_filter_12, predictions_12, average='weighted') * 100))\nprint(\"Recall Score    : {:.2f} %\".format(recall_score(target_filter_12, predictions_12, average='weighted') * 100))\nprint(\"F1 Weighted Score : {:.2f} %\".format(f1_score(target_filter_12, predictions_12, average='weighted') * 100))\nprint(\"F1 Micro Score    : {:.2f} %\".format(f1_score(target_filter_12, predictions_12, average='micro') * 100))\nprint(\"F1 Macro Score    : {:.2f} %\".format(f1_score(target_filter_12, predictions_12, average='macro') * 100))","metadata":{"execution":{"iopub.status.busy":"2024-06-18T08:02:18.841856Z","iopub.execute_input":"2024-06-18T08:02:18.842268Z","iopub.status.idle":"2024-06-18T08:02:19.224910Z","shell.execute_reply.started":"2024-06-18T08:02:18.842228Z","shell.execute_reply":"2024-06-18T08:02:19.223605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(model_history.history['accuracy'])\nplt.plot(model_history.history['val_accuracy'])\nplt.title('BERT Accuracy')\nplt.ylabel('Accuracy')\nplt.xlabel('Epoch')\nplt.legend(['Training', 'Validation'], loc='upper left')\nplt.show()\n\n# summarize history for loss\nplt.plot(model_history.history['loss'])\nplt.plot(model_history.history['val_loss'])\nplt.title('BERT loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(['Training', 'Validation'], loc='upper left')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-18T08:02:24.282979Z","iopub.execute_input":"2024-06-18T08:02:24.283413Z","iopub.status.idle":"2024-06-18T08:02:24.319888Z","shell.execute_reply.started":"2024-06-18T08:02:24.283382Z","shell.execute_reply":"2024-06-18T08:02:24.318466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nn_folds = 10\ncv = StratifiedKFold(n_splits=10, shuffle = True, random_state = 42)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T08:06:04.904523Z","iopub.execute_input":"2024-06-18T08:06:04.905456Z","iopub.status.idle":"2024-06-18T08:06:04.909913Z","shell.execute_reply.started":"2024-06-18T08:06:04.905417Z","shell.execute_reply":"2024-06-18T08:06:04.908990Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lstm_hist = pd.DataFrame()\n#lstm_best_score = []\n\naccuracy_list = []\nprecision_list = []\nrecall_list = []\nf1micro_list = []\nf1macro_list = []\nf1weighted_list = []\n\nfor index,(train_ids,test_ids) in enumerate(cv.split(x_combined, y_combined)):\n    X_train_c,X_valid_c,y_train_c,y_valid_c = x_combined[train_ids],x_combined[test_ids],y_combined[train_ids],y_combined[test_ids]\n    tf.keras.backend.clear_session()\n    with strategy.scope():\n        transformer_layer = (\n            transformers.TFBertModel.from_pretrained('bert-base-uncased')\n        )\n        model = build_model(transformer_layer, max_len=MAX_LEN)\n    checkpoint = ModelCheckpoint(\n    f'model1_{index}.h5', \n    monitor = 'val_accuracy', \n    verbose = 1, \n    save_best_only = True\n    )\n    train_dataset = (\n        tf.data.Dataset\n        .from_tensor_slices((X_train_c, y_train_c))\n        #.repeat()\n        .shuffle(2048)\n        .batch(BATCH_SIZE)\n        #.prefetch(AUTO)\n    )\n\n    valid_dataset = (\n        tf.data.Dataset\n        .from_tensor_slices((X_valid_c, y_valid_c))\n        .batch(BATCH_SIZE)\n        .cache()\n        #.prefetch(AUTO)\n    )\n    early_stopping = tf.keras.callbacks.EarlyStopping(patience=5)\n    print(\"Fold Number : \",index+1)\n    lstm_h = model.fit(train_dataset,batch_size = 16,epochs=100,validation_data=(valid_dataset),callbacks=[checkpoint,early_stopping])\n    #fold_hist_df = pd.DataFrame(lstm_h.history)\n    #lstm_hist = pd.concat([lstm_hist, fold_hist_df], ignore_index=True)\n    #lstm_best_score.append(mchpt.best)\n    model_load = load_model(f'model1_{index}.h5', custom_objects={'TFBertModel': TFBertModel})\n    y_pred_b = model_load.predict(X_test)\n    y_pred = np.argmax(y_pred_b, axis=1)\n    \n    accuracy = accuracy_score(y_test, y_pred)\n    precision = precision_score(y_test, y_pred, average='weighted')\n    recall = recall_score(y_test, y_pred, average='weighted')\n    f1_micro = f1_score(y_test, y_pred, average='micro')\n    f1_macro = f1_score(y_test, y_pred, average='macro')\n    f1_weighted = f1_score(y_test, y_pred, average='weighted')\n    \n    accuracy_list.append(accuracy)\n    precision_list.append(precision)\n    recall_list.append(recall)\n    f1micro_list.append(f1_micro)\n    f1macro_list.append(f1_macro)\n    f1weighted_list.append(f1_weighted)\n\naverage_accuracy = np.mean(accuracy_list)\naverage_precision = np.mean(precision_list)\naverage_recall = np.mean(recall_list)\naverage_f1micro = np.mean(f1micro_list)\naverage_f1macro = np.mean(f1macro_list)\naverage_f1weighted = np.mean(f1weighted_list)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T08:06:05.385150Z","iopub.execute_input":"2024-06-18T08:06:05.386096Z","iopub.status.idle":"2024-06-18T08:11:19.724154Z","shell.execute_reply.started":"2024-06-18T08:06:05.386035Z","shell.execute_reply":"2024-06-18T08:11:19.723222Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"average_accuracy:\", average_accuracy)\nprint(\"average_precision:\", average_precision)\nprint(\"average_recall:\", average_recall)\nprint(\"average_f1 micro:\", average_f1micro)\nprint(\"average_f1 macro:\", average_f1macro)\nprint(\"average_f1 weighted:\", average_f1weighted)\nprint(\"std:\", np.std(f1weighted_list))","metadata":{"execution":{"iopub.status.busy":"2024-06-18T08:12:14.563440Z","iopub.execute_input":"2024-06-18T08:12:14.564269Z","iopub.status.idle":"2024-06-18T08:12:14.570829Z","shell.execute_reply.started":"2024-06-18T08:12:14.564231Z","shell.execute_reply":"2024-06-18T08:12:14.569578Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(8, 6))\nplt.plot(range(1, len(f1weighted_list)+1), f1weighted_list, marker='o', linestyle='-')\nplt.title('Performance of BERT Across 10 Folds')\nplt.xlabel('Fold')\nplt.ylabel('Performance Score')\nplt.xticks(range(1, len(f1weighted_list)+1))\nplt.grid(True)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-06-18T08:12:26.012871Z","iopub.execute_input":"2024-06-18T08:12:26.013700Z","iopub.status.idle":"2024-06-18T08:12:26.229561Z","shell.execute_reply.started":"2024-06-18T08:12:26.013664Z","shell.execute_reply":"2024-06-18T08:12:26.228566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}