{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-07-03T08:39:46.745098Z","iopub.execute_input":"2023-07-03T08:39:46.745451Z","iopub.status.idle":"2023-07-03T08:39:46.755982Z","shell.execute_reply.started":"2023-07-03T08:39:46.745421Z","shell.execute_reply":"2023-07-03T08:39:46.754767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Imports\nDataset formulation and metric calculation adapted from https://www.kaggle.com/code/aravindanr22052001/stackoverflowrun","metadata":{}},{"cell_type":"code","source":"import torch \nimport numpy as np\nimport pandas as pd \nimport matplotlib.pyplot as plt \nimport os\nimport shutil\n\nimport tensorflow as tf\nimport tensorflow_hub as hub\nimport tensorflow_text as text\n# from official.nlp import optimization  # to create AdamW optimizer\n\nimport matplotlib.pyplot as plt\nimport tqdm","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:39:48.697912Z","iopub.execute_input":"2023-07-03T08:39:48.698613Z","iopub.status.idle":"2023-07-03T08:39:52.421916Z","shell.execute_reply.started":"2023-07-03T08:39:48.698578Z","shell.execute_reply":"2023-07-03T08:39:52.420971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tf-models-official\nfrom official.nlp import optimization  # to create AdamW optimizer\n","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:39:52.423820Z","iopub.execute_input":"2023-07-03T08:39:52.424521Z","iopub.status.idle":"2023-07-03T08:40:34.452527Z","shell.execute_reply.started":"2023-07-03T08:39:52.424487Z","shell.execute_reply":"2023-07-03T08:40:34.451468Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import the data \ndata = pd.read_csv(\"../input/predict-closed-questions-on-stack-overflow/train-sample.csv\")\ndata.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:41:22.150259Z","iopub.execute_input":"2023-07-03T08:41:22.150615Z","iopub.status.idle":"2023-07-03T08:41:25.885770Z","shell.execute_reply.started":"2023-07-03T08:41:22.150585Z","shell.execute_reply":"2023-07-03T08:41:25.884832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# look at the label distribution\ndata.OpenStatus.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:41:25.887710Z","iopub.execute_input":"2023-07-03T08:41:25.888180Z","iopub.status.idle":"2023-07-03T08:41:25.918401Z","shell.execute_reply.started":"2023-07-03T08:41:25.888146Z","shell.execute_reply":"2023-07-03T08:41:25.917041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's take 'TITLE' & 'BODYMARKDOWN' & OpenStatus Columns (only the text features and label)\ndata_train = data.loc[:, ['Title', 'BodyMarkdown', 'OpenStatus']]\ndata_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:41:27.932617Z","iopub.execute_input":"2023-07-03T08:41:27.932996Z","iopub.status.idle":"2023-07-03T08:41:27.957023Z","shell.execute_reply.started":"2023-07-03T08:41:27.932967Z","shell.execute_reply":"2023-07-03T08:41:27.956025Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preprocess","metadata":{}},{"cell_type":"code","source":"# convert all words to lowercase (not neccsary as bert tokenizer is case agnostic)\ndata_train.loc[:, 'Title'] = data_train['Title'].str.lower().values\ndata_train.loc[:, 'BodyMarkdown'] = data_train['BodyMarkdown'].str.lower().values\ndata_train.loc[:, 'OpenStatus'] = data_train['OpenStatus'].str.lower().values","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:41:33.686846Z","iopub.execute_input":"2023-07-03T08:41:33.687216Z","iopub.status.idle":"2023-07-03T08:41:34.070449Z","shell.execute_reply.started":"2023-07-03T08:41:33.687185Z","shell.execute_reply":"2023-07-03T08:41:34.069450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_mapping = {\n    'open': 0, \n    'not a real question': 1, \n    'off topic': 2, \n    'not constructive': 3, \n    'too localized': 4\n}\ndata_train['OpenStatus']= data_train['OpenStatus'].map(label_mapping) \ndata_train.head()","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:41:34.601598Z","iopub.execute_input":"2023-07-03T08:41:34.601950Z","iopub.status.idle":"2023-07-03T08:41:34.640279Z","shell.execute_reply.started":"2023-07-03T08:41:34.601921Z","shell.execute_reply":"2023-07-03T08:41:34.639435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train.OpenStatus.value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-07-01T15:04:52.549192Z","iopub.execute_input":"2023-07-01T15:04:52.549525Z","iopub.status.idle":"2023-07-01T15:04:52.558241Z","shell.execute_reply.started":"2023-07-01T15:04:52.549494Z","shell.execute_reply":"2023-07-01T15:04:52.557303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let's seperate x and y: \nx = data_train['Title'] + ' '+ data_train['BodyMarkdown']\ny = data_train['OpenStatus']\n\n# def one_hot(label):\n#     return tf.one_hot(indices = label, depth = 5, axis = -1)\n\n# y = y.apply(lambda label: one_hot(label))","metadata":{"execution":{"iopub.status.busy":"2023-07-01T15:04:52.559861Z","iopub.execute_input":"2023-07-01T15:04:52.560505Z","iopub.status.idle":"2023-07-01T15:04:52.704332Z","shell.execute_reply.started":"2023-07-01T15:04:52.560474Z","shell.execute_reply":"2023-07-01T15:04:52.703378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# xtrain, ytrain \nfrom sklearn.model_selection import train_test_split \nxtrain, xtest, ytrain, ytest = train_test_split(x, y, test_size = 0.3, random_state = 203)","metadata":{"execution":{"iopub.status.busy":"2023-07-01T15:04:52.705770Z","iopub.execute_input":"2023-07-01T15:04:52.706127Z","iopub.status.idle":"2023-07-01T15:04:52.733252Z","shell.execute_reply.started":"2023-07-01T15:04:52.706096Z","shell.execute_reply":"2023-07-01T15:04:52.732402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### preprocess the text and then batch them\n","metadata":{}},{"cell_type":"code","source":"train_dataset = tf.data.Dataset.from_tensor_slices((xtrain.values, ytrain.values))\ntest_dataset = tf.data.Dataset.from_tensor_slices((xtest.values, ytest.values))","metadata":{"execution":{"iopub.status.busy":"2023-07-02T06:04:40.924567Z","iopub.execute_input":"2023-07-02T06:04:40.925661Z","iopub.status.idle":"2023-07-02T06:04:41.141046Z","shell.execute_reply.started":"2023-07-02T06:04:40.925622Z","shell.execute_reply":"2023-07-02T06:04:41.140081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 32\n\ntrain_dataset = train_dataset.batch(batch_size)\ntrain_dataset = train_dataset.shuffle(buffer_size=len(train_dataset))\n\ntest_dataset = test_dataset.batch(batch_size)\ntest_dataset = test_dataset.shuffle(buffer_size=len(test_dataset))","metadata":{"execution":{"iopub.status.busy":"2023-07-02T06:04:43.025986Z","iopub.execute_input":"2023-07-02T06:04:43.026687Z","iopub.status.idle":"2023-07-02T06:04:43.047665Z","shell.execute_reply.started":"2023-07-02T06:04:43.026653Z","shell.execute_reply":"2023-07-02T06:04:43.046684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modelling\n\nAdapted from https://www.tensorflow.org/text/tutorials/classify_text_with_bert","metadata":{}},{"cell_type":"code","source":"tf.get_logger().setLevel('ERROR')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bert_model_name = 'small_bert/bert_en_uncased_L-4_H-512_A-8' \n\nmap_name_to_handle = {\n    'bert_en_uncased_L-12_H-768_A-12':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_L-12_H-768_A-12/3',\n    'bert_en_cased_L-12_H-768_A-12':\n        'https://tfhub.dev/tensorflow/bert_en_cased_L-12_H-768_A-12/3',\n    'bert_multi_cased_L-12_H-768_A-12':\n        'https://tfhub.dev/tensorflow/bert_multi_cased_L-12_H-768_A-12/3',\n    'small_bert/bert_en_uncased_L-2_H-128_A-2':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-2_H-128_A-2/1',\n    'small_bert/bert_en_uncased_L-2_H-256_A-4':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-2_H-256_A-4/1',\n    'small_bert/bert_en_uncased_L-2_H-512_A-8':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-2_H-512_A-8/1',\n    'small_bert/bert_en_uncased_L-2_H-768_A-12':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-2_H-768_A-12/1',\n    'small_bert/bert_en_uncased_L-4_H-128_A-2':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-4_H-128_A-2/1',\n    'small_bert/bert_en_uncased_L-4_H-256_A-4':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-4_H-256_A-4/1',\n    'small_bert/bert_en_uncased_L-4_H-512_A-8':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-4_H-512_A-8/1',\n    'small_bert/bert_en_uncased_L-4_H-768_A-12':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-4_H-768_A-12/1',\n    'small_bert/bert_en_uncased_L-6_H-128_A-2':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-6_H-128_A-2/1',\n    'small_bert/bert_en_uncased_L-6_H-256_A-4':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-6_H-256_A-4/1',\n    'small_bert/bert_en_uncased_L-6_H-512_A-8':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-6_H-512_A-8/1',\n    'small_bert/bert_en_uncased_L-6_H-768_A-12':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-6_H-768_A-12/1',\n    'small_bert/bert_en_uncased_L-8_H-128_A-2':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-8_H-128_A-2/1',\n    'small_bert/bert_en_uncased_L-8_H-256_A-4':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-8_H-256_A-4/1',\n    'small_bert/bert_en_uncased_L-8_H-512_A-8':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-8_H-512_A-8/1',\n    'small_bert/bert_en_uncased_L-8_H-768_A-12':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-8_H-768_A-12/1',\n    'small_bert/bert_en_uncased_L-10_H-128_A-2':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-10_H-128_A-2/1',\n    'small_bert/bert_en_uncased_L-10_H-256_A-4':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-10_H-256_A-4/1',\n    'small_bert/bert_en_uncased_L-10_H-512_A-8':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-10_H-512_A-8/1',\n    'small_bert/bert_en_uncased_L-10_H-768_A-12':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-10_H-768_A-12/1',\n    'small_bert/bert_en_uncased_L-12_H-128_A-2':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-12_H-128_A-2/1',\n    'small_bert/bert_en_uncased_L-12_H-256_A-4':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-12_H-256_A-4/1',\n    'small_bert/bert_en_uncased_L-12_H-512_A-8':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-12_H-512_A-8/1',\n    'small_bert/bert_en_uncased_L-12_H-768_A-12':\n        'https://tfhub.dev/tensorflow/small_bert/bert_en_uncased_L-12_H-768_A-12/1',\n    'albert_en_base':\n        'https://tfhub.dev/tensorflow/albert_en_base/2',\n    'electra_small':\n        'https://tfhub.dev/google/electra_small/2',\n    'electra_base':\n        'https://tfhub.dev/google/electra_base/2',\n    'experts_pubmed':\n        'https://tfhub.dev/google/experts/bert/pubmed/2',\n    'experts_wiki_books':\n        'https://tfhub.dev/google/experts/bert/wiki_books/2',\n    'talking-heads_base':\n        'https://tfhub.dev/tensorflow/talkheads_ggelu_bert_en_base/1',\n}\n\nmap_model_to_preprocess = {\n    'bert_en_uncased_L-12_H-768_A-12':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'bert_en_cased_L-12_H-768_A-12':\n        'https://tfhub.dev/tensorflow/bert_en_cased_preprocess/3',\n    'small_bert/bert_en_uncased_L-2_H-128_A-2':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-2_H-256_A-4':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-2_H-512_A-8':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-2_H-768_A-12':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-4_H-128_A-2':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-4_H-256_A-4':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-4_H-512_A-8':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-4_H-768_A-12':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-6_H-128_A-2':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-6_H-256_A-4':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-6_H-512_A-8':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-6_H-768_A-12':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-8_H-128_A-2':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-8_H-256_A-4':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-8_H-512_A-8':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-8_H-768_A-12':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-10_H-128_A-2':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-10_H-256_A-4':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-10_H-512_A-8':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-10_H-768_A-12':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-12_H-128_A-2':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-12_H-256_A-4':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-12_H-512_A-8':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'small_bert/bert_en_uncased_L-12_H-768_A-12':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'bert_multi_cased_L-12_H-768_A-12':\n        'https://tfhub.dev/tensorflow/bert_multi_cased_preprocess/3',\n    'albert_en_base':\n        'https://tfhub.dev/tensorflow/albert_en_preprocess/3',\n    'electra_small':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'electra_base':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'experts_pubmed':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'experts_wiki_books':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n    'talking-heads_base':\n        'https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3',\n}\n\ntfhub_handle_encoder = map_name_to_handle[bert_model_name]\ntfhub_handle_preprocess = map_model_to_preprocess[bert_model_name]\n\nprint(f'BERT model selected           : {tfhub_handle_encoder}')\nprint(f'Preprocess model auto-selected: {tfhub_handle_preprocess}')","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:41:38.958772Z","iopub.execute_input":"2023-07-03T08:41:38.959559Z","iopub.status.idle":"2023-07-03T08:41:38.975288Z","shell.execute_reply.started":"2023-07-03T08:41:38.959522Z","shell.execute_reply":"2023-07-03T08:41:38.974388Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Check that the preprocesser and model works","metadata":{}},{"cell_type":"code","source":"bert_preprocess_model = hub.KerasLayer(tfhub_handle_preprocess)","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:45:25.619631Z","iopub.execute_input":"2023-07-03T08:45:25.619980Z","iopub.status.idle":"2023-07-03T08:45:27.394228Z","shell.execute_reply.started":"2023-07-03T08:45:25.619951Z","shell.execute_reply":"2023-07-03T08:45:27.393304Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_test = ['this is such an amazing movie!']\ntext_preprocessed = bert_preprocess_model(text_test)\n\nprint(f'Keys       : {list(text_preprocessed.keys())}')\nprint(f'Shape      : {text_preprocessed[\"input_word_ids\"].shape}')\nprint(f'Word Ids   : {text_preprocessed[\"input_word_ids\"][0, :12]}')\nprint(f'Input Mask : {text_preprocessed[\"input_mask\"][0, :12]}')\nprint(f'Type Ids   : {text_preprocessed[\"input_type_ids\"][0, :12]}')","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:45:31.867137Z","iopub.execute_input":"2023-07-03T08:45:31.867997Z","iopub.status.idle":"2023-07-03T08:45:32.104256Z","shell.execute_reply.started":"2023-07-03T08:45:31.867952Z","shell.execute_reply":"2023-07-03T08:45:32.103252Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import BertTokenizer\ntz = BertTokenizer.from_pretrained(\"bert-base-uncased\")\ntz.convert_ids_to_tokens(text_preprocessed[\"input_word_ids\"][0, :12])\n# note the [CLS] token here is implicitly added by the preprocessing module","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:45:33.288577Z","iopub.execute_input":"2023-07-03T08:45:33.289295Z","iopub.status.idle":"2023-07-03T08:45:34.947848Z","shell.execute_reply.started":"2023-07-03T08:45:33.289257Z","shell.execute_reply":"2023-07-03T08:45:34.946105Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets check length of the longest text in the dataset, since the maximum sequence length accepted by small bert is 128\nxtrain.str.split().str.len().describe()","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:45:35.386957Z","iopub.execute_input":"2023-07-03T08:45:35.387531Z","iopub.status.idle":"2023-07-03T08:45:36.926611Z","shell.execute_reply.started":"2023-07-03T08:45:35.387488Z","shell.execute_reply":"2023-07-03T08:45:36.925498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bert_model = hub.KerasLayer(tfhub_handle_encoder)","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:45:38.608878Z","iopub.execute_input":"2023-07-03T08:45:38.609888Z","iopub.status.idle":"2023-07-03T08:45:44.527119Z","shell.execute_reply.started":"2023-07-03T08:45:38.609852Z","shell.execute_reply":"2023-07-03T08:45:44.525992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bert_results = bert_model(text_preprocessed)\n\nprint(f'Loaded BERT: {tfhub_handle_encoder}')\nprint(f'Pooled Outputs Shape:{bert_results[\"pooled_output\"].shape}')\nprint(f'Pooled Outputs Values:{bert_results[\"pooled_output\"][0, :12]}')\nprint(f'Sequence Outputs Shape:{bert_results[\"sequence_output\"].shape}')\nprint(f'Sequence Outputs Values:{bert_results[\"sequence_output\"][0, :12]}')","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:45:44.569289Z","iopub.execute_input":"2023-07-03T08:45:44.569654Z","iopub.status.idle":"2023-07-03T08:45:46.563762Z","shell.execute_reply.started":"2023-07-03T08:45:44.569621Z","shell.execute_reply":"2023-07-03T08:45:46.562720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DROPOUT_VAL = 0.1\nNUM_CLASSES = 5\ndef build_classifier_model():\n    text_input = tf.keras.layers.Input(shape=(), dtype=tf.string, name='text')\n    preprocessing_layer = hub.KerasLayer(tfhub_handle_preprocess, name='preprocessing')\n    encoder_inputs = preprocessing_layer(text_input)\n    encoder = hub.KerasLayer(tfhub_handle_encoder, trainable=True, name='BERT_encoder')\n    outputs = encoder(encoder_inputs)\n    net = outputs['pooled_output']\n    net = tf.keras.layers.Dropout(DROPOUT_VAL)(net)\n#     net = tf.keras.layers.Dense(1, activation=None, name='classifier')(net)\n    net = tf.keras.layers.Dense(NUM_CLASSES, activation='softmax', name='classifier')(net)\n    return tf.keras.Model(text_input, net)","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:41:57.288625Z","iopub.execute_input":"2023-07-03T08:41:57.289004Z","iopub.status.idle":"2023-07-03T08:41:57.296449Z","shell.execute_reply.started":"2023-07-03T08:41:57.288958Z","shell.execute_reply":"2023-07-03T08:41:57.295095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifier_model = build_classifier_model()\nbert_raw_result = classifier_model(tf.constant(text_test))\nprint(bert_raw_result)","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:41:58.356463Z","iopub.execute_input":"2023-07-03T08:41:58.357231Z","iopub.status.idle":"2023-07-03T08:42:13.256347Z","shell.execute_reply.started":"2023-07-03T08:41:58.357184Z","shell.execute_reply":"2023-07-03T08:42:13.254880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(classifier_model)","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:41:52.036680Z","iopub.execute_input":"2023-07-03T08:41:52.037073Z","iopub.status.idle":"2023-07-03T08:41:52.081638Z","shell.execute_reply.started":"2023-07-03T08:41:52.037021Z","shell.execute_reply":"2023-07-03T08:41:52.079782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training the model proper","metadata":{}},{"cell_type":"code","source":"loss = tf.keras.losses.SparseCategoricalCrossentropy( # use sparse categorical cross entropy as labels are not one-hot encoded\n    from_logits=False,\n    name='categorical_crossentropy'\n)\nprint(loss.reduction)\nmetrics = tf.metrics.CategoricalAccuracy()","metadata":{"execution":{"iopub.status.busy":"2023-07-02T04:00:20.998591Z","iopub.execute_input":"2023-07-02T04:00:20.999113Z","iopub.status.idle":"2023-07-02T04:00:21.013981Z","shell.execute_reply.started":"2023-07-02T04:00:20.999077Z","shell.execute_reply":"2023-07-02T04:00:21.012841Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 2\nsteps_per_epoch = tf.data.experimental.cardinality(train_dataset).numpy()\nnum_train_steps = steps_per_epoch * epochs\nnum_warmup_steps = int(0.1*num_train_steps)\n\ninit_lr = 3e-5\noptimizer = optimization.create_optimizer(init_lr=init_lr,\n                                          num_train_steps=num_train_steps,\n                                          num_warmup_steps=num_warmup_steps,\n                                          optimizer_type='adamw')","metadata":{"execution":{"iopub.status.busy":"2023-07-02T06:04:47.976833Z","iopub.execute_input":"2023-07-02T06:04:47.977211Z","iopub.status.idle":"2023-07-02T06:04:47.983901Z","shell.execute_reply.started":"2023-07-02T06:04:47.977162Z","shell.execute_reply":"2023-07-02T06:04:47.983012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classifier_model.compile(optimizer=optimizer,\n                         loss=loss,\n                         metrics=metrics)","metadata":{"execution":{"iopub.status.busy":"2023-07-02T06:04:48.798782Z","iopub.execute_input":"2023-07-02T06:04:48.801296Z","iopub.status.idle":"2023-07-02T06:04:48.817807Z","shell.execute_reply.started":"2023-07-02T06:04:48.801263Z","shell.execute_reply":"2023-07-02T06:04:48.816816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Training model with {tfhub_handle_encoder}')\nhistory = classifier_model.fit(x=train_dataset,\n                               validation_data=test_dataset,\n                               epochs=epochs)","metadata":{"execution":{"iopub.status.busy":"2023-07-02T06:04:49.487822Z","iopub.execute_input":"2023-07-02T06:04:49.488401Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Alternative if you want to use categorical cross entropy by transforming the labels into one-hot encodings first\n\nmake sure that both the input and output have the same dimension depth","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:39:42.737895Z","iopub.execute_input":"2023-07-03T08:39:42.738641Z","iopub.status.idle":"2023-07-03T08:39:42.748153Z","shell.execute_reply.started":"2023-07-03T08:39:42.738607Z","shell.execute_reply":"2023-07-03T08:39:42.747111Z"}}},{"cell_type":"code","source":"from keras.utils.np_utils import to_categorical   \n\n# Let's seperate x and y: \nx = data_train['Title'] + ' '+ data_train['BodyMarkdown']\ny = data_train['OpenStatus']\n\ndef one_hot(label):\n    return to_categorical(label, num_classes=5)\n\ny = to_categorical(y)\n\n# def one_hot(label):\n#     return tf.one_hot(indices = label, depth = 5, axis = -1)\n\n# def one_hot(label):\n#     ls = [0] * 5\n#     ls[label] = 1\n#     return ls\n\n# y = y.apply(lambda label: one_hot(label))\n\n# xtrain, ytrain \nfrom sklearn.model_selection import train_test_split \nxtrain, xtest, ytrain, ytest = train_test_split(x, y, test_size = 0.3, random_state = 203)","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:56:45.426726Z","iopub.execute_input":"2023-07-03T08:56:45.427111Z","iopub.status.idle":"2023-07-03T08:56:45.600565Z","shell.execute_reply.started":"2023-07-03T08:56:45.427081Z","shell.execute_reply":"2023-07-03T08:56:45.599611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss = tf.keras.losses.CategoricalCrossentropy( # use sparse categorical cross entropy as labels are not one-hot encoded\n    from_logits=False,\n    name='categorical_crossentropy'\n)\nprint(loss.reduction)\nmetrics = tf.metrics.CategoricalAccuracy()","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:56:46.865888Z","iopub.execute_input":"2023-07-03T08:56:46.866704Z","iopub.status.idle":"2023-07-03T08:56:46.880691Z","shell.execute_reply.started":"2023-07-03T08:56:46.866670Z","shell.execute_reply":"2023-07-03T08:56:46.879603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 2\nsteps_per_epoch = len(xtrain)\nnum_train_steps = steps_per_epoch * epochs\nnum_warmup_steps = int(0.1*num_train_steps)\n\ninit_lr = 3e-5\noptimizer = optimization.create_optimizer(init_lr=init_lr,\n                                          num_train_steps=num_train_steps,\n                                          num_warmup_steps=num_warmup_steps,\n                                          optimizer_type='adamw')\n\nclassifier_model.compile(optimizer=optimizer,\n                         loss=loss,\n                         metrics=metrics)","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:56:48.081759Z","iopub.execute_input":"2023-07-03T08:56:48.082148Z","iopub.status.idle":"2023-07-03T08:56:48.095461Z","shell.execute_reply.started":"2023-07-03T08:56:48.082116Z","shell.execute_reply":"2023-07-03T08:56:48.094526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Training model with {tfhub_handle_encoder}')\nhistory = classifier_model.fit(x=xtrain,\n                               y=ytrain,\n                               batch_size = 32,\n                               epochs=epochs)","metadata":{"execution":{"iopub.status.busy":"2023-07-03T08:56:50.426700Z","iopub.execute_input":"2023-07-03T08:56:50.427274Z","iopub.status.idle":"2023-07-03T09:16:15.830976Z","shell.execute_reply.started":"2023-07-03T08:56:50.427237Z","shell.execute_reply":"2023-07-03T09:16:15.829986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}