{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n**Info ℹ️: I refactored everything and include some new learnings - My submitted version for the competition was version 14 if you are interested. I used a different approach there and this version is a recap of everything**\n\nHi visitor,\nthis is my first NLP project and my first competition on Kaggle. I am familliar with the theoretical basics of NLP but never did a project on this topics especially with some pretrained models. So this is it. \n\nIn this project I first tried two approaches of pre-trained model. One where I load the pre-trained model manually in the embeddings layer and use that layer as a part of my model (glove) and the other one based on Huggingfaces🤗 framework, where I use the from_pretrained() function which loads the whole model (with all layers). This can be found in the notebook version where I submitted the competition with, which is version 14.\nThe current version/approach of the notebook is just the Huggingface🤗 edition because this makes all of this more readable and easier to understand 😁\n\nHINT - After the Competition:\nFor a better learning process I recaped my work and compared it with other competitions contributors work. One main notebook here was Jeremy Howards \"Iterate like a grandmaster!\" as well as the notebook of Mohamad Merchant who also wrote a blog article about \"Semantic Similarity with BERT\" on Keras, which handles the use of NLP models on Keras. The notebook which I got a lot inspired on can be found here: https://www.kaggle.com/code/mohamadmerchant/us-phrase-matching-tf-keras-train-tpu. \nI used a lot of bothes approaches in this notebook in the recap phase. Once again: If you want to see my initial approach where I got around 70% accuracy you should take a look at version 14. This was the version that I submitted to the competition. All work after this version is part of the recap phase and therefore full of inspiring code parts of other contributors.\n\nI thereforce ask you to bear with?! 🤗","metadata":{"papermill":{"duration":0.022206,"end_time":"2022-06-21T07:11:02.021009","exception":false,"start_time":"2022-06-21T07:11:01.998803","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Imports and Datasets","metadata":{"papermill":{"duration":0.02074,"end_time":"2022-06-21T07:11:02.063820","exception":false,"start_time":"2022-06-21T07:11:02.043080","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import sys\nassert sys.version_info >= (3,5)\nimport os\nimport pathlib\n\n# Is this notebook running on Colab or Kaggle?\nIS_COLAB = \"google.colab\" in sys.modules\nIS_KAGGLE = \"kaggle_secrets\" in sys.modules\n\nimport numpy as np\nimport pandas as pd\nfrom scipy import stats\nimport matplotlib.pyplot as plt\nfrom functools import partial\nimport seaborn as sns\nfrom datasets import Dataset\nfrom datasets import DatasetDict\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import Normalizer\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import StratifiedKFold\n\nimport nltk\nfrom string import punctuation\nfrom collections import Counter\n\nfrom scipy.spatial.distance import cosine\n\nimport tensorflow as tf\nfrom tensorflow import keras\nimport tensorflow_datasets as tfds\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras import layers\nfrom keras.layers import Embedding, LSTM, Dense, Dropout, CuDNNLSTM, Bidirectional\nfrom keras.layers.merge import concatenate\nfrom transformers import TrainingArguments\nfrom transformers import BertTokenizer, TFDebertaModel\nfrom transformers import RobertaTokenizer, TFRobertaModel, TFRobertaForSequenceClassification\nfrom transformers import TFAutoModel\n\n#import mlflow\n#from mlflow import log_metric, log_param, log_artifacts\n#import mlflow.tensorflow\n#from mlflow import pyfunc\n\nassert tf.__version__ >= \"2.0\"\n\nprint(f\"Tensorflow Version: {tf.__version__}\")\nprint(f\"Keras Version: {keras.__version__}\")\n\nif not tf.config.list_physical_devices('GPU'):\n    print(\"No GPU was detected. LSTMs and CNNs can be very slow without a GPU.\")\n    if IS_COLAB:\n        print(\"Go to Runtime > Change runtime and select a GPU hardware accelerator.\")\n    if IS_KAGGLE:\n        print(\"Go to Settings > Accelerator and select GPU.\")\nelse:\n    print(f'---Tensorflow is running with GPU Power now---')\n    sess = tf.compat.v1.Session(config=tf.compat.v1.ConfigProto(log_device_placement=True))\n    \n\n\nrandom_state=42\ntf.random.set_seed(random_state)\nnp.random.seed(random_state)\n\niskaggle = os.environ.get('KAGGLE_KERNEL_RUN_TYPE','')\n#kaggle = 0 # Kaggle path active = 1\n\nMAIN_PATH = os.getcwd()\n\n# change your local path here\nif iskaggle:\n    DATA_PATH = os.path.join(MAIN_PATH, '../input')\n    PHRASES_PATH = os.path.join(DATA_PATH, 'us-patent-phrase-to-phrase-matching')\nelse:\n    DATA_PATH = os.path.join(MAIN_PATH, 'data')\n    PHRASES_PATH = os.path.join(DATA_PATH,'input\\\\us-patent-phrase-to-phrase-matching')\n\n\n\nfor dirname, _, filenames in os.walk(PHRASES_PATH): \n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        ","metadata":{"papermill":{"duration":14.449925,"end_time":"2022-06-21T07:11:16.534173","exception":false,"start_time":"2022-06-21T07:11:02.084248","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-25T04:13:44.529976Z","iopub.execute_input":"2022-07-25T04:13:44.530424Z","iopub.status.idle":"2022-07-25T04:13:44.562193Z","shell.execute_reply.started":"2022-07-25T04:13:44.530383Z","shell.execute_reply":"2022-07-25T04:13:44.561445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get the Data","metadata":{"papermill":{"duration":0.021357,"end_time":"2022-06-21T07:11:16.576849","exception":false,"start_time":"2022-06-21T07:11:16.555492","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Data path and file\nCSV_FILE_TRAIN='train.csv'\nCSV_FILE_TEST='test.csv'\nCSV_FILE_COMF='sample_submission.csv'\nCSV_FILE_CPC='titles.csv'\nCPC_PATH='cpc-codes'\nDEBERTA_PATH='huggingface-deberta-variants'\nROBERTA_PATH='roberta-base'\n\ndef load_csv_data(path, csv_file):\n    csv_path = os.path.join(path, csv_file)\n    return pd.read_csv(csv_path)\n\ndef load_csv_data_manuel(path, csv_file):\n    csv_path = os.path.join(path, csv_file)\n    csv_file = open(csv_path, 'r')\n    csv_data = csv_file.readlines()\n    csv_file.close()\n    return csv_data\n    \n\ntrain = load_csv_data(PHRASES_PATH,CSV_FILE_TRAIN)\ntest = load_csv_data(PHRASES_PATH,CSV_FILE_TEST)\ncompetition_file = load_csv_data(PHRASES_PATH,CSV_FILE_COMF)\ncpc_code = load_csv_data(os.path.join(DATA_PATH, CPC_PATH), CSV_FILE_CPC)\n\n\nprint(f'Length of loaded trainset: {len(train)}')\nprint(f'Length of loaded testset: {len(test)}')\nprint(f'Length of loaded competition file: {len(competition_file)}')\nprint(f'Length of loaded cpc_codeset: {len(cpc_code)}')","metadata":{"papermill":{"duration":0.789531,"end_time":"2022-06-21T07:11:17.386787","exception":false,"start_time":"2022-06-21T07:11:16.597256","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-25T04:13:45.276465Z","iopub.execute_input":"2022-07-25T04:13:45.277199Z","iopub.status.idle":"2022-07-25T04:13:45.803399Z","shell.execute_reply.started":"2022-07-25T04:13:45.277159Z","shell.execute_reply":"2022-07-25T04:13:45.802358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.join(cpc_code.set_index('code'), on = 'context')\ntest = test.join(cpc_code.set_index('code'), on = 'context')","metadata":{"papermill":{"duration":0.191338,"end_time":"2022-06-21T07:11:17.599201","exception":false,"start_time":"2022-06-21T07:11:17.407863","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-25T04:13:45.805105Z","iopub.execute_input":"2022-07-25T04:13:45.805577Z","iopub.status.idle":"2022-07-25T04:13:45.962058Z","shell.execute_reply.started":"2022-07-25T04:13:45.805535Z","shell.execute_reply":"2022-07-25T04:13:45.961194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading Model Files","metadata":{"papermill":{"duration":0.020359,"end_time":"2022-06-21T07:11:17.640941","exception":false,"start_time":"2022-06-21T07:11:17.620582","status":"completed"},"tags":[]}},{"cell_type":"code","source":"if iskaggle:\n    ROBERTA_BASE = os.path.join(DATA_PATH, ROBERTA_PATH) # kaggle datasource location\nelse:\n    ROBERTA_BASE = 'roberta-base'","metadata":{"papermill":{"duration":0.027393,"end_time":"2022-06-21T07:11:17.783566","exception":false,"start_time":"2022-06-21T07:11:17.756173","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-25T04:13:45.968132Z","iopub.execute_input":"2022-07-25T04:13:45.970899Z","iopub.status.idle":"2022-07-25T04:13:45.974859Z","shell.execute_reply.started":"2022-07-25T04:13:45.970859Z","shell.execute_reply":"2022-07-25T04:13:45.974035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Understanding","metadata":{"papermill":{"duration":0.020152,"end_time":"2022-06-21T07:11:17.824009","exception":false,"start_time":"2022-06-21T07:11:17.803857","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## Given Attributes\n- id - a unique identifier for a pair of phrases\n- anchor - the first phrase\n- target - the second phrase\n- context - the CPC classification (version 2021.05), which indicates the subject within which the similarity is to be scored\n- score - the similarity. This is sourced from a combination of one or more manual expert ratings.\n\n\n## Score\nThe scores are in the 0-1 range with increments of 0.25 with the following meanings:\n\n- 1.0 - Very close match. This is typically an exact match except possibly for differences in conjugation, quantity (e.g. singular vs. plural), and addition or removal of stopwords (e.g. “the”, “and”, “or”).\n- 0.75 - Close synonym, e.g. “mobile phone” vs. “cellphone”. This also includes abbreviations, e.g. \"TCP\" -> \"transmission control protocol\".\n- 0.5 - Synonyms which don’t have the same meaning (same function, same properties). This includes broad-narrow (hyponym) and narrow-broad (hypernym) matches.\n- 0.25 - Somewhat related, e.g. the two phrases are in the same high level domain but are not synonyms. This also includes antonyms.\n- 0.0 - Unrelated.","metadata":{"papermill":{"duration":0.020289,"end_time":"2022-06-21T07:11:17.864877","exception":false,"start_time":"2022-06-21T07:11:17.844588","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train['anchor'].value_counts(dropna=False)","metadata":{"papermill":{"duration":0.04155,"end_time":"2022-06-21T07:11:17.926609","exception":false,"start_time":"2022-06-21T07:11:17.885059","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-25T04:13:46.667175Z","iopub.execute_input":"2022-07-25T04:13:46.667793Z","iopub.status.idle":"2022-07-25T04:13:46.681676Z","shell.execute_reply.started":"2022-07-25T04:13:46.667757Z","shell.execute_reply":"2022-07-25T04:13:46.680710Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The anchor value has 733 different values. Lets look at the target value.","metadata":{"papermill":{"duration":0.021466,"end_time":"2022-06-21T07:11:17.968895","exception":false,"start_time":"2022-06-21T07:11:17.947429","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train['target'].value_counts(dropna=False)","metadata":{"papermill":{"duration":0.047146,"end_time":"2022-06-21T07:11:18.036461","exception":false,"start_time":"2022-06-21T07:11:17.989315","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-25T04:13:47.178297Z","iopub.execute_input":"2022-07-25T04:13:47.178948Z","iopub.status.idle":"2022-07-25T04:13:47.205648Z","shell.execute_reply.started":"2022-07-25T04:13:47.178913Z","shell.execute_reply":"2022-07-25T04:13:47.204915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The target looks a little bit different. Here we have 29,340 different values.","metadata":{"papermill":{"duration":0.020871,"end_time":"2022-06-21T07:11:18.077783","exception":false,"start_time":"2022-06-21T07:11:18.056912","status":"completed"},"tags":[]}},{"cell_type":"code","source":"train['score'].value_counts(dropna=False)","metadata":{"papermill":{"duration":0.033853,"end_time":"2022-06-21T07:11:18.132060","exception":false,"start_time":"2022-06-21T07:11:18.098207","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-25T04:13:47.624658Z","iopub.execute_input":"2022-07-25T04:13:47.625270Z","iopub.status.idle":"2022-07-25T04:13:47.637350Z","shell.execute_reply.started":"2022-07-25T04:13:47.625236Z","shell.execute_reply":"2022-07-25T04:13:47.635139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['score'].value_counts(dropna=False).sort_index().plot.bar()","metadata":{"papermill":{"duration":0.22489,"end_time":"2022-06-21T07:11:18.378471","exception":false,"start_time":"2022-06-21T07:11:18.153581","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-25T04:13:47.871104Z","iopub.execute_input":"2022-07-25T04:13:47.871732Z","iopub.status.idle":"2022-07-25T04:13:48.034528Z","shell.execute_reply.started":"2022-07-25T04:13:47.871699Z","shell.execute_reply":"2022-07-25T04:13:48.033753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.groupby(['anchor', 'context']).count()","metadata":{"papermill":{"duration":0.073728,"end_time":"2022-06-21T07:11:18.474055","exception":false,"start_time":"2022-06-21T07:11:18.400327","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-25T04:13:48.092789Z","iopub.execute_input":"2022-07-25T04:13:48.093052Z","iopub.status.idle":"2022-07-25T04:13:48.140340Z","shell.execute_reply.started":"2022-07-25T04:13:48.093028Z","shell.execute_reply":"2022-07-25T04:13:48.139265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Configuration","metadata":{}},{"cell_type":"code","source":"class Config():\n    learning_rate = 1e-5\n    num_epochs = 10\n    batch_size = 32\n    decay = 0.01\n    max_line_length = 190\n    num_folds = 5\n\n    base_model = ROBERTA_BASE\n\n    root_logdir_tb = \"../../tensorboard-logs\"   # tensorboard logdir\n\n    def __init__(self, **kwargs):\n        for k, v in kwargs.items():\n            if k in self.__dict__:\n                setattr(self, k, v)\n            else:\n                raise KeyError(k)\n        \n\n\nconfig = Config()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:48.569177Z","iopub.execute_input":"2022-07-25T04:13:48.569765Z","iopub.status.idle":"2022-07-25T04:13:48.576172Z","shell.execute_reply.started":"2022-07-25T04:13:48.569732Z","shell.execute_reply":"2022-07-25T04:13:48.575165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preparation","metadata":{"papermill":{"duration":0.02174,"end_time":"2022-06-21T07:11:18.517959","exception":false,"start_time":"2022-06-21T07:11:18.496219","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"#### Loading Model","metadata":{}},{"cell_type":"code","source":"from transformers import AutoTokenizer\nfrom transformers import TFAutoModelForSequenceClassification","metadata":{"papermill":{"duration":0.198283,"end_time":"2022-06-21T07:25:28.718853","exception":false,"start_time":"2022-06-21T07:25:28.520570","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-25T04:13:49.585265Z","iopub.execute_input":"2022-07-25T04:13:49.585923Z","iopub.status.idle":"2022-07-25T04:13:49.590027Z","shell.execute_reply.started":"2022-07-25T04:13:49.585885Z","shell.execute_reply":"2022-07-25T04:13:49.588891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = AutoTokenizer.from_pretrained(config.base_model)","metadata":{"papermill":{"duration":0.321003,"end_time":"2022-06-21T07:25:29.221858","exception":false,"start_time":"2022-06-21T07:25:28.900855","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-25T04:13:49.882028Z","iopub.execute_input":"2022-07-25T04:13:49.882320Z","iopub.status.idle":"2022-07-25T04:13:49.972972Z","shell.execute_reply.started":"2022-07-25T04:13:49.882287Z","shell.execute_reply":"2022-07-25T04:13:49.972139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_pretrained = TFAutoModelForSequenceClassification.from_pretrained(config.base_model, trainable=True, return_dict=True, num_labels=5, output_hidden_states=True)","metadata":{"papermill":{"duration":7.567751,"end_time":"2022-06-21T07:25:36.971557","exception":false,"start_time":"2022-06-21T07:25:29.403806","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-25T04:13:50.146251Z","iopub.execute_input":"2022-07-25T04:13:50.147088Z","iopub.status.idle":"2022-07-25T04:13:56.795615Z","shell.execute_reply.started":"2022-07-25T04:13:50.147045Z","shell.execute_reply":"2022-07-25T04:13:56.794853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#tokenizer.add_special_tokens({'additional_special_tokens': context_list})","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:56.797345Z","iopub.execute_input":"2022-07-25T04:13:56.797854Z","iopub.status.idle":"2022-07-25T04:13:56.802307Z","shell.execute_reply.started":"2022-07-25T04:13:56.797798Z","shell.execute_reply":"2022-07-25T04:13:56.801140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Building the Input Value for the Model - The Text Corpus","metadata":{}},{"cell_type":"markdown","source":"Seperating the loaded cpc titles. They are concatenated by \";\".  ","metadata":{}},{"cell_type":"code","source":"# Seperating the cpc titles\ntrain['title'] = train.title.apply(lambda text: text.split(';'))\ntrain['title'] = train.title.apply(lambda context: ' '.join(context))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:56.803806Z","iopub.execute_input":"2022-07-25T04:13:56.804344Z","iopub.status.idle":"2022-07-25T04:13:57.061920Z","shell.execute_reply.started":"2022-07-25T04:13:56.804308Z","shell.execute_reply":"2022-07-25T04:13:57.061125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Special Tokens","metadata":{"papermill":{"duration":0.021974,"end_time":"2022-06-21T07:11:18.561571","exception":false,"start_time":"2022-06-21T07:11:18.539597","status":"completed"},"tags":[]}},{"cell_type":"code","source":"sep_token = tokenizer.sep_token\nprint(f'Seperater Token: {sep_token}')","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:57.063992Z","iopub.execute_input":"2022-07-25T04:13:57.064320Z","iopub.status.idle":"2022-07-25T04:13:57.069385Z","shell.execute_reply.started":"2022-07-25T04:13:57.064287Z","shell.execute_reply":"2022-07-25T04:13:57.068603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer.all_special_tokens","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:57.070728Z","iopub.execute_input":"2022-07-25T04:13:57.071246Z","iopub.status.idle":"2022-07-25T04:13:57.080512Z","shell.execute_reply.started":"2022-07-25T04:13:57.071209Z","shell.execute_reply":"2022-07-25T04:13:57.079484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Defining the context as special token for the Tokenizer","metadata":{}},{"cell_type":"code","source":"train['context_token'] = '[' + train['context'] + ']'\ntest['context_token'] = '[' + test['context'] + ']'\ncontext_list = list(train['context_token'].unique())","metadata":{"papermill":{"duration":0.042642,"end_time":"2022-06-21T07:11:18.625622","exception":false,"start_time":"2022-06-21T07:11:18.582980","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-25T04:13:57.082732Z","iopub.execute_input":"2022-07-25T04:13:57.083216Z","iopub.status.idle":"2022-07-25T04:13:57.100590Z","shell.execute_reply.started":"2022-07-25T04:13:57.083178Z","shell.execute_reply":"2022-07-25T04:13:57.099874Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['corpus'] = train['anchor'] + sep_token + train['target']\ntrain['corpus_w_context'] = train['context_token'] + sep_token + train['corpus']\ntrain['corpus_w_full_context'] = train['context_token'] + sep_token + train['corpus'] + sep_token + train['title']\n\ntest['corpus'] = test['anchor'] + sep_token + test['target']\ntest['corpus_w_context'] = test['context_token'] + sep_token + test['corpus']\ntest['corpus_w_full_context'] = train['context_token'] + sep_token + test['corpus'] + sep_token + test['title']","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:57.103678Z","iopub.execute_input":"2022-07-25T04:13:57.103952Z","iopub.status.idle":"2022-07-25T04:13:57.166263Z","shell.execute_reply.started":"2022-07-25T04:13:57.103929Z","shell.execute_reply":"2022-07-25T04:13:57.165379Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Train / Test / Val Data\n","metadata":{}},{"cell_type":"code","source":"anchors = train.anchor.unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:57.167686Z","iopub.execute_input":"2022-07-25T04:13:57.168001Z","iopub.status.idle":"2022-07-25T04:13:57.175653Z","shell.execute_reply.started":"2022-07-25T04:13:57.167968Z","shell.execute_reply":"2022-07-25T04:13:57.174767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Amount of diferent anchor values: {len(anchors)}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:57.177369Z","iopub.execute_input":"2022-07-25T04:13:57.177911Z","iopub.status.idle":"2022-07-25T04:13:57.183119Z","shell.execute_reply.started":"2022-07-25T04:13:57.177875Z","shell.execute_reply":"2022-07-25T04:13:57.182007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(random_state)\nnp.random.shuffle(anchors)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:57.186314Z","iopub.execute_input":"2022-07-25T04:13:57.187088Z","iopub.status.idle":"2022-07-25T04:13:57.191557Z","shell.execute_reply.started":"2022-07-25T04:13:57.187051Z","shell.execute_reply":"2022-07-25T04:13:57.190577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"anchors[:5]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:57.193015Z","iopub.execute_input":"2022-07-25T04:13:57.193676Z","iopub.status.idle":"2022-07-25T04:13:57.202309Z","shell.execute_reply.started":"2022-07-25T04:13:57.193639Z","shell.execute_reply":"2022-07-25T04:13:57.201481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This anchor set will work as the basement for the validation set slicing.","metadata":{}},{"cell_type":"code","source":"val_proportion = 0.25\nval_size = int(len(anchors)* val_proportion)\nval_anchors = anchors[:val_size]","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:57.203561Z","iopub.execute_input":"2022-07-25T04:13:57.204052Z","iopub.status.idle":"2022-07-25T04:13:57.210901Z","shell.execute_reply.started":"2022-07-25T04:13:57.204013Z","shell.execute_reply":"2022-07-25T04:13:57.210066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Slicing the data (or the over all index) with the validation index into train and validation index.","metadata":{}},{"cell_type":"code","source":"is_validation = np.isin(train.anchor, val_anchors)\nidxs = np.arange(len(train))","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:57.212293Z","iopub.execute_input":"2022-07-25T04:13:57.212698Z","iopub.status.idle":"2022-07-25T04:13:57.434404Z","shell.execute_reply.started":"2022-07-25T04:13:57.212665Z","shell.execute_reply":"2022-07-25T04:13:57.433714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_indexes = idxs[is_validation]\ntrain_indexes = idxs[~is_validation]\nlen(val_indexes), len(train_indexes)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:57.435668Z","iopub.execute_input":"2022-07-25T04:13:57.435987Z","iopub.status.idle":"2022-07-25T04:13:57.444257Z","shell.execute_reply.started":"2022-07-25T04:13:57.435954Z","shell.execute_reply":"2022-07-25T04:13:57.443328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Distribution of \"Score\" Values in Train / Val Set","metadata":{}},{"cell_type":"code","source":"train.iloc[train_indexes].score.mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:57.446190Z","iopub.execute_input":"2022-07-25T04:13:57.446664Z","iopub.status.idle":"2022-07-25T04:13:57.468463Z","shell.execute_reply.started":"2022-07-25T04:13:57.446601Z","shell.execute_reply":"2022-07-25T04:13:57.467551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.iloc[val_indexes].score.mean()","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:57.470064Z","iopub.execute_input":"2022-07-25T04:13:57.470433Z","iopub.status.idle":"2022-07-25T04:13:57.481461Z","shell.execute_reply.started":"2022-07-25T04:13:57.470397Z","shell.execute_reply":"2022-07-25T04:13:57.479978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Encoding","metadata":{}},{"cell_type":"markdown","source":"#### Tokenizer Funktion","metadata":{}},{"cell_type":"code","source":"def tokenize_fkt(text, tokenizer):\n    MAX_LINE_LENGTH = len(tokenizer(text).input_ids) # removed the tensorflow return\n    encoded_text = tokenizer.batch_encode_plus(\n        text,\n        add_special_tokens=False,\n        max_length=config.max_line_length,\n        padding='max_length',\n        truncation=True,\n        return_attention_mask=True,\n        return_token_type_ids=True,\n        return_tensors=\"tf\"\n        )\n\n    input_ids = np.array(encoded_text[\"input_ids\"], dtype=\"int32\")\n    attention_masks = np.array(encoded_text[\"attention_mask\"], dtype=\"int32\")\n    token_type_ids = np.array(encoded_text[\"token_type_ids\"], dtype=\"int32\")\n\n    return {\n        \"input_ids\": input_ids,\n        \"attention_masks\": attention_masks,\n        \"token_type_ids\": token_type_ids\n    }","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:57.484085Z","iopub.execute_input":"2022-07-25T04:13:57.484712Z","iopub.status.idle":"2022-07-25T04:13:57.492168Z","shell.execute_reply.started":"2022-07-25T04:13:57.484667Z","shell.execute_reply":"2022-07-25T04:13:57.491271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Build","metadata":{}},{"cell_type":"code","source":"def build_model(config,):\n    input_ids = tf.keras.Input(shape = (config.max_line_length ), dtype = tf.int32, name=\"input_ids\")\n    attention_masks = tf.keras.Input(shape = (config.max_line_length), dtype = tf.int32, name=\"attention_masks\")\n    token_type_ids = tf.keras.Input(shape = (config.max_line_length ), dtype = tf.int32, name=\"token_type_ids\")\n\n    base_model = TFAutoModel.from_pretrained(\n                                    config.base_model,\n                                    trainable=True,\n                                    return_dict=True,\n                                    num_labels=1,\n                                    output_hidden_states=True,\n                                    from_pt=True\n                                )\n\n    base_model_out = base_model(\n                            input_ids = input_ids,\n                            attention_mask = attention_masks,\n                            token_type_ids = token_type_ids,\n                            output_hidden_states=True\n                            )\n    \n    last_hidden_state = base_model_out.last_hidden_state\n\n    avg_pool = tf.keras.layers.GlobalAveragePooling1D()(last_hidden_state)\n    dropout = tf.keras.layers.Dropout(0.3)(avg_pool)\n    #x = tf.keras.layers.Dense(32, activation='relu')(x)\n    output = tf.keras.layers.Dense(1, activation=\"sigmoid\")(dropout)\n\n    model = tf.keras.models.Model(\n        inputs = [input_ids, attention_masks, token_type_ids],\n        outputs = output\n    )\n\n    model.compile(\n        optimizer = tf.keras.optimizers.Nadam(learning_rate=config.learning_rate),\n        loss = tf.keras.losses.BinaryCrossentropy()\n    )\n\n    return model\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:58.150118Z","iopub.execute_input":"2022-07-25T04:13:58.150444Z","iopub.status.idle":"2022-07-25T04:13:58.161721Z","shell.execute_reply.started":"2022-07-25T04:13:58.150416Z","shell.execute_reply":"2022-07-25T04:13:58.159584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Helpers for Keras Training","metadata":{}},{"cell_type":"markdown","source":"#### Pearson","metadata":{}},{"cell_type":"code","source":"class Pearsonr(tf.keras.callbacks.Callback):\n    def __init__(self, val_data, y_val):\n        self.val_data = val_data\n        self.y_val = y_val\n    \n    def on_epoch_end(self, epoch, logs):\n        val_preds = self.model.predict(self.val_data, verbose = 0)\n        \n        val_pearsonr = stats.pearsonr(self.y_val, val_preds.ravel())[0]\n        \n        print(f\"val_pearsonr: {val_pearsonr:.4f}\\n\")\n        logs[\"val_pearsonr\"] = val_pearsonr","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:13:59.254067Z","iopub.execute_input":"2022-07-25T04:13:59.254747Z","iopub.status.idle":"2022-07-25T04:13:59.261022Z","shell.execute_reply.started":"2022-07-25T04:13:59.254710Z","shell.execute_reply":"2022-07-25T04:13:59.259764Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Learningrate Scheduler","metadata":{}},{"cell_type":"code","source":"def lr_scheduler(epoch):\n    \"\"\"\n    Returns a custom learning rate that decreases as epochs progress.\n    \"\"\"\n    decay = config.decay\n    init_lr = config.learning_rate \n\n    #learning_rate = config.learning_rate * (1 / (1 + config.decay * epoch))\n    \n    if epoch == 0:\n        return init_lr * 0.05\n    else:\n        return init_lr * (0.8**epoch)\n\n    tf.summary.scalar('learning rate', data=learning_rate, step=epoch)\n    return learning_rate\n\n\nlr_callback = tf.keras.callbacks.LearningRateScheduler(lr_scheduler)\n","metadata":{"papermill":{"duration":0.955245,"end_time":"2022-06-21T07:11:22.120147","exception":false,"start_time":"2022-06-21T07:11:21.164902","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-25T04:21:15.182515Z","iopub.execute_input":"2022-07-25T04:21:15.183027Z","iopub.status.idle":"2022-07-25T04:21:15.190667Z","shell.execute_reply.started":"2022-07-25T04:21:15.182983Z","shell.execute_reply":"2022-07-25T04:21:15.189876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot([lr_scheduler(e) for e in range(10)])","metadata":{"execution":{"iopub.status.busy":"2022-07-25T04:21:17.154452Z","iopub.execute_input":"2022-07-25T04:21:17.155239Z","iopub.status.idle":"2022-07-25T04:21:17.328234Z","shell.execute_reply.started":"2022-07-25T04:21:17.155197Z","shell.execute_reply":"2022-07-25T04:21:17.327454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Tensorboard Logging","metadata":{}},{"cell_type":"code","source":"#from keras.callbacks import ReduceLROnPlateau\n#\n## Tensorboard logging structure function\n#root_logdir = \"../../tensorboard-logs\"\n#\n#def get_run_logdir(root_logdir, project):\n#    '''\n#    Returns logdir to the Tensorboard log for a specific project.\n#\n#            Parameters:\n#                    root_logdir (str) : basic logdir from Tensorboard\n#                    project (str): projectname that will be logged in TB\n#\n#            Returns:\n#                    os.path (str): Path to the final logdir\n#    '''\n#    import time\n#    run_id = time.strftime(\"run_%Y_%m_%d-%H_%M_%S\")\n#    project_logdir = os.path.join(root_logdir,project)\n#    return os.path.join(project_logdir, run_id)\n#\n#\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#tensorboard_callback = tf.keras.callbacks.TensorBoard(log_dir=get_run_logdir(config.root_logdir_tb,\"nlp_phrase2phrase_roberta\"), histogram_freq=1)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Training function with KFolds","metadata":{}},{"cell_type":"code","source":"def train_folds(train, config):\n    oof = np.zeros(len(train))\n    \n    train['score_map'] = train['score'].map({0.00: 0, 0.25: 1, 0.50: 2, 0.75: 3, 1.00: 4})\n    \n    skf = StratifiedKFold(n_splits = config.num_folds,\n                         shuffle = True,\n                         random_state = random_state)\n    \n    for fold, (train_indexes, val_indexes) in enumerate(skf.split(train, train['score_map'])):\n        print(f\"Training fold: {fold + 1}\")\n        \n        train_df = train.loc[train_indexes].reset_index(drop=True)\n        val_df = train.loc[val_indexes].reset_index(drop=True)\n        \n        train_encoded = tokenize_fkt(train_df['corpus_w_full_context'].tolist(), tokenizer)\n        val_encoded = tokenize_fkt(val_df['corpus_w_full_context'].tolist(), tokenizer)\n        \n        train_ds = tf.data.Dataset.from_tensor_slices((train_encoded, train_df['score'].tolist()))\n        val_ds = tf.data.Dataset.from_tensor_slices((val_encoded, val_df['score'].tolist()))\n        \n        train_ds = (\n            train_ds\n            .shuffle(1024)\n            .batch(config.batch_size)\n            .prefetch(tf.data.AUTOTUNE)\n        )\n\n        val_ds = (\n            val_ds\n            .batch(config.batch_size)\n            .prefetch(tf.data.AUTOTUNE)\n        )\n        \n        \n        # Callbacks\n        checkpoint = tf.keras.callbacks.ModelCheckpoint(f'model-{fold + 1}.h5',\n                                                       monitor = 'val_loss',\n                                                       mode = 'min',\n                                                       save_best_only = True,\n                                                       save_weights_only = True,\n                                                       save_freq = 'epoch',\n                                                       verbose = 1)\n        \n        earlystopping = tf.keras.callbacks.EarlyStopping(monitor='val_loss',\n                                                         mode='min',\n                                                         patience=3,\n                                                         restore_best_weights=True)\n        \n        lr_callback = tf.keras.callbacks.LearningRateScheduler(lr_scheduler)\n        \n        \n        pearsonr_callback = Pearsonr(val_ds, val_df['score'].values)\n        \n        \n        \n        num_train_steps = int(len(train) / config.batch_size * config.num_epochs)\n        \n        # Model building and training\n        model = build_model(config)\n        history = model.fit(\n            train_ds,\n            validation_data = val_ds,\n            epochs = config.num_epochs,\n            callbacks = [\n                pearsonr_callback,\n                checkpoint,\n                lr_callback,\n                earlystopping\n            ]\n        )\n        \n        print('\\nLoading best model weights ...')\n        model.load_weights(f'model-{fold + 1}.h5')\n        \n        print('Predicting OOF ...')\n        oof[val_indexes] = model.predict(val_ds,\n                                         batch_size = config.batch_size,\n                                         verbose=0\n                                        ).reshape(-1)\n        \n        score = stats.pearsonr(val_df['score'].values, oof[val_indexes])[0]\n        print(f'\\nFold {fold + 1}: OOF pearson_r: {score:.4f}')\n        print(\"*\" * 25)\n        \n    score = stats.pearsonr(train['score'].values, oof)[0]\n    print(f'\\nOverall OOF pearson_r: {score:.4f}')\n        \n    return oof\n\n        ","metadata":{"execution":{"iopub.status.busy":"2022-07-20T14:18:32.944500Z","iopub.execute_input":"2022-07-20T14:18:32.944831Z","iopub.status.idle":"2022-07-20T14:18:32.962547Z","shell.execute_reply.started":"2022-07-20T14:18:32.944798Z","shell.execute_reply":"2022-07-20T14:18:32.961633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Fitting","metadata":{}},{"cell_type":"code","source":"oof_preds = train_folds(train, config)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-20T14:18:32.963950Z","iopub.execute_input":"2022-07-20T14:18:32.964306Z","iopub.status.idle":"2022-07-20T14:21:16.485347Z","shell.execute_reply.started":"2022-07-20T14:18:32.964274Z","shell.execute_reply":"2022-07-20T14:21:16.484001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Evaluation","metadata":{"papermill":{"duration":0.689944,"end_time":"2022-06-21T08:13:56.126043","exception":false,"start_time":"2022-06-21T08:13:55.436099","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def predict_folds(test, config):\n    preds = []\n    \n    for fold in range(config.num_folds):\n        print(f'Predicting fold: {fold + 1}')\n        \n        test_encoded = tokenize_fkt(test['corpus_w_full_context'].tolist(), tokenizer)\n        \n        test_ds = tf.data.Dataset.from_tensor_slices((test_encoded))\n        \n        \n        test_ds = (\n            test_ds\n            .batch(config.batch_size)\n            .prefetch(tf.data.AUTOTUNE)\n        )\n\n               \n        # Model building and prediction\n        model = build_model(config)\n        print(f'Loading best trained model weights ...')\n        model.load_weights(f'model-{fold + 1}.h5')       \n\n        preds.append(\n            model.predict(test_ds,\n                          batch_size = config.batch_size,\n                          verbose=1).reshape(-1)\n                         )\n        \n    preds = np.mean(preds, axis=0)\n    return preds","metadata":{"execution":{"iopub.status.busy":"2022-07-20T14:21:16.492414Z","iopub.status.idle":"2022-07-20T14:21:16.493256Z","shell.execute_reply.started":"2022-07-20T14:21:16.492976Z","shell.execute_reply":"2022-07-20T14:21:16.493000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission File","metadata":{"papermill":{"duration":0.742733,"end_time":"2022-06-21T08:15:20.699473","exception":false,"start_time":"2022-06-21T08:15:19.956740","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## Training on all Data","metadata":{"papermill":{"duration":0.749132,"end_time":"2022-06-21T08:15:22.150926","exception":false,"start_time":"2022-06-21T08:15:21.401794","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"## Prediction of Test File Values","metadata":{"papermill":{"duration":0.691666,"end_time":"2022-06-21T08:15:23.526545","exception":false,"start_time":"2022-06-21T08:15:22.834879","status":"completed"},"tags":[]}},{"cell_type":"code","source":"competition_file = pd.DataFrame(columns=['score'])\ncompetition_file = pd.read_csv(PHRASES_PATH + \"/sample_submission.csv\")","metadata":{"papermill":{"duration":0.944517,"end_time":"2022-06-21T08:15:25.253826","exception":false,"start_time":"2022-06-21T08:15:24.309309","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-18T04:20:23.495764Z","iopub.execute_input":"2022-07-18T04:20:23.496123Z","iopub.status.idle":"2022-07-18T04:20:23.507767Z","shell.execute_reply.started":"2022-07-18T04:20:23.496092Z","shell.execute_reply":"2022-07-18T04:20:23.506967Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_prediction = predict_folds(test, config)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_file['score'] = test_prediction","metadata":{"papermill":{"duration":0.702957,"end_time":"2022-06-21T08:15:32.242425","exception":false,"start_time":"2022-06-21T08:15:31.539468","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-18T04:20:44.659095Z","iopub.execute_input":"2022-07-18T04:20:44.659436Z","iopub.status.idle":"2022-07-18T04:20:44.664270Z","shell.execute_reply.started":"2022-07-18T04:20:44.659406Z","shell.execute_reply":"2022-07-18T04:20:44.663242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_file['score'].hist()","metadata":{"papermill":{"duration":0.872058,"end_time":"2022-06-21T08:15:33.862954","exception":false,"start_time":"2022-06-21T08:15:32.990896","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-18T04:20:48.786406Z","iopub.execute_input":"2022-07-18T04:20:48.787345Z","iopub.status.idle":"2022-07-18T04:20:48.969181Z","shell.execute_reply.started":"2022-07-18T04:20:48.787298Z","shell.execute_reply":"2022-07-18T04:20:48.968393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"competition_file.to_csv('submission.csv', index=False)","metadata":{"papermill":{"duration":0.825722,"end_time":"2022-06-21T08:15:35.383673","exception":false,"start_time":"2022-06-21T08:15:34.557951","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-07-18T04:20:53.690046Z","iopub.execute_input":"2022-07-18T04:20:53.690389Z","iopub.status.idle":"2022-07-18T04:20:53.697884Z","shell.execute_reply.started":"2022-07-18T04:20:53.690360Z","shell.execute_reply":"2022-07-18T04:20:53.697125Z"},"trusted":true},"execution_count":null,"outputs":[]}]}