{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom termcolor import colored\nfrom wordcloud import WordCloud, STOPWORDS\nimport plotly.express as px\nimport bq_helper\nfrom bq_helper import BigQueryHelper\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-10T06:49:33.903193Z","iopub.execute_input":"2022-08-10T06:49:33.903896Z","iopub.status.idle":"2022-08-10T06:49:33.916830Z","shell.execute_reply.started":"2022-08-10T06:49:33.903841Z","shell.execute_reply":"2022-08-10T06:49:33.914384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Building a model to match phrases in order to extract contextual information. In this competition, you will train your models on a novel semantic similarity dataset to extract relevant information by matching key phrases in patent documents. Determining the semantic similarity between phrases is critically important during the patent search and examination process to determine if an invention has been described before. For example, if one invention claims \"television set\" and a prior publication describes \"TV set\", a model would ideally recognize these are the same and assist.","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/us-patent-phrase-to-phrase-matching/train.csv\")\ntest_df = pd.read_csv(\"../input/us-patent-phrase-to-phrase-matching/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:33.921564Z","iopub.execute_input":"2022-08-10T06:49:33.922639Z","iopub.status.idle":"2022-08-10T06:49:34.095482Z","shell.execute_reply.started":"2022-08-10T06:49:33.922575Z","shell.execute_reply":"2022-08-10T06:49:34.086812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Looking at the first 10 samples ","metadata":{}},{"cell_type":"code","source":"train_df.sample(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:34.103030Z","iopub.execute_input":"2022-08-10T06:49:34.103408Z","iopub.status.idle":"2022-08-10T06:49:34.141571Z","shell.execute_reply.started":"2022-08-10T06:49:34.103368Z","shell.execute_reply":"2022-08-10T06:49:34.136597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have **Anchor** and **Target** , Anchor is the patent phrase and Target is phrase needed to match. The **Score** identifies how close they are to matching ranging from 0(not at all matching) to 1(identically matching) ","metadata":{}},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:34.145466Z","iopub.execute_input":"2022-08-10T06:49:34.146090Z","iopub.status.idle":"2022-08-10T06:49:34.205565Z","shell.execute_reply.started":"2022-08-10T06:49:34.146004Z","shell.execute_reply":"2022-08-10T06:49:34.204086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.isnull().sum(axis = 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:34.209237Z","iopub.execute_input":"2022-08-10T06:49:34.210712Z","iopub.status.idle":"2022-08-10T06:49:34.245263Z","shell.execute_reply.started":"2022-08-10T06:49:34.210667Z","shell.execute_reply":"2022-08-10T06:49:34.241879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[train_df.drop(\"id\", axis =1).duplicated()]","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:34.250525Z","iopub.execute_input":"2022-08-10T06:49:34.252271Z","iopub.status.idle":"2022-08-10T06:49:34.319189Z","shell.execute_reply.started":"2022-08-10T06:49:34.252227Z","shell.execute_reply":"2022-08-10T06:49:34.317476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n1. No missing values in the training data\n2. No duplicate values in the training data","metadata":{}},{"cell_type":"code","source":"train_df.anchor.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:34.325638Z","iopub.execute_input":"2022-08-10T06:49:34.329200Z","iopub.status.idle":"2022-08-10T06:49:34.350588Z","shell.execute_reply.started":"2022-08-10T06:49:34.329155Z","shell.execute_reply":"2022-08-10T06:49:34.348905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.anchor.value_counts().head(20)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:34.356271Z","iopub.execute_input":"2022-08-10T06:49:34.359691Z","iopub.status.idle":"2022-08-10T06:49:34.382209Z","shell.execute_reply.started":"2022-08-10T06:49:34.359648Z","shell.execute_reply":"2022-08-10T06:49:34.380859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pattern = 'base'\nmask = train_df['target'].str.contains(pattern, case=False, na=False)\ntrain_df.query(\"anchor == 'component composite coating'\")[mask]","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:34.387657Z","iopub.execute_input":"2022-08-10T06:49:34.390774Z","iopub.status.idle":"2022-08-10T06:49:34.489804Z","shell.execute_reply.started":"2022-08-10T06:49:34.390733Z","shell.execute_reply":"2022-08-10T06:49:34.488428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"anchor_desc = train_df[train_df.anchor.notnull()].anchor.values\nstopwords = set(STOPWORDS)\nwordcloud = WordCloud(width = 1000,\n                     height = 600,\n                     background_color = 'white',\n                     min_font_size = 4,\n                     stopwords = stopwords).generate(\" \".join(anchor_desc))\n\nplt.figure(figsize = (8,8))\nplt.imshow(wordcloud)\nplt.axis(\"off\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:34.495518Z","iopub.execute_input":"2022-08-10T06:49:34.498720Z","iopub.status.idle":"2022-08-10T06:49:36.359731Z","shell.execute_reply.started":"2022-08-10T06:49:34.498677Z","shell.execute_reply":"2022-08-10T06:49:36.358327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['anchor_len'] = train_df['anchor'].str.split().str.len()\nprint(f\"Anchors with a maximum length of 5: \\n{(train_df.query('anchor_len == 5')['anchor'].unique())}\")\nprint(f\"\\nAnchors with a maximum length of 4: \\n{(train_df.query('anchor_len == 4')['anchor'].unique())}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:36.366206Z","iopub.execute_input":"2022-08-10T06:49:36.367744Z","iopub.status.idle":"2022-08-10T06:49:36.422771Z","shell.execute_reply.started":"2022-08-10T06:49:36.367699Z","shell.execute_reply":"2022-08-10T06:49:36.421290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.anchor_len.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:36.424702Z","iopub.execute_input":"2022-08-10T06:49:36.425600Z","iopub.status.idle":"2022-08-10T06:49:36.436651Z","shell.execute_reply.started":"2022-08-10T06:49:36.425553Z","shell.execute_reply":"2022-08-10T06:49:36.435109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n1. Anchors have a maximum length of 5\n2. Most anchors have 2 length","metadata":{}},{"cell_type":"code","source":"pattern = '[0-9]'\nmask = train_df['anchor'].str.contains(pattern, na=False)\ntrain_df['num_anchor'] = mask\ntrain_df[mask]['anchor'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:36.438718Z","iopub.execute_input":"2022-08-10T06:49:36.439641Z","iopub.status.idle":"2022-08-10T06:49:36.475136Z","shell.execute_reply.started":"2022-08-10T06:49:36.439596Z","shell.execute_reply":"2022-08-10T06:49:36.473620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n1. Only 4 observations with numbers in it","metadata":{}},{"cell_type":"code","source":"train_df.target.nunique()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:36.477539Z","iopub.execute_input":"2022-08-10T06:49:36.477968Z","iopub.status.idle":"2022-08-10T06:49:36.496765Z","shell.execute_reply.started":"2022-08-10T06:49:36.477925Z","shell.execute_reply":"2022-08-10T06:49:36.495315Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.target.value_counts().head(20)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:36.499074Z","iopub.execute_input":"2022-08-10T06:49:36.499919Z","iopub.status.idle":"2022-08-10T06:49:36.525277Z","shell.execute_reply.started":"2022-08-10T06:49:36.499877Z","shell.execute_reply":"2022-08-10T06:49:36.523881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_desc = train_df[train_df.target.notnull()].target.values\nstopwords = set(STOPWORDS)\nwordcloud = WordCloud(width = 1000,\n                     height = 600,\n                     background_color = 'white',\n                     min_font_size = 4,\n                     stopwords = stopwords).generate(\" \".join(target_desc))\n\nplt.figure(figsize = (8,8))\nplt.imshow(wordcloud)\nplt.axis(\"off\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:36.527510Z","iopub.execute_input":"2022-08-10T06:49:36.528540Z","iopub.status.idle":"2022-08-10T06:49:38.679801Z","shell.execute_reply.started":"2022-08-10T06:49:36.528496Z","shell.execute_reply":"2022-08-10T06:49:38.678420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['target_len'] = train_df['target'].str.split().str.len()\ntrain_df.target_len.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:38.681673Z","iopub.execute_input":"2022-08-10T06:49:38.682802Z","iopub.status.idle":"2022-08-10T06:49:38.736094Z","shell.execute_reply.started":"2022-08-10T06:49:38.682761Z","shell.execute_reply":"2022-08-10T06:49:38.734595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n1. The longest target character is 15 characters long","metadata":{}},{"cell_type":"code","source":"print(f\"Target with a maximum length of 15: \\n{(train_df.query('target_len == 15')['target'].unique())}\")\nprint(f\"\\nTarget with a maximum length of 13: \\n{(train_df.query('target_len == 13')['target'].unique())}\")\nprint(f\"\\nTarget with a maximum length of 12: \\n{(train_df.query('target_len == 12')['target'].unique())}\")\nprint(f\"\\nTarget with a maximum length of 11: \\n{(train_df.query('target_len == 11')['target'].unique())}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:38.737888Z","iopub.execute_input":"2022-08-10T06:49:38.739137Z","iopub.status.idle":"2022-08-10T06:49:38.764258Z","shell.execute_reply.started":"2022-08-10T06:49:38.739090Z","shell.execute_reply":"2022-08-10T06:49:38.762731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pattern = '[0-9]'\nmask = train_df['target'].str.contains(pattern, na=False)\ntrain_df['num_target'] = mask\ntrain_df[mask]['target'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:38.766327Z","iopub.execute_input":"2022-08-10T06:49:38.767667Z","iopub.status.idle":"2022-08-10T06:49:38.807116Z","shell.execute_reply.started":"2022-08-10T06:49:38.767622Z","shell.execute_reply":"2022-08-10T06:49:38.805603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n1. There's 112 observations which have numbers","metadata":{}},{"cell_type":"markdown","source":"## Model\nWe could represent the input to the model as something like \"TEXT1: abatement; TEXT2: eliminating process\". We'll need to add the context to this too. In Pandas, we just use + to concatenate,","metadata":{}},{"cell_type":"code","source":"train_df['input'] = 'TEXT1: ' + train_df.context + ' ;TEXT2: ' + train_df.target + ' ;ANC1: '+ train_df.anchor ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:38.808949Z","iopub.execute_input":"2022-08-10T06:49:38.809787Z","iopub.status.idle":"2022-08-10T06:49:38.839451Z","shell.execute_reply.started":"2022-08-10T06:49:38.809743Z","shell.execute_reply":"2022-08-10T06:49:38.838085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.input.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:38.841882Z","iopub.execute_input":"2022-08-10T06:49:38.842729Z","iopub.status.idle":"2022-08-10T06:49:38.854740Z","shell.execute_reply.started":"2022-08-10T06:49:38.842662Z","shell.execute_reply":"2022-08-10T06:49:38.853253Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Tokenization\nTransformers need Dataset object to store Dataset","metadata":{}},{"cell_type":"code","source":"from datasets import Dataset, DatasetDict\nds = Dataset.from_pandas(train_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:38.857676Z","iopub.execute_input":"2022-08-10T06:49:38.858685Z","iopub.status.idle":"2022-08-10T06:49:38.888165Z","shell.execute_reply.started":"2022-08-10T06:49:38.858640Z","shell.execute_reply":"2022-08-10T06:49:38.886611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#How it is represented\nds","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:38.891058Z","iopub.execute_input":"2022-08-10T06:49:38.892352Z","iopub.status.idle":"2022-08-10T06:49:38.901966Z","shell.execute_reply.started":"2022-08-10T06:49:38.892307Z","shell.execute_reply":"2022-08-10T06:49:38.900502Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Obviously a deep learning model cannot take text as input. It takes numbers as input. So we need to do two things:\n* Tokenization: Splitting each text into words/tokens\n* Numericalization: Converting each token into numbers","metadata":{}},{"cell_type":"code","source":"model_nm = 'microsoft/deberta-v3-small'","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:38.904669Z","iopub.execute_input":"2022-08-10T06:49:38.905785Z","iopub.status.idle":"2022-08-10T06:49:38.913386Z","shell.execute_reply.started":"2022-08-10T06:49:38.905517Z","shell.execute_reply":"2022-08-10T06:49:38.911258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Autotokenizer will create a tokenizer appropriate for the model\nfrom transformers import AutoModelForSequenceClassification, AutoTokenizer\ntokz = AutoTokenizer.from_pretrained(model_nm)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:38.915351Z","iopub.execute_input":"2022-08-10T06:49:38.916542Z","iopub.status.idle":"2022-08-10T06:49:44.449440Z","shell.execute_reply.started":"2022-08-10T06:49:38.916489Z","shell.execute_reply":"2022-08-10T06:49:44.447851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Examples of how the token is working\ntokz.tokenize(\"Good day everyone! Let's go\")","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:44.451296Z","iopub.execute_input":"2022-08-10T06:49:44.451736Z","iopub.status.idle":"2022-08-10T06:49:44.464025Z","shell.execute_reply.started":"2022-08-10T06:49:44.451694Z","shell.execute_reply":"2022-08-10T06:49:44.462461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Observations:\n1. Tokens begin with an underscore\n2. Unique words are divided into parts","metadata":{}},{"cell_type":"code","source":"def tok_func(x): return tokz(x['input'])","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:44.466844Z","iopub.execute_input":"2022-08-10T06:49:44.467338Z","iopub.status.idle":"2022-08-10T06:49:44.478536Z","shell.execute_reply.started":"2022-08-10T06:49:44.467275Z","shell.execute_reply":"2022-08-10T06:49:44.476946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Running the token function on all of our datasets using the map function\ntok_ds = ds.map(tok_func, batched =True)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:44.481053Z","iopub.execute_input":"2022-08-10T06:49:44.481593Z","iopub.status.idle":"2022-08-10T06:49:50.058734Z","shell.execute_reply.started":"2022-08-10T06:49:44.481538Z","shell.execute_reply":"2022-08-10T06:49:50.057177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This adds a new row index_ids to our dataset. Let's look at the first index_id of our first text","metadata":{}},{"cell_type":"code","source":"row = tok_ds[0]\nrow['input'], row['input_ids']","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:50.067133Z","iopub.execute_input":"2022-08-10T06:49:50.068243Z","iopub.status.idle":"2022-08-10T06:49:50.079896Z","shell.execute_reply.started":"2022-08-10T06:49:50.068195Z","shell.execute_reply":"2022-08-10T06:49:50.078147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We have to rename the score column because the Transformers deals with labels column","metadata":{}},{"cell_type":"code","source":"tok_ds = tok_ds.rename_columns({'score':'labels'})","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:50.083321Z","iopub.execute_input":"2022-08-10T06:49:50.084719Z","iopub.status.idle":"2022-08-10T06:49:50.095793Z","shell.execute_reply.started":"2022-08-10T06:49:50.084675Z","shell.execute_reply":"2022-08-10T06:49:50.094305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def corr(x,y): return np.corrcoef(x,y)[0][1]","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:50.098447Z","iopub.execute_input":"2022-08-10T06:49:50.099380Z","iopub.status.idle":"2022-08-10T06:49:50.106083Z","shell.execute_reply.started":"2022-08-10T06:49:50.099336Z","shell.execute_reply":"2022-08-10T06:49:50.104480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Creating validation sets","metadata":{}},{"cell_type":"code","source":"test_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:50.108280Z","iopub.execute_input":"2022-08-10T06:49:50.109257Z","iopub.status.idle":"2022-08-10T06:49:50.138513Z","shell.execute_reply.started":"2022-08-10T06:49:50.109200Z","shell.execute_reply":"2022-08-10T06:49:50.137276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dds = tok_ds.train_test_split(0.25, seed=42)\ndds","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:50.140289Z","iopub.execute_input":"2022-08-10T06:49:50.141410Z","iopub.status.idle":"2022-08-10T06:49:50.166782Z","shell.execute_reply.started":"2022-08-10T06:49:50.141369Z","shell.execute_reply":"2022-08-10T06:49:50.165466Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Since the train test split has named the validation dataset as Test, we'll have to keep that in mind \ntest_df['input'] = 'TEXT1: '+ test_df.context +' ; TEXT2: '+ test_df.target + ' ;ANC1: '+ test_df.anchor\neval_ds = Dataset.from_pandas(test_df).map(tok_func, batched=True)\n#Naming the dataset eval to avoid confusion","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:50.168966Z","iopub.execute_input":"2022-08-10T06:49:50.169596Z","iopub.status.idle":"2022-08-10T06:49:50.241745Z","shell.execute_reply.started":"2022-08-10T06:49:50.169514Z","shell.execute_reply":"2022-08-10T06:49:50.240219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def corr(x,y): return np.corrcoef(x,y)[0][1]","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:50.243914Z","iopub.execute_input":"2022-08-10T06:49:50.244686Z","iopub.status.idle":"2022-08-10T06:49:50.252202Z","shell.execute_reply.started":"2022-08-10T06:49:50.244641Z","shell.execute_reply":"2022-08-10T06:49:50.250741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def corr_d(eval_pred): return {'pearson': corr(*eval_pred)}","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:50.254461Z","iopub.execute_input":"2022-08-10T06:49:50.255330Z","iopub.status.idle":"2022-08-10T06:49:50.265084Z","shell.execute_reply.started":"2022-08-10T06:49:50.255287Z","shell.execute_reply":"2022-08-10T06:49:50.263656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training the model\nWe'll be needing this to train our model","metadata":{}},{"cell_type":"code","source":"from transformers import TrainingArguments, Trainer","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:50.267260Z","iopub.execute_input":"2022-08-10T06:49:50.268052Z","iopub.status.idle":"2022-08-10T06:49:50.277241Z","shell.execute_reply.started":"2022-08-10T06:49:50.267988Z","shell.execute_reply":"2022-08-10T06:49:50.275660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bs = 63 #batch size\nepochs = 7 ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:50.279422Z","iopub.execute_input":"2022-08-10T06:49:50.280310Z","iopub.status.idle":"2022-08-10T06:49:50.290383Z","shell.execute_reply.started":"2022-08-10T06:49:50.280266Z","shell.execute_reply":"2022-08-10T06:49:50.288702Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr = 4e-6 #learning rate ","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:50.294081Z","iopub.execute_input":"2022-08-10T06:49:50.294997Z","iopub.status.idle":"2022-08-10T06:49:50.301737Z","shell.execute_reply.started":"2022-08-10T06:49:50.294952Z","shell.execute_reply":"2022-08-10T06:49:50.300166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"args = TrainingArguments('outputs',learning_rate = lr, warmup_ratio = 0.1, lr_scheduler_type='cosine',fp16 =True,\n                        evaluation_strategy = 'epoch',per_device_train_batch_size = bs, per_device_eval_batch_size = bs*2,\n                        num_train_epochs = epochs, weight_decay = 0.01, report_to = 'none')","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:50.304466Z","iopub.execute_input":"2022-08-10T06:49:50.305141Z","iopub.status.idle":"2022-08-10T06:49:50.318856Z","shell.execute_reply.started":"2022-08-10T06:49:50.305097Z","shell.execute_reply":"2022-08-10T06:49:50.317128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = AutoModelForSequenceClassification.from_pretrained(model_nm, num_labels=1)\ntrainer = Trainer(model, args, train_dataset = dds['train'],eval_dataset = dds['test'],\n                 tokenizer = tokz, compute_metrics = corr_d)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:50.321689Z","iopub.execute_input":"2022-08-10T06:49:50.322526Z","iopub.status.idle":"2022-08-10T06:49:53.152731Z","shell.execute_reply.started":"2022-08-10T06:49:50.322481Z","shell.execute_reply":"2022-08-10T06:49:53.151437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer.train();","metadata":{"execution":{"iopub.status.busy":"2022-08-10T06:49:53.154580Z","iopub.execute_input":"2022-08-10T06:49:53.158887Z","iopub.status.idle":"2022-08-10T07:00:30.487452Z","shell.execute_reply.started":"2022-08-10T06:49:53.158854Z","shell.execute_reply":"2022-08-10T07:00:30.485799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = trainer.predict(eval_ds).predictions.astype(float)\npreds","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:00:30.489880Z","iopub.execute_input":"2022-08-10T07:00:30.490379Z","iopub.status.idle":"2022-08-10T07:00:30.556535Z","shell.execute_reply.started":"2022-08-10T07:00:30.490336Z","shell.execute_reply":"2022-08-10T07:00:30.554783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Removing predictions less than 0 and greater than 1","metadata":{}},{"cell_type":"code","source":"preds = np.clip(preds, 0, 1)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:00:30.558810Z","iopub.execute_input":"2022-08-10T07:00:30.559565Z","iopub.status.idle":"2022-08-10T07:00:30.567387Z","shell.execute_reply.started":"2022-08-10T07:00:30.559521Z","shell.execute_reply":"2022-08-10T07:00:30.565055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:00:30.570285Z","iopub.execute_input":"2022-08-10T07:00:30.570929Z","iopub.status.idle":"2022-08-10T07:00:30.584976Z","shell.execute_reply.started":"2022-08-10T07:00:30.570884Z","shell.execute_reply":"2022-08-10T07:00:30.583181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import datasets\nsubmission = datasets.Dataset.from_dict({\n    'id': eval_ds['id'],\n    'score': preds\n})\nsubmission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-10T07:00:30.587899Z","iopub.execute_input":"2022-08-10T07:00:30.588959Z","iopub.status.idle":"2022-08-10T07:00:30.667504Z","shell.execute_reply.started":"2022-08-10T07:00:30.588897Z","shell.execute_reply":"2022-08-10T07:00:30.666058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}