{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\nimport gc\ngc.collect()\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\ncnt = 0\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n#         print(os.path.join(dirname, filename))\n        cnt+=1\nprint(cnt)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-08T12:52:57.306830Z","iopub.execute_input":"2022-07-08T12:52:57.307357Z","iopub.status.idle":"2022-07-08T12:53:00.208391Z","shell.execute_reply.started":"2022-07-08T12:52:57.307268Z","shell.execute_reply":"2022-07-08T12:53:00.206773Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.environ[\"WANDB_DISABLED\"] = \"true\"","metadata":{"execution":{"iopub.status.busy":"2022-07-08T12:53:30.744229Z","iopub.execute_input":"2022-07-08T12:53:30.744596Z","iopub.status.idle":"2022-07-08T12:53:30.749160Z","shell.execute_reply.started":"2022-07-08T12:53:30.744564Z","shell.execute_reply":"2022-07-08T12:53:30.748097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df=pd.read_csv('../input/feedback-prize-effectiveness/train.csv')\ntrain_df.info()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T12:53:31.998632Z","iopub.execute_input":"2022-07-08T12:53:31.998981Z","iopub.status.idle":"2022-07-08T12:53:32.327435Z","shell.execute_reply.started":"2022-07-08T12:53:31.998951Z","shell.execute_reply":"2022-07-08T12:53:32.326614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T12:53:34.090955Z","iopub.execute_input":"2022-07-08T12:53:34.091697Z","iopub.status.idle":"2022-07-08T12:53:34.108620Z","shell.execute_reply.started":"2022-07-08T12:53:34.091658Z","shell.execute_reply":"2022-07-08T12:53:34.107557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df['discourse_effectiveness'].value_counts())\nprint('\\n\\n')\nprint(train_df['discourse_type'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-07-08T12:54:14.793750Z","iopub.execute_input":"2022-07-08T12:54:14.794295Z","iopub.status.idle":"2022-07-08T12:54:14.820927Z","shell.execute_reply.started":"2022-07-08T12:54:14.794249Z","shell.execute_reply":"2022-07-08T12:54:14.820207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### data pre processing","metadata":{}},{"cell_type":"code","source":"train_df['all_text'] = train_df['discourse_text']+' '+train_df['discourse_type']\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T12:57:16.699266Z","iopub.execute_input":"2022-07-08T12:57:16.699712Z","iopub.status.idle":"2022-07-08T12:57:16.758460Z","shell.execute_reply.started":"2022-07-08T12:57:16.699670Z","shell.execute_reply":"2022-07-08T12:57:16.754570Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp = train_df['all_text'].values","metadata":{"execution":{"iopub.status.busy":"2022-07-08T12:57:19.119747Z","iopub.execute_input":"2022-07-08T12:57:19.120295Z","iopub.status.idle":"2022-07-08T12:57:19.124364Z","shell.execute_reply.started":"2022-07-08T12:57:19.120247Z","shell.execute_reply":"2022-07-08T12:57:19.123609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp[:10]","metadata":{"execution":{"iopub.status.busy":"2022-07-08T12:57:19.525184Z","iopub.execute_input":"2022-07-08T12:57:19.525916Z","iopub.status.idle":"2022-07-08T12:57:19.532864Z","shell.execute_reply.started":"2022-07-08T12:57:19.525884Z","shell.execute_reply":"2022-07-08T12:57:19.532056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### To do -\n- lower case\n- punctuation\n- stop words\n- remove numbers\n- lemma ?","metadata":{}},{"cell_type":"code","source":"import re\nimport string\nfrom nltk.corpus import stopwords\n\ndef clean_text(text):\n    text = text.lower()\n    text = re.sub(r'\\W+\\s*',' ',text)\n    text = text.replace('_',' ')\n    text = re.sub(r'\\d+',' ',text)\n    text = re.sub(r'\\s\\s+',' ',text)\n    stops =  list(set(stopwords.words('english'))) \n    stops.extend(list(string.punctuation))\n    text = [word.strip() for word in text.split() if word.strip() not in stops]\n    \n    return ' '.join(text)\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-07-08T12:58:35.657787Z","iopub.execute_input":"2022-07-08T12:58:35.658183Z","iopub.status.idle":"2022-07-08T12:58:36.483478Z","shell.execute_reply.started":"2022-07-08T12:58:35.658134Z","shell.execute_reply":"2022-07-08T12:58:36.482669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(temp[6])\nprint('=============')\nclean_text(temp[6])","metadata":{"execution":{"iopub.status.busy":"2022-07-08T12:58:36.550758Z","iopub.execute_input":"2022-07-08T12:58:36.551047Z","iopub.status.idle":"2022-07-08T12:58:36.563750Z","shell.execute_reply.started":"2022-07-08T12:58:36.551020Z","shell.execute_reply":"2022-07-08T12:58:36.563036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['clean_text'] = train_df['all_text'].apply(lambda x: clean_text(x))\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T12:59:20.255352Z","iopub.execute_input":"2022-07-08T12:59:20.256138Z","iopub.status.idle":"2022-07-08T12:59:29.987687Z","shell.execute_reply.started":"2022-07-08T12:59:20.256102Z","shell.execute_reply":"2022-07-08T12:59:29.986930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop(['discourse_id','essay_id','discourse_text','discourse_type','all_text'],inplace=True,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:01:25.228905Z","iopub.execute_input":"2022-07-08T13:01:25.229368Z","iopub.status.idle":"2022-07-08T13:01:25.246942Z","shell.execute_reply.started":"2022-07-08T13:01:25.229335Z","shell.execute_reply":"2022-07-08T13:01:25.246208Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:01:25.527730Z","iopub.execute_input":"2022-07-08T13:01:25.528533Z","iopub.status.idle":"2022-07-08T13:01:25.537260Z","shell.execute_reply.started":"2022-07-08T13:01:25.528494Z","shell.execute_reply":"2022-07-08T13:01:25.536211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- prep target values","metadata":{}},{"cell_type":"code","source":"tar_map = {\"Ineffective\":0, \"Adequate\":1,\"Effective\":2}\ntrain_df[\"target\"] = train_df[\"discourse_effectiveness\"].map(tar_map)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:01:33.330809Z","iopub.execute_input":"2022-07-08T13:01:33.331195Z","iopub.status.idle":"2022-07-08T13:01:33.346763Z","shell.execute_reply.started":"2022-07-08T13:01:33.331141Z","shell.execute_reply":"2022-07-08T13:01:33.345875Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train_df['target'].value_counts())\nprint(tar_map)\nprint(train_df['discourse_effectiveness'].value_counts())","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:01:33.959532Z","iopub.execute_input":"2022-07-08T13:01:33.959890Z","iopub.status.idle":"2022-07-08T13:01:33.971687Z","shell.execute_reply.started":"2022-07-08T13:01:33.959859Z","shell.execute_reply":"2022-07-08T13:01:33.970539Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.drop(['discourse_effectiveness'],inplace=True,axis=1)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:01:43.027532Z","iopub.execute_input":"2022-07-08T13:01:43.028058Z","iopub.status.idle":"2022-07-08T13:01:43.039076Z","shell.execute_reply.started":"2022-07-08T13:01:43.028011Z","shell.execute_reply":"2022-07-08T13:01:43.038224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\nx_train,x_test,y_train_,y_test_ = train_test_split(train_df['clean_text'],train_df['target'],random_state=1234,test_size=0.25)\n\nprint(x_train.shape,y_train_.shape)\nprint(x_test.shape,y_test_.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:02:56.326248Z","iopub.execute_input":"2022-07-08T13:02:56.326675Z","iopub.status.idle":"2022-07-08T13:02:56.344172Z","shell.execute_reply.started":"2022-07-08T13:02:56.326634Z","shell.execute_reply":"2022-07-08T13:02:56.343358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_train = pd.get_dummies(y_train_).values\ny_test = pd.get_dummies(y_test_).values\n\nprint(x_train.shape,y_train_.shape)\nprint(x_test.shape,y_test_.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:02:56.543337Z","iopub.execute_input":"2022-07-08T13:02:56.543632Z","iopub.status.idle":"2022-07-08T13:02:56.553298Z","shell.execute_reply.started":"2022-07-08T13:02:56.543607Z","shell.execute_reply":"2022-07-08T13:02:56.552472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x_train.values","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:02:56.623477Z","iopub.execute_input":"2022-07-08T13:02:56.623819Z","iopub.status.idle":"2022-07-08T13:02:56.627680Z","shell.execute_reply.started":"2022-07-08T13:02:56.623788Z","shell.execute_reply":"2022-07-08T13:02:56.626656Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Model prep","metadata":{}},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport transformers\nfrom transformers import EarlyStoppingCallback\nfrom transformers import Trainer,TrainingArguments\n\nfrom transformers import BertTokenizer, BertForSequenceClassification\nfrom torch.nn.utils.clip_grad import clip_grad_norm\n\n\ntorch.cuda.is_available()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:02:58.742398Z","iopub.execute_input":"2022-07-08T13:02:58.742743Z","iopub.status.idle":"2022-07-08T13:03:05.732876Z","shell.execute_reply.started":"2022-07-08T13:02:58.742712Z","shell.execute_reply":"2022-07-08T13:03:05.732055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:03:05.734619Z","iopub.execute_input":"2022-07-08T13:03:05.735326Z","iopub.status.idle":"2022-07-08T13:03:05.739836Z","shell.execute_reply.started":"2022-07-08T13:03:05.735286Z","shell.execute_reply":"2022-07-08T13:03:05.739133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_path = '../input/huggingface-bert/bert-large-uncased'\nmax_len = 256","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:03:28.172936Z","iopub.execute_input":"2022-07-08T13:03:28.173322Z","iopub.status.idle":"2022-07-08T13:03:28.176846Z","shell.execute_reply.started":"2022-07-08T13:03:28.173287Z","shell.execute_reply":"2022-07-08T13:03:28.176140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class my_model(nn.Module):\n#     def __init__(self,model_path):\n#         super(my_model,self).__init__()\n#         self.model_path = model_path\n#         self.model = transformers.AutoModel.from_pretrained(self.model_path)\n#         self.fc_layer = nn.Sequential(\n#             nn.Dropout(0.3),\n#             nn.Linear(self.model.get_word_embedding_dimension(),256,bias=True),\n#             nn.LeakyReLU(),\n#             nn.Linear(256,3),\n#             nn.Sigmoid()\n#         )\n#     def forward(self,ids,mask,token_type_ids):\n#         out = self.bert(input_ids=ids,attention_mask=mask,token_type_ids=token_type_ids,return_dict=True)\n#         pooler_output = out.get('pooler_output')\n#         bo = self.fc_layer(pooler_output)\n#         return bo\n     ","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:03:29.443789Z","iopub.execute_input":"2022-07-08T13:03:29.444139Z","iopub.status.idle":"2022-07-08T13:03:29.448745Z","shell.execute_reply.started":"2022-07-08T13:03:29.444109Z","shell.execute_reply":"2022-07-08T13:03:29.447978Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# class my_dataset_train:\n#     def __init__(self,text,label,tokenizer,max_len):\n#         self.text=text\n#         self.label=label\n#         self.tokenizer = tokenizer\n#         self.max_len = max_len\n        \n#     def __len__(self):\n#         return len(self.text)\n    \n#     def __getitem__(self,idx):\n#         text_ = str(self.text[idx])\n#         label_ = self.label[idx]\n        \n#         inputs = self.tokenizer(\n#             text_,\n#             add_special_tokens=True,\n#             padding='max_length',\n#             truncation=True,\n#             max_length=self.max_len,\n#             return_attention_mask=True\n#         )\n        \n#         ids = inputs['input_ids']\n#         token_type_ids = inputs[\"token_type_ids\"]\n#         mask = inputs['attention_mask']\n        \n#         return {\n#             \"ids\": torch.tensor(ids,dtype=torch.long),\n#             \"mask\": torch.tensor(mask,dtype=torch.long),\n#             \"token_type_ids\": torch.tensor(token_type_ids,dtype=torch.long),\n#             \"targets\": torch.tensor(label_,dtype=torch.float),\n#         }","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:03:30.130683Z","iopub.execute_input":"2022-07-08T13:03:30.131220Z","iopub.status.idle":"2022-07-08T13:03:30.136296Z","shell.execute_reply.started":"2022-07-08T13:03:30.131183Z","shell.execute_reply":"2022-07-08T13:03:30.135218Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tokenizer = transformers.AutoTokenizer.from_pretrained(bert_path)\n\n# # Training dataset prep\n\n# train_text = list(x_train.values)\n\n# train_dataset = my_dataset_train(\n#     text = train_text,\n#     label = y_train,\n#     tokenizer=tokenizer ,\n#     max_len=max_len\n# )\n\n# train_data_loader = torch.utils.data.DataLoader(train_dataset,batch_size=train_batch_size,shuffle=True)\n\n# # validation dataset prep\n# val_text1 = list(x_test['target'].values)\n# val_text2 = list(x_test['sen1'].values)\n# val_label = list(y_test.values)\n\n# valid_dataset = my_dataset_train(\n#     text1 = val_text1,\n#     text2 = val_text2,\n#     label = val_label,\n#     tokenizer=tokenizer,\n#     max_len=max_len\n# )\n\n# valid_data_loader = torch.utils.data.DataLoader(valid_dataset,batch_size=train_batch_size,shuffle=True)\n\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:03:30.534748Z","iopub.execute_input":"2022-07-08T13:03:30.535469Z","iopub.status.idle":"2022-07-08T13:03:30.540126Z","shell.execute_reply.started":"2022-07-08T13:03:30.535432Z","shell.execute_reply":"2022-07-08T13:03:30.539305Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create torch dataset\nclass Dataset(torch.utils.data.Dataset):\n    def __init__(self, encodings, labels=None):\n        self.encodings = encodings\n        self.labels = labels\n\n    def __getitem__(self, idx):\n        item = {key: torch.tensor(val[idx]) for key, val in self.encodings.items()}\n#         print(self.labels,type(self.labels))\n        if self.labels:\n            item[\"labels\"] = torch.tensor(self.labels[idx])\n        return item\n\n    def __len__(self):\n        return len(self.encodings[\"input_ids\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:03:30.941332Z","iopub.execute_input":"2022-07-08T13:03:30.941832Z","iopub.status.idle":"2022-07-08T13:03:30.948616Z","shell.execute_reply.started":"2022-07-08T13:03:30.941798Z","shell.execute_reply":"2022-07-08T13:03:30.947567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = BertTokenizer.from_pretrained(model_path)\n\nX_train_tokenized = tokenizer(list(x_train.values),add_special_tokens=True,padding='max_length',truncation=True,max_length=max_len,return_attention_mask=True)\nX_val_tokenized = tokenizer(list(x_test.values),add_special_tokens=True,padding='max_length',truncation=True,max_length=max_len,return_attention_mask=True)\n\ntrain_dataset = Dataset(X_train_tokenized, list(y_train_))\nval_dataset = Dataset(X_val_tokenized, list(y_test_))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:03:31.908516Z","iopub.execute_input":"2022-07-08T13:03:31.909384Z","iopub.status.idle":"2022-07-08T13:04:09.662350Z","shell.execute_reply.started":"2022-07-08T13:03:31.909345Z","shell.execute_reply":"2022-07-08T13:04:09.661500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import classification_report\ndef compute_metrics(p):\n    pred, labels = p\n    pred = np.argmax(pred, axis=1)\n    \n    result = classification_report(labels,pred,output_dict=True)\n\n    return result\n","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:04:09.663829Z","iopub.execute_input":"2022-07-08T13:04:09.664181Z","iopub.status.idle":"2022-07-08T13:04:09.669600Z","shell.execute_reply.started":"2022-07-08T13:04:09.664130Z","shell.execute_reply":"2022-07-08T13:04:09.668494Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = BertForSequenceClassification.from_pretrained(model_path, num_labels=3)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:04:11.747101Z","iopub.execute_input":"2022-07-08T13:04:11.747986Z","iopub.status.idle":"2022-07-08T13:04:27.158927Z","shell.execute_reply.started":"2022-07-08T13:04:11.747941Z","shell.execute_reply":"2022-07-08T13:04:27.158195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define Trainer\nargs = TrainingArguments(\n    output_dir=\"./output/\",\n    evaluation_strategy=\"epoch\",\n    per_device_train_batch_size=8,\n    per_device_eval_batch_size=32,\n    num_train_epochs=6,\n    save_strategy=\"epoch\",\n    seed=0,\n    load_best_model_at_end=True,\n)\ntrainer = Trainer(\n    model=model,\n    args=args,\n    train_dataset=train_dataset,\n    eval_dataset=val_dataset,\n    compute_metrics=compute_metrics,\n    callbacks=[EarlyStoppingCallback(early_stopping_patience=3)],\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:04:27.160697Z","iopub.execute_input":"2022-07-08T13:04:27.161210Z","iopub.status.idle":"2022-07-08T13:04:32.234104Z","shell.execute_reply.started":"2022-07-08T13:04:27.161132Z","shell.execute_reply":"2022-07-08T13:04:32.233300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer.train()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T13:04:44.895532Z","iopub.execute_input":"2022-07-08T13:04:44.895892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Inference","metadata":{}},{"cell_type":"code","source":"test_data = pd.read_csv('../input/feedback-prize-effectiveness/test.csv')\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T09:43:04.943759Z","iopub.execute_input":"2022-06-16T09:43:04.94431Z","iopub.status.idle":"2022-06-16T09:43:04.964141Z","shell.execute_reply.started":"2022-06-16T09:43:04.944277Z","shell.execute_reply":"2022-06-16T09:43:04.963387Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data['all_text'] = test_data['discourse_text']+' '+test_data['discourse_type']\ntest_data['clean_text'] = test_data['all_text'].apply(lambda x: clean_text(x))\ntest_data.drop(['essay_id','discourse_text','discourse_type','all_text'],inplace=True,axis=1)\ntest_data.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T09:43:06.383413Z","iopub.execute_input":"2022-06-16T09:43:06.384039Z","iopub.status.idle":"2022-06-16T09:43:06.403396Z","shell.execute_reply.started":"2022-06-16T09:43:06.384001Z","shell.execute_reply":"2022-06-16T09:43:06.402441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_x = list(test_data[\"clean_text\"])\ntest_x_tokenized = tokenizer(list(test_data[\"clean_text\"]),add_special_tokens=True,padding='max_length',truncation=True,max_length=max_len,return_attention_mask=True)\n\ntest_dataset_ = Dataset(test_x_tokenized)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-16T09:44:28.432497Z","iopub.execute_input":"2022-06-16T09:44:28.433219Z","iopub.status.idle":"2022-06-16T09:44:28.448674Z","shell.execute_reply.started":"2022-06-16T09:44:28.433184Z","shell.execute_reply":"2022-06-16T09:44:28.44795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Load trained model\n# model_path_trained = \"output/checkpoint-50000\"\n# model = BertForSequenceClassification.from_pretrained(model_path_trained, num_labels=3)\n\n# test_trainer = Trainer(model)\n\n# Make prediction\nraw_pred, _, _ = trainer.predict(test_dataset_)\n","metadata":{"execution":{"iopub.status.busy":"2022-06-16T09:45:01.193167Z","iopub.execute_input":"2022-06-16T09:45:01.193519Z","iopub.status.idle":"2022-06-16T09:45:01.255838Z","shell.execute_reply.started":"2022-06-16T09:45:01.193491Z","shell.execute_reply":"2022-06-16T09:45:01.255039Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_pred.shape, test_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-16T09:45:28.722988Z","iopub.execute_input":"2022-06-16T09:45:28.723345Z","iopub.status.idle":"2022-06-16T09:45:28.731402Z","shell.execute_reply.started":"2022-06-16T09:45:28.723316Z","shell.execute_reply":"2022-06-16T09:45:28.728372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('../input/feedback-prize-effectiveness/sample_submission.csv')\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T09:46:19.410974Z","iopub.execute_input":"2022-06-16T09:46:19.411387Z","iopub.status.idle":"2022-06-16T09:46:19.430312Z","shell.execute_reply.started":"2022-06-16T09:46:19.411354Z","shell.execute_reply":"2022-06-16T09:46:19.429521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub.shape","metadata":{"execution":{"iopub.status.busy":"2022-06-16T09:47:33.70438Z","iopub.execute_input":"2022-06-16T09:47:33.704742Z","iopub.status.idle":"2022-06-16T09:47:33.70982Z","shell.execute_reply.started":"2022-06-16T09:47:33.704714Z","shell.execute_reply":"2022-06-16T09:47:33.709088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_pred","metadata":{"execution":{"iopub.status.busy":"2022-06-16T09:51:38.675013Z","iopub.execute_input":"2022-06-16T09:51:38.675378Z","iopub.status.idle":"2022-06-16T09:51:38.681369Z","shell.execute_reply.started":"2022-06-16T09:51:38.675348Z","shell.execute_reply":"2022-06-16T09:51:38.680592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x=torch.sigmoid(torch.tensor(raw_pred))\nx","metadata":{"execution":{"iopub.status.busy":"2022-06-16T09:56:23.161716Z","iopub.execute_input":"2022-06-16T09:56:23.162343Z","iopub.status.idle":"2022-06-16T09:56:23.170031Z","shell.execute_reply.started":"2022-06-16T09:56:23.16231Z","shell.execute_reply":"2022-06-16T09:56:23.169257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final = []\nids = list(test_data['discourse_id'].values)\nfor i in range(len(x)):\n    t = [ids[i]] + x[i].tolist()\n    final.append(t)\ndf_ = pd.DataFrame(final,columns=list(sub.columns))\ndf_.head()","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:03:15.396699Z","iopub.execute_input":"2022-06-16T10:03:15.397046Z","iopub.status.idle":"2022-06-16T10:03:15.410152Z","shell.execute_reply.started":"2022-06-16T10:03:15.397017Z","shell.execute_reply":"2022-06-16T10:03:15.409419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-06-16T10:04:29.466136Z","iopub.execute_input":"2022-06-16T10:04:29.466705Z","iopub.status.idle":"2022-06-16T10:04:29.475864Z","shell.execute_reply.started":"2022-06-16T10:04:29.466672Z","shell.execute_reply":"2022-06-16T10:04:29.474924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}