{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Baseline with HuggingFace + Submission for Beginners \n\nsource: https://www.kaggle.com/code/inagana/baseline-with-huggingface-training-beginners/notebook","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import preprocessing\n\nfrom transformers import AutoTokenizer\nfrom datasets import Dataset\nfrom transformers import DataCollatorWithPadding\nfrom transformers import AutoModelForSequenceClassification, TrainingArguments, Trainer\n\nimport torch\nfrom torch.utils.checkpoint import checkpoint\nimport torch.nn as nn\n\n# You can change this if you want hugginface to automatically log to wandb\n# os.environ[\"WANDB_DISABLED\"] = \"true\"\n# os.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"\n\n# Suppress warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T02:55:09.515397Z","iopub.execute_input":"2022-08-11T02:55:09.516224Z","iopub.status.idle":"2022-08-11T02:55:18.952671Z","shell.execute_reply.started":"2022-08-11T02:55:09.515725Z","shell.execute_reply":"2022-08-11T02:55:18.951331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# You can change the model name here, look up model names from huggingface docs\nmodel_name = \"microsoft/deberta-v3-base\"\n# Get tokenizer\ntokenizer = AutoTokenizer.from_pretrained(model_name)\ntokenizer.model_max_length = 512\ndef tokenize_function(examples):\n    return tokenizer(examples[\"text\"], padding=\"max_length\", truncation=True)\ndata_collator = DataCollatorWithPadding(tokenizer=tokenizer)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T02:55:18.958780Z","iopub.execute_input":"2022-08-11T02:55:18.961887Z","iopub.status.idle":"2022-08-11T02:55:24.551033Z","shell.execute_reply.started":"2022-08-11T02:55:18.961843Z","shell.execute_reply":"2022-08-11T02:55:24.550020Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer.save_pretrained('/kaggle/working/token')\nimport shutil\nimport os\nshutil.make_archive('token', 'zip', '/kaggle/working/token')\nfrom IPython.display import FileLink\nFileLink(r'./token.zip')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T02:57:26.419572Z","iopub.execute_input":"2022-08-11T02:57:26.420292Z","iopub.status.idle":"2022-08-11T02:57:27.189399Z","shell.execute_reply.started":"2022-08-11T02:57:26.420247Z","shell.execute_reply":"2022-08-11T02:57:27.188327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# adding more information\n\nBased on the information provided about different discourse types:\n\nhttps://www.kaggle.com/competitions/feedback-prize-effectiveness/data\n\n\nhttps://docs.google.com/document/d/1G51Ulb0i-nKCRQSs4p4ujauy4wjAJOae/edit\n\nWe extract some keywords/key phrases for each type (in the following variable ''description'') and use them as extra features. To be precise, in the original model, only the discourse type is appended to each text, but here we append not only the type but also the keywords/key phrases. ","metadata":{}},{"cell_type":"code","source":"classes_to_labels = {\n    \"adequate\":0,\n    \"effective\":1,\n    \"ineffective\":2,\n}\ndescription = \"\"\"lead\ngrabs the reader’s attention\npoints toward the position\n\nposition\nstates a clear stance\nrelevant to the topic\n\nclaim\nrelevant to the position, valid and acceptable\nbacks up the position with specific points or perspectives\n\ncounterclaim\nreasonable\nrelevant\n\nrebuttal\nanswer and refute the counterclaim\nstrong and valid\n\nevidence\nrelevant to and support the claim, sound and well substantiated\nconcrete facts, examples, research, statistics, studies\n\nconcluding statement\nrestates the claims\ndifferent wording \n\"\"\"\ndescription = description.split(\"\\n\")\ndescription_dict = {}\nfor i in range(7):\n    temp_des = description[4*i+1]+tokenizer.sep_token+description[4*i+2]\n    description_dict[description[4*i]] = [temp_des,]\ndescription_df = pd.DataFrame.from_dict(description_dict,orient='index')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T00:37:28.846446Z","iopub.execute_input":"2022-08-11T00:37:28.847027Z","iopub.status.idle":"2022-08-11T00:37:28.858219Z","shell.execute_reply.started":"2022-08-11T00:37:28.846990Z","shell.execute_reply":"2022-08-11T00:37:28.857258Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_essay(essay_id):\n    essay_path = os.path.join(\"../input/feedback-prize-effectiveness/\"+\"train\"+\"/\", f\"{essay_id}.txt\")\n    essay_text = open(essay_path, 'r').read()\n    return essay_text\ndef get_essay_test(essay_id):\n    essay_path = os.path.join(\"../input/feedback-prize-effectiveness/\"+\"test\"+\"/\", f\"{essay_id}.txt\")\n    essay_text = open(essay_path, 'r').read()\n    return essay_text\n# This function helps strip extra whitespaces from the text, \n# convert it all to lower case for uniformity, and remove end of line characters\ndef normalise(text):\n    text = text.lower()\n    text = text.strip()\n    text = re.sub(\"\\n\", \" \", text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-08-11T00:37:31.658093Z","iopub.execute_input":"2022-08-11T00:37:31.658803Z","iopub.status.idle":"2022-08-11T00:37:31.665365Z","shell.execute_reply.started":"2022-08-11T00:37:31.658765Z","shell.execute_reply":"2022-08-11T00:37:31.664301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Train","metadata":{}},{"cell_type":"code","source":"import wandb\nprint('Authenticating with wandb.')\nfrom kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nwandb_creds = user_secrets.get_secret(\"wandb\")\n\n!wandb login {wandb_creds}","metadata":{"execution":{"iopub.status.busy":"2022-08-11T00:37:34.228621Z","iopub.execute_input":"2022-08-11T00:37:34.229203Z","iopub.status.idle":"2022-08-11T00:37:37.112632Z","shell.execute_reply.started":"2022-08-11T00:37:34.229166Z","shell.execute_reply":"2022-08-11T00:37:37.111486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")\ndf['essay_text'] = df['essay_id'].apply(get_essay)\ndf['discourse_text'] = df['discourse_text'].apply(normalise)\ndf['discourse_type'] = df['discourse_type'].apply(normalise)\ndf['essay_text'] = df['essay_text'].apply(normalise)\ndf['discourse_effectiveness'] = df['discourse_effectiveness'].apply(normalise)\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T00:37:39.835332Z","iopub.execute_input":"2022-08-11T00:37:39.836143Z","iopub.status.idle":"2022-08-11T00:38:09.676585Z","shell.execute_reply.started":"2022-08-11T00:37:39.836099Z","shell.execute_reply":"2022-08-11T00:38:09.675531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = df.merge(description_df,left_on = \"discourse_type\", right_index = True)\ndf.rename(columns={0:\"discourse_info\"}, inplace=True)\ndf = df.sort_index()\ndf['text'] = df['discourse_text']+tokenizer.sep_token+df['essay_text']+tokenizer.sep_token+df['discourse_info']\ndf['labels'] = df['discourse_effectiveness'].replace(classes_to_labels)\ntrain_df, valid_df = train_test_split(df, test_size=0.2, random_state=42, stratify=df[\"labels\"])\ntrain_dataset = Dataset.from_pandas(train_df)\nvalid_dataset = Dataset.from_pandas(valid_df)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T00:38:12.074812Z","iopub.execute_input":"2022-08-11T00:38:12.075829Z","iopub.status.idle":"2022-08-11T00:38:12.849256Z","shell.execute_reply.started":"2022-08-11T00:38:12.075781Z","shell.execute_reply":"2022-08-11T00:38:12.848343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenized_train_dataset = train_dataset.shuffle(seed=42).map(tokenize_function, batched=True)\ntokenized_test_dataset = valid_dataset.shuffle(seed=42).map(tokenize_function, batched=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T00:38:12.851911Z","iopub.execute_input":"2022-08-11T00:38:12.852662Z","iopub.status.idle":"2022-08-11T00:39:17.963281Z","shell.execute_reply.started":"2022-08-11T00:38:12.852616Z","shell.execute_reply":"2022-08-11T00:39:17.962179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = AutoModelForSequenceClassification.from_pretrained(model_name, num_labels=3)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T00:39:17.965554Z","iopub.execute_input":"2022-08-11T00:39:17.965936Z","iopub.status.idle":"2022-08-11T00:39:41.263486Z","shell.execute_reply.started":"2022-08-11T00:39:17.965898Z","shell.execute_reply":"2022-08-11T00:39:41.262586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_args = TrainingArguments(\n    output_dir=\"./results\",\n    num_train_epochs=2,\n    evaluation_strategy=\"epoch\",\n    save_strategy=\"epoch\",\n    warmup_ratio=0.1, \n    lr_scheduler_type='cosine',\n    # Optimising\n    auto_find_batch_size=True,\n    # The num of workers may vary for different machines, if you are not sure, just comment this line out\n    dataloader_num_workers=2,\n    gradient_accumulation_steps=4,\n    fp16=True,\n    report_to=\"wandb\"\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T00:39:41.265089Z","iopub.execute_input":"2022-08-11T00:39:41.265476Z","iopub.status.idle":"2022-08-11T00:39:41.326926Z","shell.execute_reply.started":"2022-08-11T00:39:41.265438Z","shell.execute_reply":"2022-08-11T00:39:41.325970Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"w_adequate = 1-len(df[df['discourse_effectiveness'] == 'adequate'])/len(train_df)\nw_effective = 1-len(df[df['discourse_effectiveness'] == 'effective'])/len(train_df)\nw_ineffective = 1-len(df[df['discourse_effectiveness'] == 'ineffective'])/len(train_df)\nclass_weights = torch.tensor(\n    [w_adequate, w_effective, w_ineffective]\n).cuda()\n\nclass_weights","metadata":{"execution":{"iopub.status.busy":"2022-08-11T00:39:41.328840Z","iopub.execute_input":"2022-08-11T00:39:41.329175Z","iopub.status.idle":"2022-08-11T00:39:45.922342Z","shell.execute_reply.started":"2022-08-11T00:39:41.329136Z","shell.execute_reply":"2022-08-11T00:39:45.921219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# huggingface has no straightforward way to incorparate class_weights as far as I know, \n# Hence we override the compute_loss function of the Trainer and introduce our class weighgts\nclass CustomTrainer(Trainer):\n    def compute_loss(self, model, inputs, return_outputs=False):\n        labels = inputs.get(\"labels\")\n        # forward pass\n        outputs = model(**inputs)\n        logits = outputs.get('logits')\n        # compute custom loss\n        # Class weighting\n        loss_fct = nn.CrossEntropyLoss(weight=class_weights)\n        loss = loss_fct(logits.view(-1, self.model.config.num_labels), labels.view(-1))\n        return (loss, outputs) if return_outputs else loss","metadata":{"execution":{"iopub.status.busy":"2022-08-11T00:39:49.271580Z","iopub.execute_input":"2022-08-11T00:39:49.272153Z","iopub.status.idle":"2022-08-11T00:39:49.278804Z","shell.execute_reply.started":"2022-08-11T00:39:49.272116Z","shell.execute_reply":"2022-08-11T00:39:49.277793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wandb.init(project=\"Junhui_experiments\", entity=\"team-2\", name = \"baseline_1\")\ntrainer = CustomTrainer(\n    model=model,\n    args=training_args,\n    train_dataset=tokenized_train_dataset,\n    eval_dataset=tokenized_test_dataset,\n    tokenizer=tokenizer,\n    data_collator=data_collator,\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T00:39:51.976357Z","iopub.execute_input":"2022-08-11T00:39:51.977073Z","iopub.status.idle":"2022-08-11T00:39:59.092671Z","shell.execute_reply.started":"2022-08-11T00:39:51.977030Z","shell.execute_reply":"2022-08-11T00:39:59.091723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer.train()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T00:40:03.881583Z","iopub.execute_input":"2022-08-11T00:40:03.881966Z","iopub.status.idle":"2022-08-11T02:10:36.109983Z","shell.execute_reply.started":"2022-08-11T00:40:03.881929Z","shell.execute_reply":"2022-08-11T02:10:36.108889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer.save_model('/kaggle/working/model')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T02:11:30.973921Z","iopub.execute_input":"2022-08-11T02:11:30.974333Z","iopub.status.idle":"2022-08-11T02:11:32.851264Z","shell.execute_reply.started":"2022-08-11T02:11:30.974290Z","shell.execute_reply":"2022-08-11T02:11:32.850269Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nimport os\nshutil.make_archive('model', 'zip', '/kaggle/working/model')\nfrom IPython.display import FileLink\nFileLink(r'./model.zip')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T02:11:36.546564Z","iopub.execute_input":"2022-08-11T02:11:36.546915Z","iopub.status.idle":"2022-08-11T02:13:18.895938Z","shell.execute_reply.started":"2022-08-11T02:11:36.546883Z","shell.execute_reply":"2022-08-11T02:13:18.895003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Test","metadata":{}},{"cell_type":"code","source":"import shutil\nimport os\nshutil.unpack_archive('./input/model.zip', './input', \"zip\")\ntemp_model = AutoModelForSequenceClassification.from_pretrained('./input/model.zip')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T00:02:51.668408Z","iopub.execute_input":"2022-08-11T00:02:51.668791Z","iopub.status.idle":"2022-08-11T00:02:54.226051Z","shell.execute_reply.started":"2022-08-11T00:02:51.668755Z","shell.execute_reply":"2022-08-11T00:02:54.225096Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(\"../input/feedback-prize-effectiveness/test.csv\")\ndf_test['essay_text'] = df_test['essay_id'].apply(get_essay_test)\n\ndf_test['discourse_text'] = df_test['discourse_text'].apply(normalise)\ndf_test['discourse_type'] = df_test['discourse_type'].apply(normalise)\ndf_test['essay_text'] = df_test['essay_text'].apply(normalise)\n\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T02:16:36.223149Z","iopub.execute_input":"2022-08-11T02:16:36.224005Z","iopub.status.idle":"2022-08-11T02:16:36.275093Z","shell.execute_reply.started":"2022-08-11T02:16:36.223968Z","shell.execute_reply":"2022-08-11T02:16:36.274201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test.merge(description_df,left_on = \"discourse_type\", right_index = True)\ndf_test.rename(columns={0:\"discourse_info\"}, inplace=True)\ndf_test = df_test.sort_index()\ndf_test['text'] = df_test['discourse_text']+tokenizer.sep_token+df_test['essay_text']+tokenizer.sep_token+df_test['discourse_info']\ndf_test.head(4)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T02:16:40.013046Z","iopub.execute_input":"2022-08-11T02:16:40.013427Z","iopub.status.idle":"2022-08-11T02:16:40.039758Z","shell.execute_reply.started":"2022-08-11T02:16:40.013393Z","shell.execute_reply":"2022-08-11T02:16:40.038891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_temp = Dataset.from_pandas(df_test)\ntokenized_test_dataset = df_test_temp.map(tokenize_function, batched=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T02:16:46.455005Z","iopub.execute_input":"2022-08-11T02:16:46.455478Z","iopub.status.idle":"2022-08-11T02:16:46.587827Z","shell.execute_reply.started":"2022-08-11T02:16:46.455439Z","shell.execute_reply":"2022-08-11T02:16:46.586986Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.special import softmax\ntemp_model = model\ntest_trainer = Trainer(\n    model=temp_model,\n    tokenizer=tokenizer,\n    data_collator=data_collator)\npreds = softmax(test_trainer.predict(tokenized_test_dataset).predictions)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T02:16:58.735461Z","iopub.execute_input":"2022-08-11T02:16:58.735813Z","iopub.status.idle":"2022-08-11T02:16:59.097057Z","shell.execute_reply.started":"2022-08-11T02:16:58.735783Z","shell.execute_reply":"2022-08-11T02:16:59.096195Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_submission = {'discourse_id': df_test[\"discourse_id\"],'Adequate': preds[:,0], 'Effective': preds[:,1],'Ineffective': preds[:,2]}\ntemp_submission = pd.DataFrame(data=temp_submission)\ntemp_submission.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T02:17:03.686323Z","iopub.execute_input":"2022-08-11T02:17:03.686684Z","iopub.status.idle":"2022-08-11T02:17:03.702233Z","shell.execute_reply.started":"2022-08-11T02:17:03.686654Z","shell.execute_reply":"2022-08-11T02:17:03.701189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}