{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T02:59:30.379280Z","iopub.execute_input":"2022-08-11T02:59:30.379644Z","iopub.status.idle":"2022-08-11T02:59:31.570605Z","shell.execute_reply.started":"2022-08-11T02:59:30.379613Z","shell.execute_reply":"2022-08-11T02:59:31.569582Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"based on: https://www.kaggle.com/code/inagana/baseline-with-huggingface-training-beginners \n\nadd some features to the text","metadata":{}},{"cell_type":"code","source":"import os\nimport re\nimport pandas as pd\nimport numpy as np\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import preprocessing\n\nfrom transformers import AutoTokenizer\nfrom datasets import Dataset\nfrom transformers import DataCollatorWithPadding\nfrom transformers import AutoModelForSequenceClassification, TrainingArguments, Trainer\n\nimport torch\nfrom torch.utils.checkpoint import checkpoint\nimport torch.nn as nn\n\n# You can change this if you want hugginface to automatically log to wandb\n# os.environ[\"WANDB_DISABLED\"] = \"true\"\n# os.environ[\"TOKENIZERS_PARALLELISM\"] = \"false\"\n\n# Suppress warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:00:04.248608Z","iopub.execute_input":"2022-08-11T03:00:04.248984Z","iopub.status.idle":"2022-08-11T03:00:12.718890Z","shell.execute_reply.started":"2022-08-11T03:00:04.248952Z","shell.execute_reply":"2022-08-11T03:00:12.717771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# You can change the model name here, look up model names from huggingface docs\n# model_name = \"microsoft/deberta-v3-base\"\n# Get tokenizer\ntokenizer = AutoTokenizer.from_pretrained(\"../input/fpe-token\")\ntokenizer.model_max_length = 512\ndef tokenize_function(examples):\n    return tokenizer(examples[\"text\"], padding=\"max_length\", truncation=True)\ndata_collator = DataCollatorWithPadding(tokenizer=tokenizer)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:00:16.123444Z","iopub.execute_input":"2022-08-11T03:00:16.124504Z","iopub.status.idle":"2022-08-11T03:00:16.511166Z","shell.execute_reply.started":"2022-08-11T03:00:16.124463Z","shell.execute_reply":"2022-08-11T03:00:16.510160Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"classes_to_labels = {\n    \"adequate\":0,\n    \"effective\":1,\n    \"ineffective\":2,\n}\ndescription = \"\"\"lead\ngrabs the reader’s attention\npoints toward the position\n\nposition\nstates a clear stance\nrelevant to the topic\n\nclaim\nrelevant to the position, valid and acceptable\nbacks up the position with specific points or perspectives\n\ncounterclaim\nreasonable\nrelevant\n\nrebuttal\nanswer and refute the counterclaim\nstrong and valid\n\nevidence\nrelevant to and support the claim, sound and well substantiated\nconcrete facts, examples, research, statistics, studies\n\nconcluding statement\nrestates the claims\ndifferent wording \n\"\"\"\ndescription = description.split(\"\\n\")\ndescription_dict = {}\nfor i in range(7):\n    temp_des = description[4*i+1]+tokenizer.sep_token+description[4*i+2]\n    description_dict[description[4*i]] = [temp_des,]\ndescription_df = pd.DataFrame.from_dict(description_dict,orient='index')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:00:18.267838Z","iopub.execute_input":"2022-08-11T03:00:18.268658Z","iopub.status.idle":"2022-08-11T03:00:18.280969Z","shell.execute_reply.started":"2022-08-11T03:00:18.268620Z","shell.execute_reply":"2022-08-11T03:00:18.279879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_essay(essay_id):\n    essay_path = os.path.join(\"../input/feedback-prize-effectiveness/\"+\"train\"+\"/\", f\"{essay_id}.txt\")\n    essay_text = open(essay_path, 'r').read()\n    return essay_text\ndef get_essay_test(essay_id):\n    essay_path = os.path.join(\"../input/feedback-prize-effectiveness/\"+\"test\"+\"/\", f\"{essay_id}.txt\")\n    essay_text = open(essay_path, 'r').read()\n    return essay_text\n# This function helps strip extra whitespaces from the text, \n# convert it all to lower case for uniformity, and remove end of line characters\ndef normalise(text):\n    text = text.lower()\n    text = text.strip()\n    text = re.sub(\"\\n\", \" \", text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:00:18.947622Z","iopub.execute_input":"2022-08-11T03:00:18.948013Z","iopub.status.idle":"2022-08-11T03:00:18.954565Z","shell.execute_reply.started":"2022-08-11T03:00:18.947979Z","shell.execute_reply":"2022-08-11T03:00:18.953473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_model = AutoModelForSequenceClassification.from_pretrained('../input/fpe-baseline-model')","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:00:19.594064Z","iopub.execute_input":"2022-08-11T03:00:19.594832Z","iopub.status.idle":"2022-08-11T03:00:29.359194Z","shell.execute_reply.started":"2022-08-11T03:00:19.594797Z","shell.execute_reply":"2022-08-11T03:00:29.358064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(\"../input/feedback-prize-effectiveness/test.csv\")\ndf_test['essay_text'] = df_test['essay_id'].apply(get_essay_test)\n\ndf_test['discourse_text'] = df_test['discourse_text'].apply(normalise)\ndf_test['discourse_type'] = df_test['discourse_type'].apply(normalise)\ndf_test['essay_text'] = df_test['essay_text'].apply(normalise)\n\ndf_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:00:31.211224Z","iopub.execute_input":"2022-08-11T03:00:31.211675Z","iopub.status.idle":"2022-08-11T03:00:31.310153Z","shell.execute_reply.started":"2022-08-11T03:00:31.211637Z","shell.execute_reply":"2022-08-11T03:00:31.309049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = df_test.merge(description_df,left_on = \"discourse_type\", right_index = True)\ndf_test.rename(columns={0:\"discourse_info\"}, inplace=True)\ndf_test = df_test.sort_index()\ndf_test['text'] = df_test['discourse_text']+tokenizer.sep_token+df_test['essay_text']+tokenizer.sep_token+df_test['discourse_info']\ndf_test.head(4)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:00:31.995176Z","iopub.execute_input":"2022-08-11T03:00:31.995638Z","iopub.status.idle":"2022-08-11T03:00:32.035092Z","shell.execute_reply.started":"2022-08-11T03:00:31.995596Z","shell.execute_reply":"2022-08-11T03:00:32.034091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test_temp = Dataset.from_pandas(df_test)\ntokenized_test_dataset = df_test_temp.map(tokenize_function, batched=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:00:33.205001Z","iopub.execute_input":"2022-08-11T03:00:33.205441Z","iopub.status.idle":"2022-08-11T03:00:33.369314Z","shell.execute_reply.started":"2022-08-11T03:00:33.205403Z","shell.execute_reply":"2022-08-11T03:00:33.368348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_args = TrainingArguments(\n    output_dir=\"./results\",\n    num_train_epochs=2,\n    evaluation_strategy=\"epoch\",\n    save_strategy=\"epoch\",\n    warmup_ratio=0.1, \n    lr_scheduler_type='cosine',\n    auto_find_batch_size=True,\n    dataloader_num_workers=2,\n    gradient_accumulation_steps=4,\n    fp16=True,\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:00:34.142833Z","iopub.execute_input":"2022-08-11T03:00:34.143473Z","iopub.status.idle":"2022-08-11T03:00:34.216534Z","shell.execute_reply.started":"2022-08-11T03:00:34.143438Z","shell.execute_reply":"2022-08-11T03:00:34.215461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from scipy.special import softmax\ntest_trainer = Trainer(\n    model=temp_model,\n    args=training_args,\n    tokenizer=tokenizer,\n    data_collator=data_collator)\npreds = test_trainer.predict(tokenized_test_dataset).predictions","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:00:35.430994Z","iopub.execute_input":"2022-08-11T03:00:35.431572Z","iopub.status.idle":"2022-08-11T03:00:41.937203Z","shell.execute_reply.started":"2022-08-11T03:00:35.431532Z","shell.execute_reply":"2022-08-11T03:00:41.936124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = softmax(preds,axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:00:41.939127Z","iopub.execute_input":"2022-08-11T03:00:41.939841Z","iopub.status.idle":"2022-08-11T03:00:41.945844Z","shell.execute_reply.started":"2022-08-11T03:00:41.939801Z","shell.execute_reply":"2022-08-11T03:00:41.944802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"temp_submission = {'discourse_id': df_test[\"discourse_id\"],'Adequate': preds[:,0], 'Effective': preds[:,1],'Ineffective': preds[:,2]}\ntemp_submission = pd.DataFrame(data=temp_submission)\ntemp_submission.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T03:00:41.947585Z","iopub.execute_input":"2022-08-11T03:00:41.948589Z","iopub.status.idle":"2022-08-11T03:00:41.960961Z","shell.execute_reply.started":"2022-08-11T03:00:41.948551Z","shell.execute_reply":"2022-08-11T03:00:41.960006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}