{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n    \npath = '../input/feedback-prize-effectiveness/'\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session\n\n# '../input/feedback-prize-effectiveness/'\n","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2022-08-08T08:26:41.033196Z","iopub.execute_input":"2022-08-08T08:26:41.034330Z","iopub.status.idle":"2022-08-08T08:26:41.046222Z","shell.execute_reply.started":"2022-08-08T08:26:41.034239Z","shell.execute_reply":"2022-08-08T08:26:41.045246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%config Completer.use_jedi = False","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:26:41.047950Z","iopub.execute_input":"2022-08-08T08:26:41.048663Z","iopub.status.idle":"2022-08-08T08:26:41.071654Z","shell.execute_reply.started":"2022-08-08T08:26:41.048626Z","shell.execute_reply":"2022-08-08T08:26:41.070552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfile_list = os.listdir('../input/d/jellynexus/diltbert')\nprint(file_list)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:26:41.074814Z","iopub.execute_input":"2022-08-08T08:26:41.075208Z","iopub.status.idle":"2022-08-08T08:26:41.083130Z","shell.execute_reply.started":"2022-08-08T08:26:41.075167Z","shell.execute_reply":"2022-08-08T08:26:41.082082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Get data from server","metadata":{}},{"cell_type":"code","source":"train_data = pd.read_csv(path + 'train.csv')\ntest_data = pd.read_csv(path + 'test.csv')\n\ndisplay(train_data.head())\ntrain_data.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:26:41.087102Z","iopub.execute_input":"2022-08-08T08:26:41.087623Z","iopub.status.idle":"2022-08-08T08:26:41.257661Z","shell.execute_reply.started":"2022-08-08T08:26:41.087596Z","shell.execute_reply":"2022-08-08T08:26:41.256441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### set tokenizer for preprocessing","metadata":{}},{"cell_type":"code","source":"from transformers import DistilBertTokenizer, DistilBertModel\n\n# ../input/dberta/model\nmodel_name = \"../input/d/jellynexus/diltbert\"\n\ntknz = DistilBertTokenizer.from_pretrained(model_name)\n\nsep = tknz._sep_token\n\ntknz(\"do any of you know which software LTT uses for their video script writing\", return_attention_mask=False, truncation=True, )","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:26:41.259550Z","iopub.execute_input":"2022-08-08T08:26:41.259904Z","iopub.status.idle":"2022-08-08T08:26:42.532184Z","shell.execute_reply.started":"2022-08-08T08:26:41.259869Z","shell.execute_reply":"2022-08-08T08:26:42.531171Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Streamlined preprocessing","metadata":{}},{"cell_type":"code","source":"from datasets import Dataset","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:26:42.534294Z","iopub.execute_input":"2022-08-08T08:26:42.535284Z","iopub.status.idle":"2022-08-08T08:26:42.747510Z","shell.execute_reply.started":"2022-08-08T08:26:42.535245Z","shell.execute_reply":"2022-08-08T08:26:42.746514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\" Tokenizing mapping function \"\"\"\n\"\"\" \n    input_tok() -> tokenizing sentence input\n\"\"\"\n\ndef input_tok(x): \n    return tknz(x['input'], return_attention_mask=False, truncation=True)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:26:42.749540Z","iopub.execute_input":"2022-08-08T08:26:42.750532Z","iopub.status.idle":"2022-08-08T08:26:42.756119Z","shell.execute_reply.started":"2022-08-08T08:26:42.750488Z","shell.execute_reply":"2022-08-08T08:26:42.754954Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\" Labeling effectiveness \"\"\"\n\n\"\"\" \n    Encode the effectiveness string to integer type\n\"\"\"\n\ndef labeling(eff):\n    if eff == 'Ineffective': return 0\n    elif eff == 'Adequate': return 1\n    else: return 2","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:26:42.757806Z","iopub.execute_input":"2022-08-08T08:26:42.758174Z","iopub.status.idle":"2022-08-08T08:26:42.769227Z","shell.execute_reply.started":"2022-08-08T08:26:42.758139Z","shell.execute_reply":"2022-08-08T08:26:42.768132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\" Training pipeline\"\"\"\n\n\"\"\" \n    Whole pipeline of preprocessing written in one function\n\"\"\"\n\n#   olddf -> not preprocessed data\n#   isTrain -> train data or not\n\nfrom sklearn import preprocessing\nlabel_encoder = preprocessing.LabelEncoder()\n\n\ndef processing_func(olddf, isTrain):\n    df = pd.DataFrame()\n    df['input'] = olddf['discourse_type'] + sep + olddf['discourse_text']\n    df['essay_id'] = olddf['essay_id'].astype('category')\n    df['essay_id']= label_encoder.fit_transform(olddf['essay_id']) \n    if isTrain:\n        df['label'] = [labeling(x) for x in olddf['discourse_effectiveness']]\n    ds = Dataset.from_pandas(df)\n    ds = ds.map(input_tok, remove_columns=['input'])\n    return ds\n    ","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:26:42.771543Z","iopub.execute_input":"2022-08-08T08:26:42.772424Z","iopub.status.idle":"2022-08-08T08:26:43.114762Z","shell.execute_reply.started":"2022-08-08T08:26:42.772359Z","shell.execute_reply":"2022-08-08T08:26:43.113756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = processing_func(train_data, True)\ndisplay(train_dataset[0])","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:26:43.116093Z","iopub.execute_input":"2022-08-08T08:26:43.116458Z","iopub.status.idle":"2022-08-08T08:27:33.571706Z","shell.execute_reply.started":"2022-08-08T08:26:43.116422Z","shell.execute_reply":"2022-08-08T08:27:33.570790Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset[0]","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:33.572880Z","iopub.execute_input":"2022-08-08T08:27:33.573511Z","iopub.status.idle":"2022-08-08T08:27:33.584155Z","shell.execute_reply.started":"2022-08-08T08:27:33.573471Z","shell.execute_reply":"2022-08-08T08:27:33.583034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = processing_func(test_data, False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:33.585929Z","iopub.execute_input":"2022-08-08T08:27:33.586984Z","iopub.status.idle":"2022-08-08T08:27:33.657925Z","shell.execute_reply.started":"2022-08-08T08:27:33.586945Z","shell.execute_reply":"2022-08-08T08:27:33.656809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset[0]","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:33.662806Z","iopub.execute_input":"2022-08-08T08:27:33.663403Z","iopub.status.idle":"2022-08-08T08:27:33.671846Z","shell.execute_reply.started":"2022-08-08T08:27:33.663345Z","shell.execute_reply":"2022-08-08T08:27:33.670734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Split tokenized test dataset\nSplit tokenized train dataset into 80:20 ratio.","metadata":{}},{"cell_type":"code","source":"splitted_train_dataset = (train_dataset).train_test_split(test_size=0.2, shuffle=True)\n\ntrain_dataset_test = splitted_train_dataset['train']\ntrain_dataset_valid = splitted_train_dataset['test']\n\nprint(len(train_dataset_test), len(train_dataset_valid), sep=\", \")","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:33.673905Z","iopub.execute_input":"2022-08-08T08:27:33.674352Z","iopub.status.idle":"2022-08-08T08:27:33.696076Z","shell.execute_reply.started":"2022-08-08T08:27:33.674311Z","shell.execute_reply":"2022-08-08T08:27:33.695080Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Making model","metadata":{}},{"cell_type":"code","source":"import torch\nfrom transformers import TrainingArguments,Trainer\nfrom transformers import AutoModelForSequenceClassification","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:33.697470Z","iopub.execute_input":"2022-08-08T08:27:33.697997Z","iopub.status.idle":"2022-08-08T08:27:35.268107Z","shell.execute_reply.started":"2022-08-08T08:27:33.697957Z","shell.execute_reply":"2022-08-08T08:27:35.267030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lr,bs = 8e-5,30\nwd,epochs = 0.01,1\n\nimport torch, gc\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:35.269581Z","iopub.execute_input":"2022-08-08T08:27:35.270585Z","iopub.status.idle":"2022-08-08T08:27:35.476725Z","shell.execute_reply.started":"2022-08-08T08:27:35.270545Z","shell.execute_reply":"2022-08-08T08:27:35.475414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import log_loss\nimport torch.nn.functional as F","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:35.478442Z","iopub.execute_input":"2022-08-08T08:27:35.479096Z","iopub.status.idle":"2022-08-08T08:27:35.485342Z","shell.execute_reply.started":"2022-08-08T08:27:35.479057Z","shell.execute_reply":"2022-08-08T08:27:35.484228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def score(preds): return {'log loss': log_loss(preds.label_ids, F.softmax(torch.Tensor(preds.predictions)))}","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:35.486914Z","iopub.execute_input":"2022-08-08T08:27:35.487447Z","iopub.status.idle":"2022-08-08T08:27:35.496319Z","shell.execute_reply.started":"2022-08-08T08:27:35.487409Z","shell.execute_reply":"2022-08-08T08:27:35.495310Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"args = TrainingArguments('outputs', learning_rate=lr, warmup_ratio=0.1, lr_scheduler_type='cosine', fp16=True,\n    evaluation_strategy=\"epoch\", per_device_train_batch_size=bs, per_device_eval_batch_size=bs,\n    num_train_epochs=epochs, weight_decay=wd, report_to='none', save_steps=1000, optim='adagrad')\n\n\"\"\"\nvalid optimizers : ['adamw_hf', 'adamw_torch', 'adamw_torch_xla', 'adamw_apex_fused', 'adafactor', 'adamw_bnb_8bit', 'sgd', 'adagrad']\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:35.497893Z","iopub.execute_input":"2022-08-08T08:27:35.498431Z","iopub.status.idle":"2022-08-08T08:27:35.543458Z","shell.execute_reply.started":"2022-08-08T08:27:35.498391Z","shell.execute_reply":"2022-08-08T08:27:35.542446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoModelForSequenceClassification\nmodel = AutoModelForSequenceClassification.from_pretrained(model_name, num_labels=3)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:35.545467Z","iopub.execute_input":"2022-08-08T08:27:35.546165Z","iopub.status.idle":"2022-08-08T08:27:36.451485Z","shell.execute_reply.started":"2022-08-08T08:27:35.546127Z","shell.execute_reply":"2022-08-08T08:27:36.450397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer = Trainer(model, args, train_dataset=train_dataset_test, eval_dataset=train_dataset_valid,tokenizer=tknz, compute_metrics=score)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:36.453052Z","iopub.execute_input":"2022-08-08T08:27:36.453503Z","iopub.status.idle":"2022-08-08T08:27:38.596795Z","shell.execute_reply.started":"2022-08-08T08:27:36.453458Z","shell.execute_reply":"2022-08-08T08:27:38.595800Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# trainer.train()\n\n# trainer.save_model()","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:38.598429Z","iopub.execute_input":"2022-08-08T08:27:38.598774Z","iopub.status.idle":"2022-08-08T08:27:38.604466Z","shell.execute_reply.started":"2022-08-08T08:27:38.598738Z","shell.execute_reply":"2022-08-08T08:27:38.603446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trainer.predict(test_dataset).predictions","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:38.605898Z","iopub.execute_input":"2022-08-08T08:27:38.606939Z","iopub.status.idle":"2022-08-08T08:27:38.870248Z","shell.execute_reply.started":"2022-08-08T08:27:38.606899Z","shell.execute_reply":"2022-08-08T08:27:38.869320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = F.softmax(torch.Tensor(trainer.predict(test_dataset).predictions)).numpy().astype(float)\npreds","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:38.871975Z","iopub.execute_input":"2022-08-08T08:27:38.872385Z","iopub.status.idle":"2022-08-08T08:27:38.918486Z","shell.execute_reply.started":"2022-08-08T08:27:38.872347Z","shell.execute_reply":"2022-08-08T08:27:38.916434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission_df = pd.read_csv(path + 'sample_submission.csv')\nsubmission_df = pd.read_csv(path + 'sample_submission.csv')\nsubmission_df['Ineffective'] = preds[:,0]\nsubmission_df['Adequate'] = preds[:,1]\nsubmission_df['Effective'] = preds[:,2]\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:27:56.400297Z","iopub.execute_input":"2022-08-08T08:27:56.401000Z","iopub.status.idle":"2022-08-08T08:27:56.420818Z","shell.execute_reply.started":"2022-08-08T08:27:56.400962Z","shell.execute_reply":"2022-08-08T08:27:56.419770Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-08-08T08:28:07.134415Z","iopub.execute_input":"2022-08-08T08:28:07.135121Z","iopub.status.idle":"2022-08-08T08:28:07.141701Z","shell.execute_reply.started":"2022-08-08T08:28:07.135085Z","shell.execute_reply":"2022-08-08T08:28:07.140576Z"},"trusted":true},"execution_count":null,"outputs":[]}]}