{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install transformers==4.20.1","metadata":{"execution":{"iopub.status.busy":"2022-08-03T00:15:23.397483Z","iopub.execute_input":"2022-08-03T00:15:23.397817Z","iopub.status.idle":"2022-08-03T00:15:39.359572Z","shell.execute_reply.started":"2022-08-03T00:15:23.397738Z","shell.execute_reply":"2022-08-03T00:15:39.358711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport re\nimport spacy\nfrom spacy import displacy\n\nimport nltk\nfrom nltk import word_tokenize\nfrom nltk import tokenize\nfrom collections import Counter\n\nimport os\nos.environ[\"WANDB_SILENT\"] = \"true\"\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3' # Disable tensorflow debugging logs\n\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input, Dropout\nfrom tensorflow.keras.optimizers import Adam,SGD\nfrom tensorflow.keras.models import Model\nimport transformers\n\nfrom sklearn.model_selection import GroupKFold\nfrom sklearn.metrics import confusion_matrix\nfrom sklearn.metrics import log_loss\n\nimport warnings # Supress warnings\nwarnings.filterwarnings(\"ignore\")\n\nSEED = 42\n# 1. Set the `PYTHONHASHSEED` environment variable at a fixed value\nos.environ['PYTHONHASHSEED'] = str(SEED)\n\n# 2. Set the `python` built-in pseudo-random generator at a fixed value\nimport random\nrandom.seed(SEED)\n\n# 3. Set the `numpy` pseudo-random generator at a fixed value\nnp.random.seed(SEED)\n\n# 4. Set the `tensorflow` pseudo-random generator at a fixed value\ntf.random.set_seed(SEED)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-03T00:15:39.363377Z","iopub.execute_input":"2022-08-03T00:15:39.363602Z","iopub.status.idle":"2022-08-03T00:15:47.594743Z","shell.execute_reply.started":"2022-08-03T00:15:39.363575Z","shell.execute_reply":"2022-08-03T00:15:47.593960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T00:15:47.595930Z","iopub.execute_input":"2022-08-03T00:15:47.596275Z","iopub.status.idle":"2022-08-03T00:15:47.857814Z","shell.execute_reply.started":"2022-08-03T00:15:47.596212Z","shell.execute_reply":"2022-08-03T00:15:47.857055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"label\"] = train[\"discourse_effectiveness\"].replace({\"Ineffective\": 0, \"Adequate\": 1, \"Effective\": 2})","metadata":{"execution":{"iopub.status.busy":"2022-08-03T00:15:47.859921Z","iopub.execute_input":"2022-08-03T00:15:47.860191Z","iopub.status.idle":"2022-08-03T00:15:47.888692Z","shell.execute_reply.started":"2022-08-03T00:15:47.860159Z","shell.execute_reply":"2022-08-03T00:15:47.887876Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Configuration\nBATCH_SIZE = 16\nMAX_LEN = 256 \nLEARNING_RATE = 1e-5\nEPOCHS = 10\nAUTO = tf.data.experimental.AUTOTUNE\nMODEL = \"distilbert\"\nMODEL_PATH = f\"distilbert-base-uncased\"","metadata":{"execution":{"iopub.status.busy":"2022-08-03T00:15:47.890093Z","iopub.execute_input":"2022-08-03T00:15:47.890363Z","iopub.status.idle":"2022-08-03T00:15:47.895066Z","shell.execute_reply.started":"2022-08-03T00:15:47.890328Z","shell.execute_reply":"2022-08-03T00:15:47.894268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = transformers.AutoTokenizer.from_pretrained(MODEL_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T00:15:47.896786Z","iopub.execute_input":"2022-08-03T00:15:47.897125Z","iopub.status.idle":"2022-08-03T00:15:53.117239Z","shell.execute_reply.started":"2022-08-03T00:15:47.897087Z","shell.execute_reply":"2022-08-03T00:15:53.116469Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer.save_pretrained(\"./\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:37:33.293544Z","iopub.execute_input":"2022-08-03T07:37:33.294236Z","iopub.status.idle":"2022-08-03T07:37:33.344732Z","shell.execute_reply.started":"2022-08-03T07:37:33.294204Z","shell.execute_reply":"2022-08-03T07:37:33.344063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sep = tokenizer.sep_token\nprint(sep)\n\ntrain['inputs'] = train.discourse_type + sep + train.discourse_text","metadata":{"execution":{"iopub.status.busy":"2022-08-03T00:15:53.118614Z","iopub.execute_input":"2022-08-03T00:15:53.119033Z","iopub.status.idle":"2022-08-03T00:15:53.155720Z","shell.execute_reply.started":"2022-08-03T00:15:53.118995Z","shell.execute_reply":"2022-08-03T00:15:53.154955Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef bert_encode(texts, tokenizer, max_len = MAX_LEN):\n    input_ids = np.zeros((len(texts), max_len), dtype=\"int32\")\n    #token_type_ids = np.zeros((len(texts), max_len), dtype=\"int32\")\n    attention_mask = np.zeros((len(texts), max_len), dtype=\"int32\")\n    \n    for i, text in enumerate(texts):\n        token = tokenizer(text, \n                          max_length         = max_len, \n                          truncation         = True, \n                          padding            = \"max_length\",\n                          add_special_tokens = True,\n                          return_tensors     = \"np\")\n        \n        input_ids[i] = token['input_ids']\n        #token_type_ids[i] = token['token_type_ids']\n        attention_mask[i] = token['attention_mask']\n    return input_ids, attention_mask   #token_type_ids, ","metadata":{"execution":{"iopub.status.busy":"2022-08-03T00:15:53.156844Z","iopub.execute_input":"2022-08-03T00:15:53.157225Z","iopub.status.idle":"2022-08-03T00:15:53.164238Z","shell.execute_reply.started":"2022-08-03T00:15:53.157191Z","shell.execute_reply":"2022-08-03T00:15:53.163200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(bert_mdoel):\n    input_ids = Input(shape = (MAX_LEN, ), dtype = tf.int32, name = \"input_ids\")\n    attention_mask = Input(shape = (MAX_LEN, ), dtype = tf.int32, name = \"attention_mask\")\n\n    sequence_output = bert_model(input_ids, \n                                        attention_mask = attention_mask)[0]\n\n    clf_output = sequence_output[:, 0, :]\n    clf_output1 = Dropout(0.1)(clf_output)\n    clf_output2 = Dropout(0.2)(clf_output)\n    clf_output3 = Dropout(0.3)(clf_output)\n    clf_output4 = Dropout(0.4)(clf_output)\n    clf_output5 = Dropout(0.5)(clf_output)\n    out1 = Dense(3, activation='softmax')(clf_output1)\n    out2 = Dense(3, activation='softmax')(clf_output2)\n    out3 = Dense(3, activation='softmax')(clf_output3)\n    out4 = Dense(3, activation='softmax')(clf_output4)\n    out5 = Dense(3, activation='softmax')(clf_output5)\n    out = (out1 + out2 + out3 + out4 + out5) / 5\n\n    model = Model(inputs = [input_ids, attention_mask], \n                  outputs = out)\n\n    model.compile(Adam(learning_rate = LEARNING_RATE, \n                      #decay=1e-6\n                      ), \n                  loss = 'sparse_categorical_crossentropy', \n                  metrics = ['accuracy'])\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-08-03T00:15:53.165460Z","iopub.execute_input":"2022-08-03T00:15:53.165839Z","iopub.status.idle":"2022-08-03T00:15:53.179159Z","shell.execute_reply.started":"2022-08-03T00:15:53.165801Z","shell.execute_reply":"2022-08-03T00:15:53.178337Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X = train['inputs']\ny = train['label']\n\nkf = GroupKFold(n_splits = 5)\n\nbert_model = transformers.TFAutoModel.from_pretrained(MODEL_PATH)\nmodel = build_model(bert_model)\n\nfor i, (train_index, val_index) in enumerate(kf.split(X, y, train[\"essay_id\"])):  \n    print(f\"Fold {i+1}: Train Set: {train.loc[train_index, 'essay_id'].nunique()}, Validation Set: {train.loc[val_index, 'essay_id'].nunique()}\")\n\n    X_train = X.loc[train_index].values\n    X_train = bert_encode(X_train.astype(str), tokenizer)\n\n    X_valid = X.loc[val_index].values\n    X_valid = bert_encode(X_valid.astype(str), tokenizer)\n\n    y_train = y[train_index].values\n    y_valid = y[val_index].values\n\n    train_dataset = (\n        tf.data.Dataset\n        .from_tensor_slices((X_train, y_train))\n        .repeat()\n        .shuffle(SEED)\n        .batch(BATCH_SIZE)\n        .prefetch(AUTO)\n    )\n    valid_dataset = (\n        tf.data.Dataset\n        .from_tensor_slices((X_valid, y_valid))\n        .batch(BATCH_SIZE)\n        .cache()\n        .prefetch(AUTO)\n    )\n\n    print(f\"Steps per Epoch: {len(train_index) // BATCH_SIZE}\")\n    save_best = tf.keras.callbacks.ModelCheckpoint(\"distilbert_{}.h5\".format(i),monitor=\"val_accuracy\",\n                                                  save_best_only=True,verbose=1)\n    model.fit(\n        train_dataset,\n        steps_per_epoch = len(train_index) // BATCH_SIZE,\n        validation_data=valid_dataset,\n        epochs=EPOCHS, \n        callbacks=[save_best],\n        verbose = 2,\n    )\n\n    # Validation\n    y_valid_pred = model.predict(X_valid, verbose=1)\n    print(f\"Validation Log Loss {log_loss(y_valid, y_valid_pred):.2f}\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T00:15:53.182073Z","iopub.execute_input":"2022-08-03T00:15:53.182417Z","iopub.status.idle":"2022-08-03T07:15:11.317781Z","shell.execute_reply.started":"2022-08-03T00:15:53.182309Z","shell.execute_reply":"2022-08-03T07:15:11.317073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from IPython.display import FileLinks\nFileLinks('./')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T07:37:43.894773Z","iopub.execute_input":"2022-08-03T07:37:43.895457Z","iopub.status.idle":"2022-08-03T07:37:43.902867Z","shell.execute_reply.started":"2022-08-03T07:37:43.895422Z","shell.execute_reply":"2022-08-03T07:37:43.902161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}