{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T00:40:42.132879Z","iopub.execute_input":"2022-07-12T00:40:42.133195Z","iopub.status.idle":"2022-07-12T00:40:42.296260Z","shell.execute_reply.started":"2022-07-12T00:40:42.133163Z","shell.execute_reply":"2022-07-12T00:40:42.295513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_PATH = '../input/huggingface-bert-variants/bert-base-cased/bert-base-cased' ","metadata":{"execution":{"iopub.status.busy":"2022-07-10T23:19:11.110019Z","iopub.execute_input":"2022-07-10T23:19:11.110388Z","iopub.status.idle":"2022-07-10T23:19:11.114115Z","shell.execute_reply.started":"2022-07-10T23:19:11.110359Z","shell.execute_reply":"2022-07-10T23:19:11.113568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pyforest","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:40:53.705023Z","iopub.execute_input":"2022-07-12T00:40:53.705372Z","iopub.status.idle":"2022-07-12T00:41:06.986876Z","shell.execute_reply.started":"2022-07-12T00:40:53.705345Z","shell.execute_reply":"2022-07-12T00:41:06.985816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from pyforest import *\nimport pandas as pd\nimport numpy as np\nimport os\nimport warnings\nwarnings.filterwarnings('ignore')\nimport nltk\nfrom nltk.corpus import stopwords\nfrom nltk.stem import SnowballStemmer\nfrom string import punctuation\nfrom nltk.stem.wordnet import WordNetLemmatizer\nfrom tqdm import tqdm\nimport re\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport time\nimport datetime\nfrom scipy import sparse\nimport datasets, transformers\nfrom transformers import TrainingArguments, Trainer\nfrom transformers import AutoModelForSequenceClassification, AutoTokenizer\nfrom transformers import AutoModelForMaskedLM\nos.environ['WANDB_DISABLED'] = 'true'\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input, Dropout\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom transformers import TFBertModel","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:10.539622Z","iopub.execute_input":"2022-07-12T00:41:10.540371Z","iopub.status.idle":"2022-07-12T00:41:20.026345Z","shell.execute_reply.started":"2022-07-12T00:41:10.540333Z","shell.execute_reply":"2022-07-12T00:41:20.025393Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\nEPOCHS = 20\nBATCH_SIZE = 32\nMAX_LEN = 128","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:20.028234Z","iopub.execute_input":"2022-07-12T00:41:20.028932Z","iopub.status.idle":"2022-07-12T00:41:20.039825Z","shell.execute_reply.started":"2022-07-12T00:41:20.028902Z","shell.execute_reply":"2022-07-12T00:41:20.035066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/train.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:21.495762Z","iopub.execute_input":"2022-07-12T00:41:21.496396Z","iopub.status.idle":"2022-07-12T00:41:21.780221Z","shell.execute_reply.started":"2022-07-12T00:41:21.496360Z","shell.execute_reply":"2022-07-12T00:41:21.779421Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:22.037203Z","iopub.execute_input":"2022-07-12T00:41:22.037894Z","iopub.status.idle":"2022-07-12T00:41:22.047630Z","shell.execute_reply.started":"2022-07-12T00:41:22.037864Z","shell.execute_reply":"2022-07-12T00:41:22.046852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample = pd.read_csv('/kaggle/input/feedback-prize-effectiveness/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:22.833460Z","iopub.execute_input":"2022-07-12T00:41:22.833872Z","iopub.status.idle":"2022-07-12T00:41:22.849297Z","shell.execute_reply.started":"2022-07-12T00:41:22.833838Z","shell.execute_reply":"2022-07-12T00:41:22.848489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:23.793675Z","iopub.execute_input":"2022-07-12T00:41:23.794315Z","iopub.status.idle":"2022-07-12T00:41:23.817305Z","shell.execute_reply.started":"2022-07-12T00:41:23.794281Z","shell.execute_reply":"2022-07-12T00:41:23.816564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:24.481682Z","iopub.execute_input":"2022-07-12T00:41:24.482338Z","iopub.status.idle":"2022-07-12T00:41:24.492415Z","shell.execute_reply.started":"2022-07-12T00:41:24.482304Z","shell.execute_reply":"2022-07-12T00:41:24.491414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:25.076296Z","iopub.execute_input":"2022-07-12T00:41:25.076925Z","iopub.status.idle":"2022-07-12T00:41:25.089085Z","shell.execute_reply.started":"2022-07-12T00:41:25.076891Z","shell.execute_reply":"2022-07-12T00:41:25.088065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:25.527299Z","iopub.execute_input":"2022-07-12T00:41:25.527644Z","iopub.status.idle":"2022-07-12T00:41:25.533155Z","shell.execute_reply.started":"2022-07-12T00:41:25.527617Z","shell.execute_reply":"2022-07-12T00:41:25.532324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:27.511499Z","iopub.execute_input":"2022-07-12T00:41:27.512119Z","iopub.status.idle":"2022-07-12T00:41:27.517810Z","shell.execute_reply.started":"2022-07-12T00:41:27.512087Z","shell.execute_reply":"2022-07-12T00:41:27.516891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:27.985620Z","iopub.execute_input":"2022-07-12T00:41:27.986335Z","iopub.status.idle":"2022-07-12T00:41:27.992286Z","shell.execute_reply.started":"2022-07-12T00:41:27.986302Z","shell.execute_reply":"2022-07-12T00:41:27.991085Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def cleanup_text(text):\n    words = re.sub(pattern = '[^a-zA-Z]', repl = ' ', string = text)\n    words = words.lower()\n    return words\ncleanup_text('Every Mountain Speaks Different Ways!')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:28.623692Z","iopub.execute_input":"2022-07-12T00:41:28.624075Z","iopub.status.idle":"2022-07-12T00:41:28.634412Z","shell.execute_reply.started":"2022-07-12T00:41:28.624047Z","shell.execute_reply":"2022-07-12T00:41:28.633611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_preprocessed = train_df['discourse_text'].apply(cleanup_text)\nprint(text_preprocessed)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:29.135787Z","iopub.execute_input":"2022-07-12T00:41:29.136470Z","iopub.status.idle":"2022-07-12T00:41:29.966159Z","shell.execute_reply.started":"2022-07-12T00:41:29.136432Z","shell.execute_reply":"2022-07-12T00:41:29.964427Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['text_preprocessed'] = text_preprocessed\ndisplay(train_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:29.967915Z","iopub.execute_input":"2022-07-12T00:41:29.968280Z","iopub.status.idle":"2022-07-12T00:41:29.981700Z","shell.execute_reply.started":"2022-07-12T00:41:29.968245Z","shell.execute_reply":"2022-07-12T00:41:29.980786Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def bert_encode(texts, tokenizer, max_len=MAX_LEN):\n    input_ids = []\n    token_type_ids = []\n    attention_mask = []\n    \n    for text in texts:\n        token = tokenizer(text, max_length=max_len, truncation=True, padding='max_length', add_special_tokens=True)\n        input_ids.append(token['input_ids'])\n        token_type_ids.append(token['token_type_ids'])\n        attention_mask.append(token['attention_mask'])\n    return np.array(input_ids), np.array(token_type_ids), np.array(attention_mask)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:30.141459Z","iopub.execute_input":"2022-07-12T00:41:30.141816Z","iopub.status.idle":"2022-07-12T00:41:30.147617Z","shell.execute_reply.started":"2022-07-12T00:41:30.141788Z","shell.execute_reply":"2022-07-12T00:41:30.146794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = transformers.BertTokenizer.from_pretrained('../input/huggingface-bert-variants/distilbert-base-cased/distilbert-base-cased')\ntokenizer.save_pretrained('.')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:30.727119Z","iopub.execute_input":"2022-07-12T00:41:30.727841Z","iopub.status.idle":"2022-07-12T00:41:30.804435Z","shell.execute_reply.started":"2022-07-12T00:41:30.727806Z","shell.execute_reply":"2022-07-12T00:41:30.803596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sep = tokenizer.sep_token\nsep","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:33.523003Z","iopub.execute_input":"2022-07-12T00:41:33.523354Z","iopub.status.idle":"2022-07-12T00:41:33.528358Z","shell.execute_reply.started":"2022-07-12T00:41:33.523326Z","shell.execute_reply":"2022-07-12T00:41:33.527558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['inputs'] = train_df.discourse_type + sep + train_df.text_preprocessed\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:34.034293Z","iopub.execute_input":"2022-07-12T00:41:34.034855Z","iopub.status.idle":"2022-07-12T00:41:34.070808Z","shell.execute_reply.started":"2022-07-12T00:41:34.034822Z","shell.execute_reply":"2022-07-12T00:41:34.070090Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bin_map = {'discourse_effectiveness': {'Ineffective': 0, 'Adequate': 1, 'Effective': 2}}\ntrain_df = train_df.replace(bin_map)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:34.501672Z","iopub.execute_input":"2022-07-12T00:41:34.502211Z","iopub.status.idle":"2022-07-12T00:41:34.550280Z","shell.execute_reply.started":"2022-07-12T00:41:34.502178Z","shell.execute_reply":"2022-07-12T00:41:34.549481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:34.920265Z","iopub.execute_input":"2022-07-12T00:41:34.920609Z","iopub.status.idle":"2022-07-12T00:41:34.932452Z","shell.execute_reply.started":"2022-07-12T00:41:34.920579Z","shell.execute_reply":"2022-07-12T00:41:34.931575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_valid, y_train, y_valid = train_test_split(train_df['inputs'], train_df['discourse_effectiveness'], test_size=0.2, random_state=12)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:35.316050Z","iopub.execute_input":"2022-07-12T00:41:35.316750Z","iopub.status.idle":"2022-07-12T00:41:35.329148Z","shell.execute_reply.started":"2022-07-12T00:41:35.316695Z","shell.execute_reply":"2022-07-12T00:41:35.328256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train = bert_encode(X_train.astype(str), tokenizer)\nX_valid = bert_encode(X_valid.astype(str), tokenizer)\ny_train = y_train.values\ny_valid = y_valid.values","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:41:36.704677Z","iopub.execute_input":"2022-07-12T00:41:36.705231Z","iopub.status.idle":"2022-07-12T00:42:29.060556Z","shell.execute_reply.started":"2022-07-12T00:41:36.705194Z","shell.execute_reply":"2022-07-12T00:42:29.059757Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((X_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((X_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:42:29.062370Z","iopub.execute_input":"2022-07-12T00:42:29.062743Z","iopub.status.idle":"2022-07-12T00:42:34.313110Z","shell.execute_reply.started":"2022-07-12T00:42:29.062687Z","shell.execute_reply":"2022-07-12T00:42:34.308010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model(bert_model, max_len=MAX_LEN):\n    input_ids = Input(shape=(max_len,), dtype=tf.int32, name='input_ids')\n    token_type_ids = Input(shape=(max_len,), dtype=tf.int32, name='token_type_ids')\n    attention_mask = Input(shape=(max_len,), dtype=tf.int32, name='attention_mask')\n    \n    sequence_output = bert_model(input_ids, token_type_ids=token_type_ids, attention_mask=attention_mask)[0]\n    clf_output = sequence_output[:, 0, :]\n    clf_output = Dropout(.1)(clf_output)\n    out = Dense(3, activation='softmax')(clf_output)\n    model = Model(inputs=[input_ids, token_type_ids, attention_mask], outputs=out)\n    model.compile(Adam(lr=1e-5), loss = 'sparse_categorical_crossentropy', metrics=['accuracy'])\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:42:34.315132Z","iopub.execute_input":"2022-07-12T00:42:34.316457Z","iopub.status.idle":"2022-07-12T00:42:34.333594Z","shell.execute_reply.started":"2022-07-12T00:42:34.316418Z","shell.execute_reply":"2022-07-12T00:42:34.332117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntransformer_layer = (TFBertModel.from_pretrained('../input/huggingface-bert-variants/distilbert-base-cased/distilbert-base-cased'))\nmodel = build_model(transformer_layer, max_len=MAX_LEN)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:42:34.340297Z","iopub.execute_input":"2022-07-12T00:42:34.340703Z","iopub.status.idle":"2022-07-12T00:42:42.728326Z","shell.execute_reply.started":"2022-07-12T00:42:34.340668Z","shell.execute_reply":"2022-07-12T00:42:42.727484Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.utils import plot_model\nplot_model(model)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:42:42.730163Z","iopub.execute_input":"2022-07-12T00:42:42.731470Z","iopub.status.idle":"2022-07-12T00:42:43.726602Z","shell.execute_reply.started":"2022-07-12T00:42:42.731430Z","shell.execute_reply":"2022-07-12T00:42:43.725766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_history = model.fit(\n        train_dataset,\n        steps_per_epoch = 200,\n        validation_data = valid_dataset,\n        epochs = EPOCHS\n)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:42:43.728252Z","iopub.execute_input":"2022-07-12T00:42:43.728842Z","iopub.status.idle":"2022-07-12T01:27:21.948406Z","shell.execute_reply.started":"2022-07-12T00:42:43.728802Z","shell.execute_reply":"2022-07-12T01:27:21.947373Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_preprocessed = test_df['discourse_text'].apply(cleanup_text)\ntest_preprocessed","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:30:36.328228Z","iopub.execute_input":"2022-07-12T01:30:36.328665Z","iopub.status.idle":"2022-07-12T01:30:36.343082Z","shell.execute_reply.started":"2022-07-12T01:30:36.328627Z","shell.execute_reply":"2022-07-12T01:30:36.341793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['text_preprocessed'] = test_preprocessed\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:31:16.153328Z","iopub.execute_input":"2022-07-12T01:31:16.153675Z","iopub.status.idle":"2022-07-12T01:31:16.168576Z","shell.execute_reply.started":"2022-07-12T01:31:16.153647Z","shell.execute_reply":"2022-07-12T01:31:16.167660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df['processed'] = test_df.discourse_type + sep + test_df.text_preprocessed","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:32:30.675115Z","iopub.execute_input":"2022-07-12T01:32:30.675539Z","iopub.status.idle":"2022-07-12T01:32:30.684521Z","shell.execute_reply.started":"2022-07-12T01:32:30.675502Z","shell.execute_reply":"2022-07-12T01:32:30.683727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_processed = bert_encode(test_df.processed.astype(str), tokenizer)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:35:02.988929Z","iopub.execute_input":"2022-07-12T01:35:02.989836Z","iopub.status.idle":"2022-07-12T01:35:03.009671Z","shell.execute_reply.started":"2022-07-12T01:35:02.989795Z","shell.execute_reply":"2022-07-12T01:35:03.008649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = model.predict(test_processed, verbose = 1)\npreds","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:35:31.683691Z","iopub.execute_input":"2022-07-12T01:35:31.684281Z","iopub.status.idle":"2022-07-12T01:35:35.295475Z","shell.execute_reply.started":"2022-07-12T01:35:31.684248Z","shell.execute_reply":"2022-07-12T01:35:35.294586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample['Ineffective'] = preds[:,0]\nsample['Adequate'] = preds[:,1]\nsample['Effective'] = preds[:,2]\nsample.sample(2)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:38:37.920502Z","iopub.execute_input":"2022-07-12T01:38:37.920896Z","iopub.status.idle":"2022-07-12T01:38:37.933706Z","shell.execute_reply.started":"2022-07-12T01:38:37.920866Z","shell.execute_reply":"2022-07-12T01:38:37.932581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample.to_csv(\"submission.csv\", index=False)\nprint('Success')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:38:56.503022Z","iopub.execute_input":"2022-07-12T01:38:56.503371Z","iopub.status.idle":"2022-07-12T01:38:56.510300Z","shell.execute_reply.started":"2022-07-12T01:38:56.503343Z","shell.execute_reply":"2022-07-12T01:38:56.509440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}