{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-19T11:15:03.062020Z","iopub.execute_input":"2022-07-19T11:15:03.062629Z","iopub.status.idle":"2022-07-19T11:15:04.297526Z","shell.execute_reply.started":"2022-07-19T11:15:03.062589Z","shell.execute_reply":"2022-07-19T11:15:04.296561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install tensorflow_text","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-19T11:15:04.299637Z","iopub.execute_input":"2022-07-19T11:15:04.300481Z","iopub.status.idle":"2022-07-19T11:15:13.479701Z","shell.execute_reply.started":"2022-07-19T11:15:04.300438Z","shell.execute_reply":"2022-07-19T11:15:13.478536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport re\n\nimport tensorflow_text \nimport tensorflow as tf\nimport tensorflow_hub as hub\n\n# Suppress warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:15:13.483003Z","iopub.execute_input":"2022-07-19T11:15:13.483314Z","iopub.status.idle":"2022-07-19T11:15:13.490699Z","shell.execute_reply.started":"2022-07-19T11:15:13.483284Z","shell.execute_reply":"2022-07-19T11:15:13.489646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DATA = pd.read_csv(\"../input/feedback-prize-effectiveness/train.csv\")\nTEST_DATA = pd.read_csv(\"../input/feedback-prize-effectiveness/test.csv\")\nsample_data = pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")\n\nprint(f\"Train size : {TRAIN_DATA.shape}\\nTest size : {TEST_DATA.shape}\\nSample size : {sample_data.shape}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:15:29.690800Z","iopub.execute_input":"2022-07-19T11:15:29.691646Z","iopub.status.idle":"2022-07-19T11:15:30.007424Z","shell.execute_reply.started":"2022-07-19T11:15:29.691607Z","shell.execute_reply":"2022-07-19T11:15:30.006488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DATA.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:15:32.403953Z","iopub.execute_input":"2022-07-19T11:15:32.405016Z","iopub.status.idle":"2022-07-19T11:15:32.423658Z","shell.execute_reply.started":"2022-07-19T11:15:32.404960Z","shell.execute_reply":"2022-07-19T11:15:32.422834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DATA.discourse_effectiveness.unique()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:15:32.992309Z","iopub.execute_input":"2022-07-19T11:15:32.992700Z","iopub.status.idle":"2022-07-19T11:15:33.007282Z","shell.execute_reply.started":"2022-07-19T11:15:32.992666Z","shell.execute_reply":"2022-07-19T11:15:33.006227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DATA['discourse_text'].str.len().plot()","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:15:33.464759Z","iopub.execute_input":"2022-07-19T11:15:33.465496Z","iopub.status.idle":"2022-07-19T11:15:33.735779Z","shell.execute_reply.started":"2022-07-19T11:15:33.465446Z","shell.execute_reply":"2022-07-19T11:15:33.734889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"new_label = {\"discourse_effectiveness\": {\"Ineffective\": 0, \"Adequate\": 1, \"Effective\": 2}}\nTRAIN_DATA = TRAIN_DATA.replace(new_label)\nTRAIN_DATA = TRAIN_DATA.rename(columns = {\"discourse_effectiveness\": \"label\"})","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:15:33.737695Z","iopub.execute_input":"2022-07-19T11:15:33.738374Z","iopub.status.idle":"2022-07-19T11:15:33.774617Z","shell.execute_reply.started":"2022-07-19T11:15:33.738335Z","shell.execute_reply":"2022-07-19T11:15:33.773697Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = TRAIN_DATA[:round(TRAIN_DATA.shape[0] * 0.8)]\ntest = TRAIN_DATA[round(TRAIN_DATA.shape[0] * 0.8):]\ntrain.shape,test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:15:33.810308Z","iopub.execute_input":"2022-07-19T11:15:33.810825Z","iopub.status.idle":"2022-07-19T11:15:33.819101Z","shell.execute_reply.started":"2022-07-19T11:15:33.810785Z","shell.execute_reply":"2022-07-19T11:15:33.818137Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_data = TEST_DATA['discourse_text']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:15:34.137517Z","iopub.execute_input":"2022-07-19T11:15:34.137836Z","iopub.status.idle":"2022-07-19T11:15:34.143068Z","shell.execute_reply.started":"2022-07-19T11:15:34.137808Z","shell.execute_reply":"2022-07-19T11:15:34.142024Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preprocessor = \"https://tfhub.dev/tensorflow/bert_en_uncased_preprocess/3\"\nencoder = \"https://tfhub.dev/tensorflow/bert_en_uncased_L-12_H-768_A-12/4\"","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:15:34.273308Z","iopub.execute_input":"2022-07-19T11:15:34.275367Z","iopub.status.idle":"2022-07-19T11:15:34.279795Z","shell.execute_reply.started":"2022-07-19T11:15:34.275334Z","shell.execute_reply":"2022-07-19T11:15:34.278739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bert_preprocessing_model = hub.KerasLayer(preprocessor) # preprocessing step in bert-base model\nbert_model = hub.KerasLayer(encoder) # BERT-base model encoder","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:15:34.487727Z","iopub.execute_input":"2022-07-19T11:15:34.488059Z","iopub.status.idle":"2022-07-19T11:15:55.792857Z","shell.execute_reply.started":"2022-07-19T11:15:34.488030Z","shell.execute_reply":"2022-07-19T11:15:55.791901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"text_input = tf.keras.layers.Input(shape= (),dtype = tf.string,name = 'text')\npreprocessing = bert_preprocessing_model(text_input)\noutput = bert_model(preprocessing)\n\n# Linear layer\nlayer_1 = tf.keras.layers.Dropout(0.15,name = 'dropout')(output['pooled_output'])\nlayer_2 = tf.keras.layers.Dense(3,activation = 'softmax',name = 'output')(layer_1)\n\nmy_model = tf.keras.Model(inputs = text_input,outputs = layer_2)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:15:55.794880Z","iopub.execute_input":"2022-07-19T11:15:55.795315Z","iopub.status.idle":"2022-07-19T11:15:56.536674Z","shell.execute_reply.started":"2022-07-19T11:15:55.795276Z","shell.execute_reply":"2022-07-19T11:15:56.535739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ls = tf.keras.losses.SparseCategoricalCrossentropy()\nmy_model.compile(loss = ls,optimizer = 'adam',metrics = ['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:15:56.538053Z","iopub.execute_input":"2022-07-19T11:15:56.538380Z","iopub.status.idle":"2022-07-19T11:15:56.553640Z","shell.execute_reply.started":"2022-07-19T11:15:56.538346Z","shell.execute_reply":"2022-07-19T11:15:56.552600Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"initial_learning_rate = 0.01\nsgd = tf.keras.optimizers.SGD(learning_rate=initial_learning_rate)\nmy_model.compile(loss = 'sparse_categorical_crossentropy',optimizer = sgd,metrics = ['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:15:56.556311Z","iopub.execute_input":"2022-07-19T11:15:56.556729Z","iopub.status.idle":"2022-07-19T11:15:56.568131Z","shell.execute_reply.started":"2022-07-19T11:15:56.556692Z","shell.execute_reply":"2022-07-19T11:15:56.567154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x = TRAIN_DATA['discourse_text']\ny = TRAIN_DATA['label']","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:15:56.569775Z","iopub.execute_input":"2022-07-19T11:15:56.570470Z","iopub.status.idle":"2022-07-19T11:15:56.575409Z","shell.execute_reply.started":"2022-07-19T11:15:56.570434Z","shell.execute_reply":"2022-07-19T11:15:56.574352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = my_model.fit(x,y,batch_size = 32,epochs = 3)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:20:32.038919Z","iopub.execute_input":"2022-07-19T11:20:32.039593Z","iopub.status.idle":"2022-07-19T11:32:51.588574Z","shell.execute_reply.started":"2022-07-19T11:20:32.039554Z","shell.execute_reply":"2022-07-19T11:32:51.587649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = my_model.predict(test_data)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:37:25.804636Z","iopub.execute_input":"2022-07-19T11:37:25.805022Z","iopub.status.idle":"2022-07-19T11:37:25.932191Z","shell.execute_reply.started":"2022-07-19T11:37:25.804991Z","shell.execute_reply":"2022-07-19T11:37:25.931303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred > 0.5","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:37:27.436893Z","iopub.execute_input":"2022-07-19T11:37:27.437753Z","iopub.status.idle":"2022-07-19T11:37:27.444216Z","shell.execute_reply.started":"2022-07-19T11:37:27.437704Z","shell.execute_reply":"2022-07-19T11:37:27.443236Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.read_csv(\"../input/feedback-prize-effectiveness/sample_submission.csv\")\nsubmission_df['Ineffective'] = pred[:,0]\nsubmission_df['Adequate'] = pred[:,1]\nsubmission_df['Effective'] = pred[:,2]\nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:37:31.873363Z","iopub.execute_input":"2022-07-19T11:37:31.875762Z","iopub.status.idle":"2022-07-19T11:37:31.898997Z","shell.execute_reply.started":"2022-07-19T11:37:31.875724Z","shell.execute_reply":"2022-07-19T11:37:31.898139Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-19T11:37:33.266679Z","iopub.execute_input":"2022-07-19T11:37:33.267879Z","iopub.status.idle":"2022-07-19T11:37:33.277769Z","shell.execute_reply.started":"2022-07-19T11:37:33.267807Z","shell.execute_reply":"2022-07-19T11:37:33.276604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}