{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n   # for filename in filenames:\n        #print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-08T09:36:21.148203Z","iopub.execute_input":"2022-07-08T09:36:21.149184Z","iopub.status.idle":"2022-07-08T09:36:21.176972Z","shell.execute_reply.started":"2022-07-08T09:36:21.149072Z","shell.execute_reply":"2022-07-08T09:36:21.175959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(\"/kaggle/input/feedback-prize-effectiveness/train.csv\")\ndata[\"text_type\"] = data[\"discourse_type\"] + ' ' + data[\"discourse_text\"]\n#X = data[\"text_type\"] ","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:36:21.180875Z","iopub.execute_input":"2022-07-08T09:36:21.182194Z","iopub.status.idle":"2022-07-08T09:36:21.602991Z","shell.execute_reply.started":"2022-07-08T09:36:21.182152Z","shell.execute_reply":"2022-07-08T09:36:21.601749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndata_train = data.groupby(by=\"discourse_effectiveness\").sample(frac=0.995)\ndata_test = data.drop(index = data_train.index)\n\ndata_train = data_train.sample(frac=1)\ndata_test = data_test.sample(frac=1)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:36:21.605388Z","iopub.execute_input":"2022-07-08T09:36:21.605747Z","iopub.status.idle":"2022-07-08T09:36:21.660787Z","shell.execute_reply.started":"2022-07-08T09:36:21.605711Z","shell.execute_reply":"2022-07-08T09:36:21.659740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(data_train))\nprint(len(data_test))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:36:21.662146Z","iopub.execute_input":"2022-07-08T09:36:21.662841Z","iopub.status.idle":"2022-07-08T09:36:21.669868Z","shell.execute_reply.started":"2022-07-08T09:36:21.662803Z","shell.execute_reply":"2022-07-08T09:36:21.668264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn import preprocessing\n\nle = preprocessing.LabelEncoder()\ny_train = le.fit_transform(data_train[\"discourse_effectiveness\"].values)\ny_test = le.transform(data_test[\"discourse_effectiveness\"].values)\n\n#y = le.fit_transform(data[\"discourse_effectiveness\"].values)\n\nX_train = data_train[\"text_type\"]\nX_test = data_test[\"text_type\"]","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:36:21.672528Z","iopub.execute_input":"2022-07-08T09:36:21.673506Z","iopub.status.idle":"2022-07-08T09:36:22.721132Z","shell.execute_reply.started":"2022-07-08T09:36:21.673470Z","shell.execute_reply":"2022-07-08T09:36:22.719902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n\nVOCAB_LEN = 10000\ntokenizer = tf.keras.preprocessing.text.Tokenizer(num_words=VOCAB_LEN, lower=True, oov_token='<UNK>')\ntokenizer.fit_on_texts(X_train)\ntrain_seq = tokenizer.texts_to_sequences(X_train)\ntest_seq = tokenizer.texts_to_sequences(X_test)\n\n#tokenizer.fit_on_texts(X)\n#seq = tokenizer.texts_to_sequences(X)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:36:22.722897Z","iopub.execute_input":"2022-07-08T09:36:22.723688Z","iopub.status.idle":"2022-07-08T09:36:32.089791Z","shell.execute_reply.started":"2022-07-08T09:36:22.723647Z","shell.execute_reply":"2022-07-08T09:36:32.088802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def FindMaxLength(lst):\n    maxList = max(lst, key = lambda i: len(i))\n    maxLength = len(maxList)\n      \n    return maxLength\n\nMAX_LEN = FindMaxLength(train_seq)\nprint(MAX_LEN)\n#MAX_LEN = FindMaxLength(seq)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:36:32.091454Z","iopub.execute_input":"2022-07-08T09:36:32.092098Z","iopub.status.idle":"2022-07-08T09:36:32.108038Z","shell.execute_reply.started":"2022-07-08T09:36:32.092070Z","shell.execute_reply":"2022-07-08T09:36:32.107162Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\n\npadded_train_seq = np.array(pad_sequences(train_seq, maxlen=MAX_LEN, padding='post', truncating='post'))\npadded_test_seq = np.array(pad_sequences(test_seq, maxlen=MAX_LEN, padding='post', truncating='post'))\n\nprint(padded_train_seq.shape)\nprint(padded_test_seq.shape)\n\n#padded_seq = np.array(pad_sequences(seq, maxlen=MAX_LEN, padding='post', truncating='post'))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:36:32.109750Z","iopub.execute_input":"2022-07-08T09:36:32.110396Z","iopub.status.idle":"2022-07-08T09:36:32.460153Z","shell.execute_reply.started":"2022-07-08T09:36:32.110347Z","shell.execute_reply":"2022-07-08T09:36:32.458929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model():\n        \n        inputs = tf.keras.layers.Input(shape=(MAX_LEN,))\n        h = tf.keras.layers.Embedding(input_dim=VOCAB_LEN, output_dim=768)(inputs)\n        h = tf.keras.layers.Lambda(lambda x: tf.keras.backend.mean(x, axis=1))(h)\n\n        h = tf.keras.layers.Dense(512, activation='relu')(h)\n        h = tf.keras.layers.Dropout(0.1)(h)\n        h = tf.keras.layers.BatchNormalization()(h)\n       \n        h = tf.keras.layers.Dense(256, activation='relu')(h)\n        h = tf.keras.layers.Dropout(0.1)(h)\n        \n        outputs = tf.keras.layers.Dense(3, activation='softmax')(h)\n\n        model = tf.keras.Model(inputs=inputs, outputs=outputs, name=\"my_model\")\n        model.compile(optimizer=tf.keras.optimizers.Adam(3e-4),\n                      loss='sparse_categorical_crossentropy',\n                      metrics=['accuracy'])\n\n        \n        return model","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:36:32.461670Z","iopub.execute_input":"2022-07-08T09:36:32.464978Z","iopub.status.idle":"2022-07-08T09:36:32.475258Z","shell.execute_reply.started":"2022-07-08T09:36:32.464944Z","shell.execute_reply":"2022-07-08T09:36:32.473943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"callback = tf.keras.callbacks.EarlyStopping(\n    monitor='val_loss',\n    patience=3)\n'''\ncallback_train = tf.keras.callbacks.EarlyStopping(\n    monitor='loss',\n    patience=1)\n'''","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:36:32.476754Z","iopub.execute_input":"2022-07-08T09:36:32.477445Z","iopub.status.idle":"2022-07-08T09:36:32.490388Z","shell.execute_reply.started":"2022-07-08T09:36:32.477403Z","shell.execute_reply":"2022-07-08T09:36:32.489179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model()\n\nmodel.fit(padded_train_seq,\n          y_train, \n          epochs=50, \n          batch_size=64, \n          validation_data=(padded_test_seq, y_test),\n          callbacks=[callback])\n'''\nmodel.fit(padded_seq,\n          y, \n          epochs=5, \n          batch_size=64, \n          callbacks=[callback_train])\n'''","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:36:32.495874Z","iopub.execute_input":"2022-07-08T09:36:32.496842Z","iopub.status.idle":"2022-07-08T09:37:19.184940Z","shell.execute_reply.started":"2022-07-08T09:36:32.496773Z","shell.execute_reply":"2022-07-08T09:37:19.183618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_eval = pd.read_csv(\"/kaggle/input/feedback-prize-effectiveness/test.csv\")\ndata_eval[\"text_type\"] = data_eval[\"discourse_type\"] + ' ' + data_eval[\"discourse_text\"]\nX_eval =  data_eval[\"text_type\"]\neval_seq = tokenizer.texts_to_sequences(X_eval)\npadded_eval_seq = np.array(pad_sequences(eval_seq, maxlen=MAX_LEN, padding='post', truncating='post'))\npredictions = model.predict(padded_eval_seq)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:37:58.870733Z","iopub.execute_input":"2022-07-08T09:37:58.871168Z","iopub.status.idle":"2022-07-08T09:37:58.928092Z","shell.execute_reply.started":"2022-07-08T09:37:58.871127Z","shell.execute_reply":"2022-07-08T09:37:58.926923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"col = [\"Adequate\", \"Effective\", \"Ineffective\"]\nsub_df = pd.DataFrame(predictions, columns=col)\nsub_df = pd.concat([sub_df, data_eval[\"discourse_id\"]], axis = 1)\nsub_df = sub_df[['discourse_id', 'Ineffective', 'Adequate', 'Effective']]\nsub_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:37:58.931114Z","iopub.execute_input":"2022-07-08T09:37:58.931550Z","iopub.status.idle":"2022-07-08T09:37:58.948604Z","shell.execute_reply.started":"2022-07-08T09:37:58.931510Z","shell.execute_reply":"2022-07-08T09:37:58.947351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df.to_csv('/kaggle/working/submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T09:37:58.857187Z","iopub.execute_input":"2022-07-08T09:37:58.857575Z","iopub.status.idle":"2022-07-08T09:37:58.866670Z","shell.execute_reply.started":"2022-07-08T09:37:58.857538Z","shell.execute_reply":"2022-07-08T09:37:58.865392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}