{"cells":[{"metadata":{"_uuid":"6e7067ffdfb062d93f1ab7c66ce1c44c12afb17d"},"cell_type":"markdown","source":"# CNN-GRU Model\nFor my first submission I will develop a simple CNN-GRU model based on my ongoing learning of model architectures for various text classification problems. \n\nThis notebook will follow a typical deep learning workflow with an initial EDA from which we select important parameters such as max sequence length and max number of words. We then vectorize both training and test data followed by model training and evaluation and possibly model tuning if results are lower than expected. Finally, I post my first submission and review the model for further enhancements."},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"_kg_hide-input":true,"_kg_hide-output":false},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"Custom helper functions for reading in glove embeddings as well as vectorizing training data"},{"metadata":{"trusted":true,"_uuid":"f8266e5be212344b350d28f23a0dcdaa79dfb634","_kg_hide-input":true,"_kg_hide-output":true},"cell_type":"code","source":"# functions for reading in embedding data and\n# tokenizing and processing sequences with padding and\n# function for plotting model accuracy and loss\n# modify line.split to line.split(\" \") as 300D contains spaces\n\nimport os\nimport numpy as np\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\n# vectorizer and sequence function\n# takes in raw text and labels\n# params for max sequence length and max words\n# default arg for Shuffle=True to randomise data\n# returns tokenizer object. x_train,y_train, x_val,y_val\ndef tokenize_and_sequence(texts, labels, max_len, max_words, validation_samples, shuffle=True):\n    #initialise tokenizer with num_words param\n    tokenizer = Tokenizer(num_words=max_words)\n    tokenizer.fit_on_texts(texts)\n    # convert texts to sequences\n    sequences = tokenizer.texts_to_sequences(texts)\n    # generate work index\n    word_index = tokenizer.word_index\n    # print top words count\n    print('{} of unique tokens found'.format(len(word_index)))\n    # pad sequences using max_len param\n    data = pad_sequences(sequences, maxlen=max_len)\n    # convert list of labels into numpy array\n    labels = np.asarray(labels)\n    # print shape of text and label tensors\n    print('data tensor shape: {}\\nlabel tensor shape:{}'.format(data.shape, labels.shape))\n\n    # shuffle data=True as labels are ordered\n    # randomise data to vary class distribution\n    if shuffle:\n        # get length of data sequence and create array\n        indices = np.arange(data.shape[0])\n        np.random.shuffle(indices)\n        # shuffle data and labels\n        data = data[indices]\n        labels = labels[indices]\n    else:\n        pass\n\n    # split training data into training and validation splits\n    # split using validation length\n    # validation split\n    x_val = data[:validation_samples]\n    y_val = labels[:validation_samples]\n    # training split\n    x_train = data[validation_samples:]\n    y_train = labels[validation_samples:]\n\n    # return tokenizer, word_index, training and validation data\n    return tokenizer, word_index, x_train, y_train, x_val, y_val\n\n\n# function to lpad pretrained glove embeddings\n# takes in embedding dim for variable embedding sizes\n# and base directory as well as txt file\n# embedding dim should match the file name dimension\n# and max words and word_index for embedding features\ndef load_glove(base_directory, f_name, max_words, word_index, embedding_dim=None):\n    # check file name ends in .txt\n    # read file name embedding value if not specified\n    if f_name[-4:] == '.txt':\n        # check embedding value\n        if embedding_dim is not None:\n            dim = f_name[-8:-5]\n            dim = int(dim)\n            embedding_dim = dim\n        else:\n            # assuming dimension is not none for manual input\n            pass\n        # continue\n\n        # create embedding dictionary\n        embeddings_index = {}\n        # open embeddings file\n        try:\n            f = open(os.path.join(base_directory, f_name))\n            # iterate over lines and split on individual words\n            # split coefficient of word values\n            # map words and coefficients to embeddings dictionary\n            for line in f:\n                values = line.split(\" \") # returns list of [word, coeff]\n                word = values[0] # gets first list element\n                coeff = np.asarray(values[1:], dtype='float32')  # slice coefficiennt value array from remainder of list\n                # assign mapping to dictionary\n                embeddings_index[word] = coeff\n            f.close()\n        except IOError:\n            print('cannot read file. check file paths')\n\n        # prepare glove word-embedding matrix\n        # create empty embedding tensor\n        embedding_matrix = np.zeros((max_words,embedding_dim ))\n        # map the top words of the data into the glove embedding matrix\n        # words not found from the data in glove will be zeroed\n        for word, i in word_index.items():\n            if i < max_words:\n                embedding_vector = embeddings_index.get(word)\n                if embedding_vector is not None:\n                    embedding_matrix[i] = embedding_vector\n\n        # return embedding matrix\n        return embedding_matrix\n\n\n# function to visualise keras model history metrics\n# function takes in acc, val_acc, loss, val_loss for model params\n# range is defined by epochs in range len(acc)\n\nimport matplotlib.pyplot as plt\n\ndef plot_training_and_validation(acc, val_acc, loss, val_loss):\n    epochs = range(1, len(acc) + 1)\n    plt.plot(epochs, acc, 'bo', label='Training acc')\n    plt.plot(epochs, val_acc, 'b', label='Validation acc')\n    plt.title('Training and validation accuracy')\n    plt.legend()\n    plt.figure()\n    plt.plot(epochs, loss, 'bo', label='Training loss')\n    plt.plot(epochs, val_loss, 'b', label='Validation loss')\n    plt.title('Training and validation loss')\n    plt.legend()\n    plt.show()\n\n# end","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d73c5e5a42d1bd9f434c62753d0edda0c4578cac"},"cell_type":"markdown","source":"# 1 - Data exploration\nLoading data and simple EDA"},{"metadata":{"_uuid":"25e7d654d528eaafd57ab37d6a8077b15e764af3"},"cell_type":"markdown","source":"# 1.1 - Loading in Training and Test data\nSet directory paths for data as well as GloVe embeddings"},{"metadata":{"trusted":true,"_uuid":"ed2bc6614340d3820f8fa156947155d21eaf2df9"},"cell_type":"code","source":"base_dir ='../input'\nprint(base_dir)\n# list files in current directory\nprint(os.listdir(base_dir))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1105b9a3e20d4f5fbf107cd94fc95070848da06d"},"cell_type":"code","source":"# set train and test data set paths\ntrain_path = os.path.join(base_dir, 'train.csv')\ntest_path = os.path.join(base_dir, 'test.csv')\nprint(train_path, test_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3f9af5a442f829ef7fb75c649d72f2556678bd11"},"cell_type":"code","source":"# set embedding file path\nprint(os.listdir(os.path.join(base_dir, 'embeddings')))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c28cef327d39f89d809f4b8517b962813e6d3bf1"},"cell_type":"code","source":"glove_file = 'glove.840B.300d.txt'\nbase_embedding_dir = os.path.join(base_dir, 'embeddings/glove.840B.300d')\nprint(os.path.join(base_embedding_dir, glove_file))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bddde659ed7e960941a9718214281ae099a460a5"},"cell_type":"code","source":"# load data into DataFrames\ntrain_df = pd.read_csv(train_path)\ntest_df = pd.read_csv(test_path)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4ca22a0c4d6fc08a3641e46ccc77f63ceb754dd6"},"cell_type":"code","source":"# verify and inspect DataFrames\ntrain_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a030f1461da1e64b12eef880133ac0e61011a8f"},"cell_type":"code","source":"test_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c5d3c574bdfe852e349a8905c1b165dda346892a"},"cell_type":"code","source":"# simple statistics of training data\ntrain_df.info()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5ae0e4c2e5d6c6a4866bc3051f8ed2412c3e0159"},"cell_type":"code","source":"train_df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b63e0b5aa1ea2a984e8d20827d082fbaea4d8fe1"},"cell_type":"code","source":"# check for null values in labels in training data\ntrain_df.isnull().any()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fbec88cba9e40a664266bb29aa7fddd775ff3391"},"cell_type":"code","source":"# define variable to index data frame question column\nquestions = 'question_text'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e5abf8b24695883b97f924f936ba07605cbe33ac"},"cell_type":"code","source":"# find maximum sequence length of questions\nnp.max(train_df[questions].apply(lambda x: len(x.split())))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e671cc4492b16bbe429fa94338339512b2c25a5f"},"cell_type":"markdown","source":"A max sequence length of 134 means a good max_len parameter would be 100 as in previous experiments that establish a good baseline"},{"metadata":{"_uuid":"7bdf71ebde714109189ebc86d252673b0cde53e3"},"cell_type":"markdown","source":"# 1.2 - Split text and labels ready for vectorization\nSplit training text and labels into lists, split test text into a list"},{"metadata":{"trusted":true,"_uuid":"45d7cc8d565de56c9ff2fc78d6704ea17bacc7c1"},"cell_type":"code","source":"# split labels from training data into numpy array\ny_train = train_df['target'].values\ny_train = np.asarray(y_train)\nprint(type(y_train))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d7b2e76d934e5c75200072884ef4bcba78ec3051"},"cell_type":"code","source":"# verify label array shape\ny_train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b77437381b0aafb6ea606f87716ee8f5437eedc1"},"cell_type":"code","source":"# inspect label\ny_train[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aec9b2139f643bc5caadc59d7a0df1ff5792a4a4"},"cell_type":"code","source":"# extract questions from Series objects of train and test data\nquestions_train = train_df[questions]\nquestions_test = test_df[questions]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6e7fc4dec176b36f7af6d5b7d534056a6481ed77"},"cell_type":"code","source":"# transforms Series into lists\nx_train = list(questions_train)\nx_test = list(questions_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"282e135ffb86680fc8a60a97fa5429e3e35bd754"},"cell_type":"code","source":"# verify and inspect data\nprint(x_train[0], y_train[0], type(x_train), len(x_train))\nprint(x_test[0], len(x_test))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0acce5fde736b1816cf9781197aecb212fe1781a"},"cell_type":"markdown","source":"# 1.3 - Vectorization\nTokenize and pad text sequences"},{"metadata":{"trusted":true,"_uuid":"c310affb7ba8ead3e4aaeebf782f889c54d4160f"},"cell_type":"code","source":"# define a 10% validation split from training data\nvalidation_samples = int((len(x_train) // 10))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"197948325dcf2665fe74c826ac2d92de34a47bd6"},"cell_type":"code","source":"# verify 90:10 split\nprint(len(x_train), validation_samples)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"acab5aeff5f9aa30ad9007bcc4ae83f6cdfd976f"},"cell_type":"code","source":"# define max sequence length and total dictionary words\nmax_len = 100\nmax_words = 10000","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bd06e12b9183486492808f4b97d005f3b144fc78"},"cell_type":"code","source":"# Vectorize training data and return tokenizer and word_index as well as validation splits\ntokenizer, word_index, X_train, Y_train, x_val, y_val = tokenize_and_sequence(\n    x_train, y_train, max_len=max_len, max_words=max_words, validation_samples=validation_samples, shuffle=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7968e2d4fb511d6a9fcfa2665086cd5a344d4d2d"},"cell_type":"code","source":"# verify train and validation text and labels\nprint('training:',X_train.shape, Y_train.shape, '\\nvalidation:', x_val.shape, y_val.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ea043bc0044eb8619989210393ccbe6580f1610d"},"cell_type":"markdown","source":"# 1.4 - Load Glove Embeddings\nLoad 300D pre-trained embeddings"},{"metadata":{"trusted":true,"_uuid":"ae3843cc91d86fdf9aae9d7a7b598f5fcb49f06c"},"cell_type":"code","source":"# define embedding dimension\nembedding_dim = 300","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8e6eb92a4af4bbcca3f7986fbd5ab0399fdacb1b"},"cell_type":"code","source":"# load in glove embedding using custom function from earlier\n# function takes as input the raw file, word_index returned from the tokenizer and max_words\nglove_embedding_300d = load_glove('../input/embeddings/glove.840B.300d/', glove_file, max_words=max_words, word_index=word_index, embedding_dim=embedding_dim)\n# *error in loading* needs investigation \n# * 300D needs line.split(' ') compared to smaller dimensions","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c22f16862d60d733195cbd09c72d03781f067bc3"},"cell_type":"code","source":"# verify embeddings loaded correctly\nglove_embedding_300d.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a21934806c91cbda9b4363932d5419070d288e39"},"cell_type":"markdown","source":"# 2 - Model Architecture"},{"metadata":{"_uuid":"5a43ca499485af702d41fe4ebf21d7a9ac617019"},"cell_type":"markdown","source":"# 2.1 CNN-GRU Baseline\nDesign a simple model using a 1D convolution layer followed by a GRU layer to establish a benchmark performance score for further improvements.\n\nUsing the pre-trained embeddings with weights frozen to prevent re-training of word vectors during model training.\n\nTraining for 5 epochs on batch sizes of 128\n\nenhancements -\nUsing a CuDNNGRU layer for GPU acceleration as well as SpatialDropout1D of 0.2 for the convolution layer\n\n## Design\nA sequential model with the following layers:\n\nEmbedding(dimension=200)\nConv1D(64, 3, 'relu') 64 convolutions with a kernel size of 3, can be extended up to 7\nSpatialDropout1D(0.2)\nMaxPooling1D(4) standard practive following convolutions\nGRU(64, dropout=0.1, recurrent_dropout=0.5) *using a layer dropout of 10% and a recurent unit dropout of 50%, as seen in research to return good performance*\nDense(1, activation='sigmoid') dense classifier layer of six outputs"},{"metadata":{"trusted":true,"_uuid":"bbdeef62fd81fa1e32ba482b63e0936f80aa2e50"},"cell_type":"code","source":"# import layers\nfrom keras.layers import Input, Embedding, GRU, LSTM, MaxPooling1D, GlobalMaxPool1D, CuDNNGRU\nfrom keras.layers import Dropout, Dense, Activation, Flatten, Conv1D, SpatialDropout1D\nfrom keras.models import Sequential","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"eff8eb4972511cd2b87c1e8788ec3529558883db"},"cell_type":"code","source":"# import AUC ROC metrics from sklearn\nfrom sklearn.metrics import roc_auc_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"55bf0626eb2688483173efd41ee32b172b4630d3"},"cell_type":"code","source":"# define model architecture\nmodel = Sequential()\nmodel.add(Embedding(max_words, embedding_dim, input_length=max_len))\nmodel.add(Conv1D(64, 3, activation='relu'))\nmodel.add(SpatialDropout1D(0.2))\nmodel.add(MaxPooling1D(4))\nmodel.add(CuDNNGRU(64))\nmodel.add(Dropout(0.1))\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f1775fcb01481b94aa783dae7718c80c511f1e42"},"cell_type":"code","source":"# load pre-trained Glove embeddings in the first layer\nmodel.layers[0].set_weights([glove_embedding_300d])\n# freeze embedding layer weights\nmodel.layers[0].trainable = False\n# compile model with adam optimizer\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6cba50273a8e6dc4bbc9ae70422fc1062e446275"},"cell_type":"code","source":"# fit model and train on training data and validate on validation samples\n# train for 5 epochs to establish baseline overfitting model\n# saves results to histroy object\nhistory = model.fit(X_train, Y_train, epochs=5, batch_size=128, validation_data=(x_val, y_val))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e2071e96af63dc3201acb404cc822bd66a00df5b"},"cell_type":"code","source":"# save model\nmodel.save('cnn_cudnngru_300d.h5')","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"63a4a5e8cc7d25df64324666f5fbe4a7aa7a0fe1"},"cell_type":"markdown","source":"# Model Results\nPlot model training and validation performance"},{"metadata":{"trusted":true,"_uuid":"faf711d789b2c06f8a4c3bb58358a782f2784d9d"},"cell_type":"code","source":"# define plotting metrics\nacc = history.history['acc']\nval_acc = history.history['val_acc']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n# plot model training and validation accuracy and loss\nplot_training_and_validation(acc, val_acc, loss, val_loss)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"edddb987739b7c5fd4269404df57f0a87babb0b5"},"cell_type":"code","source":"# further enhancements to be made and model checkpointing\n# plots show model performs the best at epoch 3","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c55602133a49de2ffb86244463f19d966b373644"},"cell_type":"markdown","source":"Let's use the AUC ROC score metrics for a slightly better understanding of model performance.\n\nThe training and validation performance remains consistent but let's review it on the validation data and plot the AUC graph for a better representation of model generalization"},{"metadata":{"trusted":true,"_uuid":"618d788e49edd41f1b4e5824223cfff39fac24be"},"cell_type":"code","source":"y_hat = model.predict(x_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1c20c0744d7ea66104da91e54a8aadbdb1cd0182"},"cell_type":"code","source":"# print auc roc score\n\"{:0.2f}\".format(roc_auc_score(y_val, y_hat)*100.0)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"3b8da4c314bb1025490b398842066883c941ab8f"},"cell_type":"markdown","source":"94% under the curve is not a bad starting point. \n\nDefinitely better scores to gain using advanced model tuning and hyperparameters."},{"metadata":{"_uuid":"94bc48618dd31dfc5d2e37de15cb1b44bf811ff2"},"cell_type":"markdown","source":"# 2.2 Evaluate on Test set and submit\nLet's evaluate the model on the test set, round the predicted values as per requirements of the competition, and submit the csv"},{"metadata":{"trusted":true,"_uuid":"323b73570268d4f01fc7049e91ae8106f09ce980"},"cell_type":"code","source":"# we first need to tokenize and pad the raw text from the test data\nsequences_test = tokenizer.texts_to_sequences(x_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6a47f1d402c8c7a0c46273713d85024f1e248d40"},"cell_type":"code","source":"test_data = pad_sequences(sequences_test, maxlen=max_len)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f827150d3c495a1b750a227c1c932e7ef1890848"},"cell_type":"code","source":"test_data.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"46dd2491822c9d17f194e060a9b5a782f5d0552d"},"cell_type":"code","source":"# verify test data sample\ntest_data[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"157e4b66eb1a662b33f20233d323f839e65fa9aa"},"cell_type":"code","source":"# iterate over test sequences and predict y_hat values\ny_hat = model.predict(test_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"418c8a6305db478f9f3235c08771998577121ea7"},"cell_type":"code","source":"y_hat.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc1da5f1d76d1a8e540948cf8a133af3d1c8b13d"},"cell_type":"code","source":"y_hat","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"94c509bcd3e136390920fc01783f807532131d2e"},"cell_type":"code","source":"predictions = (np.array(y_hat) > 0.5).astype(np.int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"df83ca346671a0c2b0a7417e644ae04ae645c515"},"cell_type":"code","source":"# create dataframe for precitions\nsubmit_df = pd.DataFrame({\"qid\": test_df[\"qid\"], \"prediction\": predictions.flatten()})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ae0b2ce0b7a5a7e448b1c69f3821beefc0f3844c"},"cell_type":"code","source":"submit_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d6595e4fc99028fa2d67fd9e9e307dae7b0eff12"},"cell_type":"code","source":"# save as csv for submission\n# submit_df.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e450d57178b4101b479d2a746672971506ffaf10"},"cell_type":"markdown","source":"# 3. Baseline ++"},{"metadata":{"_uuid":"5e606e45862d90facc7c8cf926e151613f97e63b"},"cell_type":"markdown","source":"# Improving upon the baseline model\nIn this experiment we increase the model complexity while modifying the vecotrization process slightly by fitting on test and train texts. \n\nWe also implement callbacks, dropout, batch normalisation as well as early stopping."},{"metadata":{"_uuid":"1c137b01d8704c21ed084eab0b5965bf215240a4"},"cell_type":"markdown","source":"## 3.1 Redefine custom helper functions"},{"metadata":{"trusted":true,"_uuid":"5e96dd43d8a0c3f6bae682d8568c050c32b23290"},"cell_type":"code","source":"# functions for reading in embedding data and\n# tokenizing and processing sequences with padding and\n# function for plotting model accuracy and loss\n# modify line.split to line.split(\" \") as 300D contains spaces\n\nimport os\nimport numpy as np\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\n# vectorizer and sequence function\n# takes in raw text and labels\n# params for max sequence length and max words\n# default arg for Shuffle=True to randomise data\n# returns tokenizer object. x_train,y_train, x_val,y_val\ndef tokenize_and_sequence(full_texts, texts, labels, max_len, max_words, validation_samples, shuffle=True):\n    #initialise tokenizer with num_words param\n    tokenizer = Tokenizer(num_words=max_words)\n    tokenizer.fit_on_texts(full_texts)\n    # convert texts to sequences\n    sequences = tokenizer.texts_to_sequences(texts)\n    # generate work index\n    word_index = tokenizer.word_index\n    # print top words count\n    print('{} of unique tokens found'.format(len(word_index)))\n    # pad sequences using max_len param\n    data = pad_sequences(sequences, maxlen=max_len)\n    # convert list of labels into numpy array\n    labels = np.asarray(labels)\n    # print shape of text and label tensors\n    print('data tensor shape: {}\\nlabel tensor shape:{}'.format(data.shape, labels.shape))\n\n    # shuffle data=True as labels are ordered\n    # randomise data to vary class distribution\n    if shuffle:\n        # get length of data sequence and create array\n        indices = np.arange(data.shape[0])\n        np.random.shuffle(indices)\n        # shuffle data and labels\n        data = data[indices]\n        labels = labels[indices]\n    else:\n        pass\n\n    # split training data into training and validation splits\n    # split using validation length\n    # validation split\n    x_val = data[:validation_samples]\n    y_val = labels[:validation_samples]\n    # training split\n    x_train = data[validation_samples:]\n    y_train = labels[validation_samples:]\n\n    # return tokenizer, word_index, training and validation data\n    return tokenizer, word_index, x_train, y_train, x_val, y_val\n\n\n# function to lpad pretrained glove embeddings\n# takes in embedding dim for variable embedding sizes\n# and base directory as well as txt file\n# embedding dim should match the file name dimension\n# and max words and word_index for embedding features\ndef load_glove(base_directory, f_name, max_words, word_index, embedding_dim=None):\n    # check file name ends in .txt\n    # read file name embedding value if not specified\n    if f_name[-4:] == '.txt':\n        # check embedding value\n        if embedding_dim is not None:\n            dim = f_name[-8:-5]\n            dim = int(dim)\n            embedding_dim = dim\n        else:\n            # assuming dimension is not none for manual input\n            pass\n        # continue\n\n        # create embedding dictionary\n        embeddings_index = {}\n        # open embeddings file\n        try:\n            f = open(os.path.join(base_directory, f_name))\n            # iterate over lines and split on individual words\n            # split coefficient of word values\n            # map words and coefficients to embeddings dictionary\n            for line in f:\n                values = line.split(\" \") # returns list of [word, coeff]\n                word = values[0] # gets first list element\n                coeff = np.asarray(values[1:], dtype='float32')  # slice coefficiennt value array from remainder of list\n                # assign mapping to dictionary\n                embeddings_index[word] = coeff\n            f.close()\n        except IOError:\n            print('cannot read file. check file paths')\n\n        # prepare glove word-embedding matrix\n        # create empty embedding tensor\n        embedding_matrix = np.zeros((max_words,embedding_dim ))\n        # map the top words of the data into the glove embedding matrix\n        # words not found from the data in glove will be zeroed\n        for word, i in word_index.items():\n            if i < max_words:\n                embedding_vector = embeddings_index.get(word)\n                if embedding_vector is not None:\n                    embedding_matrix[i] = embedding_vector\n\n        # return embedding matrix\n        return embedding_matrix\n\n\n# function to visualise keras model history metrics\n# function takes in acc, val_acc, loss, val_loss for model params\n# range is defined by epochs in range len(acc)\n\nimport matplotlib.pyplot as plt\n\ndef plot_training_and_validation(acc, val_acc, loss, val_loss):\n    epochs = range(1, len(acc) + 1)\n    plt.plot(epochs, acc, 'bo', label='Training acc')\n    plt.plot(epochs, val_acc, 'b', label='Validation acc')\n    plt.title('Training and validation accuracy')\n    plt.legend()\n    plt.figure()\n    plt.plot(epochs, loss, 'bo', label='Training loss')\n    plt.plot(epochs, val_loss, 'b', label='Validation loss')\n    plt.title('Training and validation loss')\n    plt.legend()\n    plt.show()\n\n# end","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"c03937029f39a8e1ae2309764fa6364040e7a63d"},"cell_type":"markdown","source":"## 3.2 Vectorize data\nWe fit our tokenizer on both the training and test sets to capture as many words as possible when passing to the embeddings"},{"metadata":{"trusted":true,"_uuid":"611536d8438e4680df0e1c3730d515b7cbf0a475"},"cell_type":"code","source":"full_texts = x_train + x_test","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1be5c1ee7d921a55098d53d815bbccff72739f82"},"cell_type":"code","source":"max_len = 100\nmax_words = 10000","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0f9d71c7adf3443dbf25fbc74873e1ee8423131f"},"cell_type":"code","source":"# vectorize training data\n# Vectorize training data and return tokenizer and word_index as well as validation splits\ntokenizer, word_index, X_train, Y_train, x_val, y_val = tokenize_and_sequence(\n    full_texts, x_train, y_train, max_len=max_len, max_words=max_words, validation_samples=validation_samples, shuffle=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a7517a6be217a3376ca25b99d49c0e9fe48b6a24"},"cell_type":"code","source":"# verify train and validation text and labels\nprint('training:',X_train.shape, Y_train.shape, '\\nvalidation:', x_val.shape, y_val.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e38849874c5b33d5d5e5c1ddf2f022007c1aa6f4"},"cell_type":"markdown","source":"## 3.3 Load GloVe embeddings"},{"metadata":{"trusted":true,"_uuid":"4f8415e09e132e04710680b88c9f70b5569af00b"},"cell_type":"code","source":"# define embedding dimension\nembedding_dim = 300","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"46eba248b30640e36b8ea7dad360271b43bb2063"},"cell_type":"code","source":"# load in glove embedding using custom function from earlier\n# function takes as input the raw file, word_index returned from the tokenizer and max_words\nglove_embedding_300d = load_glove('../input/embeddings/glove.840B.300d/', glove_file, max_words=max_words, word_index=word_index, embedding_dim=embedding_dim)\n# * 300D needs line.split(' ') compared to smaller dimensions","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"39732909a3301e7b8954097e7e9c4dea69226a25"},"cell_type":"code","source":"# verify embeddings loaded correctly\nglove_embedding_300d.shape","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d60262b936bc554c7580280e9e12a7c14e1eab0a"},"cell_type":"markdown","source":"## 3.4 Model Architecture\n## Baseline++\n\nWe take the same core design of the baseline model of a CNN-GRU but add features such as Dropout, Batch normalization.\n\nWe optimise the model with callbacks and monitor progress to compare if the modifications in input sequence length and size improves our results."},{"metadata":{"trusted":true,"_uuid":"0a0ba208d4621c99f3f8858f85351a7f52a53abf"},"cell_type":"code","source":"# import keras layers\nimport keras","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"329077189fd1bf51936f3cc39b78fa134b73069d"},"cell_type":"code","source":"# define custom ROC callback\n# import AUC ROC metrics from sklearn\nfrom sklearn.metrics import roc_auc_score\n\n# define class for ROC AUC callback with simple name modifications\n# credit to https://www.kaggle.com/yekenot\nclass roc_auc_validation(keras.callbacks.Callback):\n    def __init__(self, validation_data=(), interval=1):\n        super(Callback, self).__init__()\n        self.interval = interval\n        self.x_val, self.y_val = validation_data\n\n    def on_epoch_end(self, epoch, logs={}):\n        if epoch % self.interval == 0:\n            y_pred = self.model.predict(self.x_val, verbose=0)\n            score = roc_auc_score(self.y_val, y_pred)\n            print(\"\\n ROC-AUC - epoch: {:d} - score: {:.6f}\".format(epoch+1, score))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1a4c8fc6ed0fd3ab09fb3619ba17816a6cd531e1"},"cell_type":"code","source":"# import keras layers \nfrom keras.layers import Input, Embedding, GRU, LSTM, MaxPooling1D, GlobalMaxPool1D, CuDNNGRU, CuDNNLSTM\nfrom keras.layers import Dropout, Dense, Activation, Flatten,Conv1D, Bidirectional, SpatialDropout1D, BatchNormalization\nfrom keras.models import Sequential\nfrom keras.optimizers import RMSprop ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7dd9bd1c211540ce741b6d0747231c5d7bc9da88"},"cell_type":"code","source":"# define model architecture\nmodel = Sequential()\nmodel.add(Embedding(max_words, embedding_dim, input_length=max_len))\nmodel.add(SpatialDropout1D(0.2)) # add spatial dropout\nmodel.add(Conv1D(64, 5, activation='relu')) # increase kernel size to 5\nmodel.add(MaxPooling1D(4))\nmodel.add(BatchNormalization()) # add batch normalization\nmodel.add(Dropout(0.1))\n# modify to CuDNNGRU\n#model.add(GRU(64, dropout=0.1, recurrent_dropout=0.5)) # defaults inclide tanh activation\nmodel.add(CuDNNGRU(64)) # does not have a dropout or recurrent dropout param\nmodel.add(BatchNormalization())\nmodel.add(Dropout(0.1))\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9af0bde2970c039764551cedeefccc2ade4fd9b2"},"cell_type":"code","source":"# define callbacks\nfrom keras.callbacks import Callback, EarlyStopping, ReduceLROnPlateau","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a0e9efddf671809ce5d0c76822af514e15fb6b94"},"cell_type":"code","source":"# initialise customer roc callback\nroc_callback = roc_auc_validation(validation_data=(x_val, y_val), interval=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"32e901470785ff604179013eb501a71031d5d17f"},"cell_type":"code","source":"# define early stopping and reduce lr callbacks\ncallback_list = [keras.callbacks.EarlyStopping(monitor='acc', patience=1),\n                 keras.callbacks.ModelCheckpoint(filepath='baseline_plus_.h5', monitor='val_loss',\n                                                 save_best_only=True)]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e7c33ae9e6e31f7ca71126ee76bf52bed977fd91"},"cell_type":"code","source":"# add roc to callbacks list\ncallback_list.append(roc_callback)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"201a52febdd131d684cd463c81747e739d608ed0"},"cell_type":"code","source":"callback_list","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b162c18e3e7f25f34a65e62bf2434bf0fca4ffbd"},"cell_type":"code","source":"# load pre-trained Glove embeddings in the first layer\nmodel.layers[0].set_weights([glove_embedding_300d])\n# freeze embedding layer weights\nmodel.layers[0].trainable = False\n# compile model with adam optimizer\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d5b0a5460574ac80516405139352980577f28a5f"},"cell_type":"code","source":"# fit model and train on training data and validate on validation samples\n# train for 5 epochs to establish baseline overfitting model\n# saves results to histroy object\nhistory = model.fit(X_train, Y_train, epochs=20, batch_size=512, callbacks=callback_list,validation_data=(x_val, y_val))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"fd3578d317485c4c997fb9f0ae92203e1472268b"},"cell_type":"markdown","source":"## 3.5 Evaluate and submit test results"},{"metadata":{"trusted":true,"_uuid":"23e07f5384a5a71c4f1b4b89f501cceb6019405d"},"cell_type":"code","source":"# evaluate model on test set and submit results\n# we first need to tokenize and pad the raw text from the test data\nsequences_test = tokenizer.texts_to_sequences(x_test)\ntest_data = pad_sequences(sequences_test, maxlen=max_len)\n# iterate over test sequences and predict y_hat values\ny_hat = model.predict(test_data)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1ab0f0f45e68e53e3a956be793cd5afcbf2b6d19"},"cell_type":"code","source":"predictions = (np.array(y_hat) > 0.5).astype(np.int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a27dfcde33f046befac3ae200cec3e7e44a83e5d"},"cell_type":"code","source":"# create dataframe for precitions\nsubmit_df = pd.DataFrame({\"qid\": test_df[\"qid\"], \"prediction\": predictions.flatten()})","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2a50ad75ddeb3d8c609729d2e55886b0371a4780"},"cell_type":"code","source":"submit_df.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5efcd3eb298344f40671bee88976bf99bc854cf3"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}