{"cells":[{"metadata":{"_uuid":"55a1eba723db84fbd09252d3a864a5babadd8669"},"cell_type":"markdown","source":"# Importing Libraries"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"from __future__ import print_function, division\nfrom builtins import range\n# Note: you may need to update your version of future\n# sudo pip install -U future\n\nimport os\nimport sys\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n\nfrom keras.models import Model\nfrom keras.layers import Dense, Embedding, Input\nfrom keras.layers import LSTM, Bidirectional, GlobalMaxPool1D, Dropout\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.optimizers import Adam\nfrom sklearn.metrics import roc_auc_score\n\nimport keras.backend as K\nif len(K.tensorflow_backend._get_available_gpus()) > 0:\n  from keras.layers import CuDNNLSTM as LSTM\n  from keras.layers import CuDNNGRU as GRU\n\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"# Setting some configuration"},{"metadata":{"trusted":true,"_uuid":"1faf9d84bee5c3c908e6f48dde33a52a53397129"},"cell_type":"code","source":"# configuration setting\nMAX_SEQUENCE_LENGTH = 100\nMAX_VOCAB_SIZE = 20000\nEMBEDDING_DIM = 50\nVALIDATION_SPLIT = 0.2\nBATCH_SIZE = 128\nEPOCHS = 2\n\npath = '../input/'\n\nEMBEDDING_FILE=f'{path}glove6b50d/glove.6B.50d.txt'\ncomp = 'jigsaw-toxic-comment-classification-challenge/'\nTRAIN_DATA_FILE=f'{path}{comp}train.csv'\nTEST_DATA_FILE=f'{path}{comp}test.csv'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1fe07e8b079a2415f4960711b62eb6e3dbd9fb08"},"cell_type":"code","source":"# load in pre-trained word vectors\nprint('Loading word vectors...')\nword2vec = {}\nwith open(EMBEDDING_FILE) as f:\n  # is just a space-separated text file in the format:\n  # word vec[0] vec[1] vec[2] ...\n  for line in f:\n    values = line.split()\n    word = values[0]\n    vec = np.asarray(values[1:], dtype='float32')\n    word2vec[word] = vec\nprint('Found %s word vectors.' % len(word2vec))\n\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"487ce299657048d7e594afe92e412a4fa97eabf2"},"cell_type":"markdown","source":"# Preparing text samples and their labels"},{"metadata":{"trusted":true,"_uuid":"50449ed294a90c895cdd3e68e66aeb2f9c900d85"},"cell_type":"code","source":"\nprint('Loading in comments...')\n\ntrain = pd.read_csv(TRAIN_DATA_FILE)\n\nsentences = train[\"comment_text\"].fillna(\"DUMMY_VALUE\").values\npossible_labels = [\"toxic\", \"severe_toxic\", \"obscene\", \"threat\", \"insult\", \"identity_hate\"]\ntargets = train[possible_labels].values\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c3335f0ba3afc122f489a94ec88ae869229719ad"},"cell_type":"code","source":"# convert the sentences (strings) into integers\ntokenizer = Tokenizer(num_words=MAX_VOCAB_SIZE)\ntokenizer.fit_on_texts(sentences)\nsequences = tokenizer.texts_to_sequences(sentences)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d04ad214d2c3bc02c7234bd7087087e2c1297b9f"},"cell_type":"code","source":"# get word -> integer mapping\nword2idx = tokenizer.word_index\nprint('Found %s unique tokens.' % len(word2idx))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"990b92d0651e481446ea66a7c1059bc60b64ec48"},"cell_type":"code","source":"# pad sequences so that we get a N x T matrix\ndata = pad_sequences(sequences, maxlen=MAX_SEQUENCE_LENGTH)\nprint('Shape of data tensor:', data.shape)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8b91ec334ea5aa785dae9c2a3de07f576b5e9b7d"},"cell_type":"markdown","source":"# Preparation of Embedding Matrix"},{"metadata":{"trusted":true,"_uuid":"2f37c5823cb7ca9344e5b66f13fe0cb4241b252c"},"cell_type":"code","source":"# prepare embedding matrix\nprint('Filling pre-trained embeddings...')\nnum_words = min(MAX_VOCAB_SIZE, len(word2idx) + 1)\nembedding_matrix = np.zeros((num_words, EMBEDDING_DIM))\nfor word, i in word2idx.items():\n  if i < MAX_VOCAB_SIZE:\n    embedding_vector = word2vec.get(word)\n    if embedding_vector is not None:\n      # words not found in embedding index will be all zeros.\n      embedding_matrix[i] = embedding_vector\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ca8f04c034867e96e80dbd19eb101b1ede9f325c"},"cell_type":"code","source":"# load pre-trained word embeddings into an Embedding layer\n# note that we set trainable = False so as to keep the embeddings fixed\nembedding_layer = Embedding(\n  num_words,\n  EMBEDDING_DIM,\n  weights=[embedding_matrix],\n  input_length=MAX_SEQUENCE_LENGTH,\n  trainable=False\n)\n\n","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"37c766dc45c62416b992ded323b0c42ca7fd13fb"},"cell_type":"markdown","source":"# Model building : Bidirectional LSTM"},{"metadata":{"trusted":true,"_uuid":"903fcb3ae022aeb5d6803c9ada4357497b212156"},"cell_type":"code","source":"print('Building model...')\n\n# create an LSTM network with a single LSTM\ninput_ = Input(shape=(MAX_SEQUENCE_LENGTH,))\nx = embedding_layer(input_)\n# x = LSTM(15, return_sequences=True)(x)\nx = Bidirectional(LSTM(15, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\noutput = Dense(len(possible_labels), activation=\"sigmoid\")(x)\n\nmodel = Model(input_, output)\nmodel.compile(\n  loss='binary_crossentropy',\n  optimizer=Adam(lr=0.01),\n  metrics=['accuracy']\n)\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1107d393729a3a22603f0e4fdd007509903c8b27"},"cell_type":"code","source":"print('Training model...')\nr = model.fit(\n  data,\n  targets,\n  batch_size=BATCH_SIZE,\n  epochs=EPOCHS,\n  validation_split=VALIDATION_SPLIT\n)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1525c9d1d0b3b8537a3d60e5b5d32b29c23fc4e3"},"cell_type":"code","source":"# plot some data\nplt.plot(r.history['loss'], label='loss')\nplt.plot(r.history['val_loss'], label='val_loss')\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"51982f97454c07ec660acd943b5b2bd971066738"},"cell_type":"code","source":"# Plotting accuracies\nplt.plot(r.history['acc'], label='acc')\nplt.plot(r.history['val_acc'], label='val_acc')\nplt.legend()\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a066d55712599803c110b9bea4ba6a9f9f09f0bc"},"cell_type":"code","source":"p = model.predict(data)\naucs = []\nfor j in range(6):\n    auc = roc_auc_score(targets[:,j], p[:,j])\n    aucs.append(auc)\nprint(np.mean(aucs))\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a7de947b2d4430a0a94914e2da994c6a992b5bd1"},"cell_type":"code","source":"test = pd.read_csv(TEST_DATA_FILE)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2193a7025663503ecb85a9b2bac3b7b6c78f9f0f"},"cell_type":"code","source":"list_sentences_test = test[\"comment_text\"].fillna(\"DUMMY_VALUE\").values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ec4777482785d4e99ea0e13ed0430d6cddc7e3c6"},"cell_type":"code","source":"list_tokenized_test = tokenizer.texts_to_sequences(list_sentences_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d91ec548e48b303985e90457c363f8c2f7034c05"},"cell_type":"code","source":"X_te = pad_sequences(list_tokenized_test, maxlen=MAX_SEQUENCE_LENGTH)\nprint('Shape of data tensor:', X_te.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2c20134ea3516af58b7ca44eab23c7fe6912ba72"},"cell_type":"code","source":"y_test = model.predict([X_te], batch_size=BATCH_SIZE, verbose=1)\nsubmission = pd.read_csv(f'{path}{comp}sample_submission.csv')\nsubmission[possible_labels] = y_test\nsubmission.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e80d7c8bdec1b60fbc714472171a1f0b5c995466"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}