{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\nX_train = train_df[\"question_text\"].fillna(\"Is it unethical to sabotage a public kernel?\").values\ntest_df = pd.read_csv(\"../input/test.csv\")\nX_test = test_df[\"question_text\"].fillna(\"Is it unethical to sabotage a public kernel?\").values\ny = train_df[\"target\"]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"137a416d36f56aba95bc0ebddfe279ceb99a79b5"},"cell_type":"markdown","source":"Lets look at some of these insightful questions."},{"metadata":{"trusted":true,"_uuid":"072b89a556f3991368d675d328e6cace6aac3b24"},"cell_type":"code","source":"\ntrain_df[train_df[\"target\"] != 0][[\"question_text\", \"target\"]]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1d603ea06829973bb81e49f1e42fc8cd3304fb42"},"cell_type":"markdown","source":"Let's load in some embeddings and run a quick model"},{"metadata":{"trusted":true,"_uuid":"38a308e5661046f97af541061420bb29f0d5eadb"},"cell_type":"code","source":"\nfrom keras.models import Model\nfrom keras.layers import Input, Dense, Embedding, concatenate\nfrom keras.layers import CuDNNGRU, Bidirectional, GlobalAveragePooling1D, GlobalMaxPooling1D\nfrom keras.preprocessing import text, sequence","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c2fe4020ebc3ddf16a3c2f1c53eb67d376f25561"},"cell_type":"code","source":"maxlen = 10\nmax_features = 30000\n\ntokenizer = text.Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(X_train) + list(X_test))\nword_index = tokenizer.word_index\nX_train = tokenizer.texts_to_sequences(X_train)\nX_test = tokenizer.texts_to_sequences(X_test)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e9f131a4332f5dc140d88acc2c13bf422b56d974"},"cell_type":"code","source":"train_df[\"seq\"] = sequence.pad_sequences(X_train, maxlen=maxlen).tolist()\ntest_df[\"seq\"] = sequence.pad_sequences(X_test, maxlen=maxlen).tolist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"089d91cf40f9ec8d98f9a8d8d3cfc01f88672a89"},"cell_type":"code","source":"import os\nimport zipfile\nimport numpy as np\nembeddings_index = {}\nGLOVE_DIR = \"../input/embeddings/glove.840B.300d/glove.840B.300d.txt\"\nprint(os.listdir(\"../input/embeddings/glove.840B.300d\"))\n\nf = open(GLOVE_DIR)\nfor line in f:\n    values = line.split()\n    word = values[0]\n    #print(values)\n    try:\n        coefs = np.asarray(values[1:], dtype='float32')\n    except:\n        pass\n    embeddings_index[word] = coefs\nf.close()\n\nprint('Found %s word vectors.' % len(embeddings_index))\n\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1f835dc0bef876849b03fdc53cbabc48db199d86"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f3cba3d8bb2a77e7b52381c328678a1b3cfb23e0"},"cell_type":"code","source":"embedding_matrix = np.zeros((len(word_index) + 1, 300))\nfor word, i in word_index.items():\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None:\n        # words not found in embedding index will be all-zeros.\n        embedding_matrix[i] = embedding_vector\n        \nfrom keras.layers import Embedding","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b64a206168170d83b0736f7c5a0576ef514121c6"},"cell_type":"code","source":"embedding_layer = Embedding(len(word_index) + 1,\n                            300,\n                            weights=[embedding_matrix],\n                            input_length=maxlen,\n                            trainable=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d702fb1a3ae2b88745f178a2753c60f43ecd32ce"},"cell_type":"code","source":"from keras.layers import concatenate, Flatten, Lambda, Permute, Reshape, merge\ndef get_model():\n    inp = Input(shape=(maxlen, ))\n    x = embedding_layer(inp)\n    x = Bidirectional(CuDNNGRU(10, return_sequences=True))(x)\n    x = Bidirectional(CuDNNGRU(10, return_sequences=True))(x)\n    avg_pool = GlobalAveragePooling1D()(x)\n    max_pool = GlobalMaxPooling1D()(x)\n    conc = concatenate([\n                        avg_pool, \n                        max_pool])\n\n    outp = Dense(1, activation=\"sigmoid\")(conc)\n    \n    model = Model(inputs=[inp], outputs=outp)\n    model.compile(loss='binary_crossentropy',\n                  optimizer='adam',\n                  metrics=['accuracy'])\n\n    return model\n\nmodel = get_model()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fec390becf54d8d33f71ca13d613696980a0776e"},"cell_type":"code","source":"batch_size = 512\nepochs = 8","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c6dc865296d4f0254fc04aefe07176d7917f3397"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_tra, X_val, y_tra, y_val = train_test_split(train_df, y, test_size = 0.05, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2139b1c0f9884f05b7b2d946a87df0d966f7ff20"},"cell_type":"code","source":"from keras import backend as K\n\ndef f1(y_true, y_pred):\n    def recall(y_true, y_pred):\n        \"\"\"Recall metric.\n\n        Only computes a batch-wise average of recall.\n\n        Computes the recall, a metric for multi-label classification of\n        how many relevant items are selected.\n        \"\"\"\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n        recall = true_positives / (possible_positives + K.epsilon())\n        return recall\n\n    def precision(y_true, y_pred):\n        \"\"\"Precision metric.\n\n        Only computes a batch-wise average of precision.\n\n        Computes the precision, a metric for multi-label classification of\n        how many selected items are relevant.\n        \"\"\"\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n        precision = true_positives / (predicted_positives + K.epsilon())\n        return precision\n    precision = precision(y_true, y_pred)\n    recall = recall(y_true, y_pred)\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))\n\n\nmodel.compile(loss='binary_crossentropy',\n          optimizer= \"adam\",\n          metrics=[\"acc\", f1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"64ddd381a691539935733944cb9935566dccf31f"},"cell_type":"code","source":"from keras.callbacks import ModelCheckpoint, LearningRateScheduler, EarlyStopping, ReduceLROnPlateau\ncheckpoint = ModelCheckpoint('gru.h5', monitor='val_loss', verbose=1, \n                             save_best_only=True, mode='min', save_weights_only = False)\nreduceLROnPlat = ReduceLROnPlateau(monitor='val_loss', factor=0.5, patience=3, \n                                   verbose=1, mode='min', epsilon=0.0001)\nearly = EarlyStopping(monitor='val_loss', \n                      mode=\"min\", \n                      patience=10)\ncallbacks_list = [checkpoint, early, reduceLROnPlat]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0b112231e50369d9e45a687d8e985be999adcb04","scrolled":false},"cell_type":"code","source":"#with self train\nfrom keras.models import load_model\n\nhist = model.fit([np.array(X_tra[\"seq\"].tolist())], y_tra, batch_size=batch_size, epochs=epochs,\n                  validation_data=([np.array(X_val[\"seq\"].tolist())], y_val),\n                  verbose=True, callbacks = callbacks_list)\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d24b2c7d1c82c5f7385cb97f0849f7a783c66996"},"cell_type":"code","source":"model = load_model('gru.h5', custom_objects={'f1': f1})\n\nval_pred1 = model.predict([np.array(X_val[\"seq\"].tolist())], batch_size=128)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bd39657a510c2a51e4512a35cfa4b27c77354467"},"cell_type":"code","source":"positives = y_val[y_val > 0]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"5e5a7e9ac0b2638d16043a7c8d0e1b904a05915b"},"cell_type":"markdown","source":"Let's look at the predictions of the positives. My hypothesis is it will be very difficult to detect some of these as they are using difficult sarcasm or words out of vocabulary"},{"metadata":{"trusted":true,"_uuid":"49dd2a51d62a5f2223ebec8c94b4048e0f926065"},"cell_type":"code","source":"positive_scores = val_pred1[:, 0][y_val > 0]","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cf542e9b28553c432dbad7ee6636a942ef7938ec"},"cell_type":"markdown","source":"Here are the positives from our validation set"},{"metadata":{"trusted":true,"_uuid":"1adc3130c032614db76f28e196bf95e3afc7bb70"},"cell_type":"code","source":"pos_text = X_val.loc[positives.index.values]\npos_text","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f1472ddfed6db0dbaa6c14c7e486dd5822a54a49"},"cell_type":"markdown","source":"Let's look at what our predictions across the whole set look like"},{"metadata":{"trusted":true,"_uuid":"0274f948f95e59a3678c9d5322fb8aa9d499deec"},"cell_type":"code","source":"pd.DataFrame(val_pred1).describe()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d4b1c9208167608c5d219a3f752b6e1838df3315"},"cell_type":"markdown","source":"Now lets look at our predictions for just the positives and then sort them and take the 250 worst predictions. Then we can look at the text for these and get an idea of why our model might be having such a hard time with these"},{"metadata":{"trusted":true,"_uuid":"5c1cbc9a31947dde8335bec7efffda7e3739152d"},"cell_type":"code","source":"pred_sort = val_pred1[:, 0][y_val > 0].argsort()[:250][::-1]\npd.DataFrame(val_pred1[y_val > 0][pred_sort]).describe()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2e406ddbe4eb4df3a712ca28a4aab6a496e8a160"},"cell_type":"markdown","source":"Distribution is interesting. Let's look at the text"},{"metadata":{"trusted":true,"_uuid":"3c6d2d5463d3c60fb58e0b51d0a2a47ae4cc12d6"},"cell_type":"code","source":"X_val[y_val > 0].iloc[pred_sort]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"14992bf13b9dd6e7859c750515b6221ec254ebcb"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"172cfa9f47e1768a53594074bffccbee526c2629"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}