{"cells":[{"metadata":{"_uuid":"38e7d605b52588dfa82fb54def70d25e511df5bd"},"cell_type":"markdown","source":"Inspired by:\n* https://www.kaggle.com/sudalairajkumar/a-look-at-different-embeddings\n* https://www.kaggle.com/shujian/single-rnn-with-4-folds-v1-9\n* http://mlexplained.com/2018/01/13/weight-normalization-and-layer-normalization-explained-normalization-in-deep-learning-part-2/\n* https://arxiv.org/abs/1607.06450\n* https://github.com/keras-team/keras/issues/3878\n* https://www.kaggle.com/lystdo/lstm-with-word2vec-embeddings\n* https://www.kaggle.com/jhoward/improved-lstm-baseline-glove-dropout\n* https://www.kaggle.com/aquatic/entity-embedding-neural-net\n\n(and other links in notebook)\n\nRemark:\nmodel overfits like hell...\n\nv7: \n- normalization \n- dropout\n- alphadrop\n- selu"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nnp.set_printoptions(threshold=np.nan)\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\nprint(os.listdir(\"../input/embeddings\"))\nprint(os.listdir(\"../input/embeddings/GoogleNews-vectors-negative300\"))\n\n# Any results you write to the current directory are saved as output.\n\nimport gensim\nfrom gensim.utils import simple_preprocess\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.callbacks import Callback\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report,f1_score,precision_recall_fscore_support,recall_score,precision_score\nfrom keras import backend as K\nfrom sklearn.utils import class_weight\nimport matplotlib.pyplot as plt\n\n#https://www.kaggle.com/shujian/single-rnn-with-4-folds-v1-9\ndef threshold_search(y_true, y_proba):\n    best_threshold = 0\n    best_score = 0\n    for threshold in [i * 0.01 for i in range(100)]:\n        score = f1_score(y_true=y_true, y_pred=y_proba > threshold)\n        print('\\rthreshold = %f | score = %f'%(threshold,score),end='')\n        if score > best_score:\n            best_threshold = threshold\n            best_score = score\n    print('\\nbest threshold is % f with score %f'%(best_threshold,best_score))\n    search_result = {'threshold': best_threshold, 'f1': best_score}\n    return search_result","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"77b7c5199c59943744495e62d7c0f73f68769e17"},"cell_type":"code","source":"df = pd.read_csv('../input/train.csv')\ndf[\"question_text\"].fillna(\"_##_\",inplace=True)\nmax_len = df['question_text'].apply(lambda x:len(x)).max()\nprint('max length of sequences:',max_len)\n# df = df.sample(frac=0.1)\n\nprint('columns:',df.columns)\npd.set_option('display.max_columns',None)\nprint('df head:',df.head())\nprint('example of the question text values:',df['question_text'].head().values)\nprint('what values contains target:',df.target.unique())\n\nprint('Computing class weights....')\n#https://datascience.stackexchange.com/questions/13490/how-to-set-class-weights-for-imbalanced-classes-in-keras\nclass_weights = class_weight.compute_class_weight('balanced',\n                                                 np.unique(df.target.values),\n                                                 df.target.values)\nprint('class_weights:',class_weights)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f41b23c1f3f4eed0d8d419974fe795b63f3df50b"},"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\n#dim of vectors\ndim = 300\n# max words in vocab\nnum_words = 50000\n# max number in questions\nmax_len = 100 \n\nprint('Fiting tokenizer')\n## Tokenize the sentences\ntokenizer = Tokenizer(num_words=num_words)\ntokenizer.fit_on_texts(df['question_text'])\n\nprint('spliting data')\ndf_train,df_test = train_test_split(df, random_state=1)\n\nprint('text to sequence')\nx_train = tokenizer.texts_to_sequences(df_train['question_text'])\nx_test = tokenizer.texts_to_sequences(df_test['question_text'])\n\nprint('pad sequence')\n## Pad the sentences \nx_train = pad_sequences(x_train,maxlen=max_len)\nx_test = pad_sequences(x_test, maxlen=max_len)\n\n## Get the target values\ny_train = df_train['target'].values\ny_test = df_test['target'].values\n\nprint(x_train.shape)\nprint(y_train.shape)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"db71c4ce3c1e3743f964c1a6e43a11644ee53cb4","scrolled":true},"cell_type":"code","source":"# https://www.kaggle.com/jhoward/improved-lstm-baseline-glove-dropout\nprint('loading word2vec model...')\nword2vec = gensim.models.KeyedVectors.load_word2vec_format('../input/embeddings/GoogleNews-vectors-negative300/GoogleNews-vectors-negative300.bin', binary=True)\nprint('vocab:',len(word2vec.vocab))\n\nall_embs = word2vec.vectors\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nprint(emb_mean,emb_std)\n\nprint(num_words,' from ',len(tokenizer.word_index.items()))\n# num_words = min(num_words, len(tokenizer.word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (num_words, dim))\n\n# embedding_matrix = np.zeros((num_words, dim))\ncount = 0\nfor word, i in tokenizer.word_index.items():\n    if i>=num_words:\n        break\n    if word in word2vec.vocab:\n        embedding_matrix[i] = word2vec.word_vec(word)\n    else:\n        count += 1\nprint('embedding matrix size:',embedding_matrix.shape)\nprint('Number of words not in vocab:',count)\n\ndel word2vec\nimport gc\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8d6abe3bd0a68cf06bc006b9b6efd799b90c7dfb"},"cell_type":"code","source":"def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open('../input/embeddings/glove.840B.300d/glove.840B.300d.txt'))\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nprint(len(all_embs))\n\n\nword_index = tokenizer.word_index\n# num_words = min(num_words, len(word_index))\nembedding_matrix_glov = np.random.normal(emb_mean, emb_std, (num_words, dim))\ncount=0\nfor word, i in word_index.items():\n    if i >= num_words: \n        break\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: \n        embedding_matrix_glov[i] = embedding_vector\n    else:\n        count += 1\nprint('embedding matrix size:',embedding_matrix_glov.shape)\nprint('Number of words not in vocab:',count)\n\ndel embeddings_index,all_embs\nimport gc\ngc.collect()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e78bb6fa9a5f4c8bdc6d9fd09534076276bc703e"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nprint(len(all_embs))\n\n\nword_index = tokenizer.word_index\n# num_words = min(num_words, len(word_index))\nembedding_matrix_para = np.random.normal(emb_mean, emb_std, (num_words, dim))\ncount=0\nfor word, i in word_index.items():\n    if i >= num_words: \n        break\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: \n        embedding_matrix_para[i] = embedding_vector\n    else:\n        count += 1\nprint('embedding matrix size:',embedding_matrix_glov.shape)\nprint('Number of words not in vocab:',count)\n\ndel embeddings_index,all_embs\nimport gc\ngc.collect()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"504f0213725bb0ffc728c58fa0845b35c542179c"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\n\nprint(len(all_embs))\n\nword_index = tokenizer.word_index\nembedding_matrix_wiki = np.random.normal(emb_mean, emb_std, (num_words, dim))\n\ncount=0\nfor word, i in word_index.items():\n    if i >= num_words: \n        break\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: \n        embedding_matrix_wiki[i] = embedding_vector\n    else:\n        count += 1\nprint('embedding matrix size:',embedding_matrix_wiki.shape)\nprint('Number of words not in vocab:',count)\n\ndel embeddings_index,all_embs\nimport gc\ngc.collect()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"55cb2ac1dca7de9fba51e8a7e5dba402159be302"},"cell_type":"code","source":"from keras.layers import Dense, Input,Embedding, Dropout, Activation, CuDNNLSTM,BatchNormalization,SpatialDropout1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D, Concatenate, GlobalAveragePooling1D,Average,AlphaDropout\nfrom keras.models import Model\nfrom keras.callbacks import Callback,EarlyStopping\nfrom keras.engine import Layer\nfrom keras.initializers import Ones, Zeros\nimport keras.backend as K\nfrom keras import regularizers\nfrom keras import constraints\n\n\n#model looks to be from here: https://www.kaggle.com/CVxTz/keras-bidirectional-lstm-baseline-lb-0-069\n\ndef complex_layer(embedding_matrix = None):\n    inp1 = Input(shape=(max_len,))\n    if embedding_matrix is not None:\n        e1 = Embedding(num_words, dim, weights=[embedding_matrix],trainable=True)(inp1)\n    else:\n        e1 = Embedding(num_words, dim,)(inp1)\n    e1 = SpatialDropout1D(0.1)(e1)\n    e1 = Bidirectional(CuDNNLSTM(50,return_sequences=True))(e1)\n    e1 = GlobalMaxPool1D()(e1)\n#     e1 = Dropout(0.1)(e1)\n#     e1 = Dense(50, activation=\"selu\")(e1)\n    e1 = Dropout(0.4)(e1)\n    e1 = Dense(1, activation=\"sigmoid\",kernel_regularizer='l2')(e1)\n    return inp1,e1\n    \n    \nmatrixes = [embedding_matrix,embedding_matrix_glov,embedding_matrix_wiki,embedding_matrix_para,None]\ninputs = []\noutputs=[]\nfor emb_matrix in matrixes:\n    inp,outp =  complex_layer(emb_matrix)\n    inputs.append(inp)\n    outputs.append(outp)\n\nx = Average()(outputs)\n\nmodel = Model(inputs=inputs, outputs=x)\n\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\nprint(model.summary())\n\n# for commiting the model to competition i need to comment these sections....otherwise the running time will be more then 2h on gpu...\nhistory = model.fit([x_train,x_train,x_train,x_train,x_train],y_train, \n                      batch_size=512, \n                      validation_split=0.2,\n                      epochs=100,\n                      #overfits rather soon\n                      callbacks=[EarlyStopping(patience=5)])\n\nprint('training done....')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"da921fc2d216b1e2db6819ea13ab0b0c4a0d731d"},"cell_type":"code","source":"_, ax = plt.subplots(1, 2, figsize=(12, 6))\nax[0].plot(history.history['loss'], label='loss')\nax[0].plot(history.history['val_loss'], label='val_loss')\nax[0].legend()\nax[0].set_title('loss')\n\nax[1].plot(history.history['acc'], label='acc')\nax[1].plot(history.history['val_acc'], label='val_acc')\nax[1].legend()\nax[1].set_title('acc')\n\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cde59508feb4a5c5f398128f9a5176e9a1912236","scrolled":true},"cell_type":"code","source":"#for train set\ny_pred = model.predict([x_train,x_train,x_train,x_train,x_train],batch_size=1024, verbose=1)\nsearch_result = threshold_search(y_train, y_pred)\nprint(search_result)\ny_pred = y_pred>search_result['threshold']\ny_pred = y_pred.astype(int)\n\nprint('RESULTS ON TRAINING SET:\\n',classification_report(y_train,y_pred))\n\n\n#for test set\ny_pred = model.predict([x_test,x_test,x_test,x_test,x_test],batch_size=1024, verbose=1)\nsearch_result = threshold_search(y_test, y_pred)\nprint(search_result)\ny_pred = y_pred>search_result['threshold']\ny_pred = y_pred.astype(int)\n\nprint('RESULTS ON TEST SET:\\n',classification_report(y_test,y_pred))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"de2f170bc8d3f50b4d270b5b00ab934cfdc379ed"},"cell_type":"markdown","source":"# Results\n\n# layernom + drop 0.1\n\n    979591/979591 [==============================] - 235s 240us/step\n    threshold = 0.990000 | score = 0.097414\n    best threshold is  0.380000 with score 0.790592\n    {'threshold': 0.38, 'f1': 0.7905919979733687}\n    RESULTS ON TRAINING SET:\n                  precision    recall  f1-score   support\n\n              0       0.99      0.98      0.99    919150\n              1       0.76      0.83      0.79     60441\n\n    avg / total       0.97      0.97      0.97    979591\n\n    326531/326531 [==============================] - 78s 240us/step\n    threshold = 0.990000 | score = 0.045348\n    best threshold is  0.360000 with score 0.656955\n    {'threshold': 0.36, 'f1': 0.6569553683622655}\n    RESULTS ON TEST SET:\n                  precision    recall  f1-score   support\n\n              0       0.98      0.97      0.98    306162\n              1       0.62      0.69      0.66     20369\n\n    avg / total       0.96      0.95      0.96    326531\n    \n# Batchnorm + drop 0.1\n\n    979591/979591 [==============================] - 236s 241us/step\n    threshold = 0.990000 | score = 0.235990\n    best threshold is  0.350000 with score 0.805080\n    {'threshold': 0.35000000000000003, 'f1': 0.805080118694362}\n    RESULTS ON TRAINING SET:\n                  precision    recall  f1-score   support\n\n              0       0.99      0.98      0.99    919150\n              1       0.77      0.84      0.81     60441\n\n    avg / total       0.98      0.97      0.98    979591\n\n    326531/326531 [==============================] - 78s 240us/step\n    threshold = 0.990000 | score = 0.125359\n    best threshold is  0.250000 with score 0.645530\n    {'threshold': 0.25, 'f1': 0.6455297976304692}\n    RESULTS ON TEST SET:\n                  precision    recall  f1-score   support\n\n              0       0.98      0.97      0.97    306162\n              1       0.59      0.71      0.65     20369\n\n    avg / total       0.96      0.95      0.95    326531\n    \n# layernom + drop 0.5\n\n    979591/979591 [==============================] - 235s 240us/step\n    threshold = 0.990000 | score = 0.183250\n    best threshold is  0.370000 with score 0.789071\n    {'threshold': 0.37, 'f1': 0.7890714222964069}\n    RESULTS ON TRAINING SET:\n                  precision    recall  f1-score   support\n\n              0       0.99      0.98      0.99    919150\n              1       0.76      0.82      0.79     60441\n\n    avg / total       0.97      0.97      0.97    979591\n\n    326531/326531 [==============================] - 78s 240us/step\n    threshold = 0.990000 | score = 0.092049\n    best threshold is  0.280000 with score 0.651629\n    {'threshold': 0.28, 'f1': 0.6516285112674164}\n    RESULTS ON TEST SET:\n                  precision    recall  f1-score   support\n\n              0       0.98      0.97      0.97    306162\n              1       0.60      0.71      0.65     20369\n\n    avg / total       0.96      0.95      0.95    326531\n  \n  # 2xlayernom, no drop\n  \n      979591/979591 [==============================] - 236s 241us/step\n    threshold = 0.990000 | score = 0.000066\n    best threshold is  0.360000 with score 0.800982\n    {'threshold': 0.36, 'f1': 0.80098241415684}\n    RESULTS ON TRAINING SET:\n                  precision    recall  f1-score   support\n\n              0       0.99      0.98      0.99    919150\n              1       0.76      0.84      0.80     60441\n\n    avg / total       0.98      0.97      0.97    979591\n\n    326531/326531 [==============================] - 78s 240us/step\n    threshold = 0.990000 | score = 0.000000\n    best threshold is  0.330000 with score 0.658858\n    {'threshold': 0.33, 'f1': 0.6588583861858248}\n    RESULTS ON TEST SET:\n                  precision    recall  f1-score   support\n\n              0       0.98      0.97      0.98    306162\n              1       0.62      0.70      0.66     20369\n\n    avg / total       0.96      0.95      0.96    326531\n    \n# 3xlayernom, no drop\n    979591/979591 [==============================] - 285s 291us/step\n    threshold = 0.990000 | score = 0.087948\n    best threshold is  0.380000 with score 0.807504\n    {'threshold': 0.38, 'f1': 0.8075041306405301}\n    RESULTS ON TRAINING SET:\n                  precision    recall  f1-score   support\n\n              0       0.99      0.98      0.99    919150\n              1       0.78      0.83      0.81     60441\n\n    avg / total       0.98      0.98      0.98    979591\n\n    326531/326531 [==============================] - 95s 290us/step\n    threshold = 0.990000 | score = 0.040397\n    best threshold is  0.310000 with score 0.657994\n    {'threshold': 0.31, 'f1': 0.6579935743698809}\n    RESULTS ON TEST SET:\n                  precision    recall  f1-score   support\n\n              0       0.98      0.97      0.98    306162\n              1       0.61      0.71      0.66     20369\n\n    avg / total       0.96      0.95      0.96    326531\n\n# no layernorm, no drop, kernel_reg\n"},{"metadata":{"trusted":true,"_uuid":"3beb75a2101c603f182934220444bf573647220c"},"cell_type":"code","source":"#fit final model on all data\nprint('text to sequence')\nx = tokenizer.texts_to_sequences(df['question_text'])\n\nprint('pad sequence')\n## Pad the sentences \nx = pad_sequences(x,maxlen=max_len)\n\n## Get the target values\ny = df['target'].values\n\nprint('fiting final model...')\nhistory = model.fit([x,x,x,x,x],y, batch_size=512, epochs=2)\n\nprint('fitting on full data done...')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"49e1bd5b69b428e57ad8248868d4b23c9c6f9d84"},"cell_type":"code","source":"_, ax = plt.subplots(1, 2, figsize=(12, 6))\nax[0].plot(history.history['loss'], label='loss')\nax[0].legend()\nax[0].set_title('loss')\n\nax[1].plot(history.history['acc'], label='acc')\nax[1].legend()\nax[1].set_title('acc')\n\nplt.show()\n\ny_pred = model.predict([x,x,x,x,x],batch_size=1024, verbose=1)\nsearch_result = threshold_search(y, y_pred)\ny_pred = y_pred>search_result['threshold']\ny_pred = y_pred.astype(int)\n\nprint(classification_report(y,y_pred))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0ad34df4ae095e1de08c64079fa6b0ecbc944423"},"cell_type":"code","source":"#submission\nprint('Loading test data...')\ndf_final = pd.read_csv('../input/test.csv')\ndf_final[\"question_text\"].fillna(\"_##_\", inplace=True)\n\nx_final=tokenizer.texts_to_sequences(df_final['question_text'])\nx_final = pad_sequences(x_final,maxlen=max_len)\n\ny_pred = model.predict([x_final,x_final,x_final,x_final,x_final],batch_size=1024,verbose=1)\ny_pred = y_pred > search_result['threshold']\ny_pred = y_pred.astype(int)\nprint(y_pred[:5])\n\ndf_subm = pd.DataFrame()\ndf_subm['qid'] = df_final.qid\ndf_subm['prediction'] = y_pred\nprint(df_subm.head())\ndf_subm.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}