{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom keras.callbacks import ModelCheckpoint, History","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"333192afcca6ca91ebce8b54982e5a09cb5486ae"},"cell_type":"code","source":"\n## some config values \nEMBED_SIZE = 300 #how big is each word vector\nMAX_FEATURES= 50000 # how many unique words to use (i.e num rows in embedding vector)\nMAX_LEN = 100 # max number of words in a question to use\n\ndef load_preprocess_data(test_size=0.1,  max_features = MAX_FEATURES, maxlen = MAX_LEN ):\n    \n    train_df = pd.read_csv(\"../input/train.csv\")\n    test_df = pd.read_csv(\"../input/test.csv\")\n#     print(\"Train shape : \",train_df.shape)\n#     print(\"Test shape : \",test_df.shape)\n\n    ## split to train and val\n    train_df_1, val_df = train_test_split(train_df, test_size=test_size, random_state=2018)\n    \n    ## fill up the missing values\n    train_X = train_df[\"question_text\"].fillna(\"_na_\").values\n    val_X = val_df[\"question_text\"].fillna(\"_na_\").values\n    test_X = test_df[\"question_text\"].fillna(\"_na_\").values\n\n    ## Tokenize the sentences\n    tokenizer = Tokenizer(num_words=max_features)\n    tokenizer.fit_on_texts(list(train_X))\n    train_X = tokenizer.texts_to_sequences(train_X)\n    val_X = tokenizer.texts_to_sequences(val_X)\n    test_X = tokenizer.texts_to_sequences(test_X)\n\n    ## Pad the sentences \n    train_X = pad_sequences(train_X, maxlen=maxlen)\n    val_X = pad_sequences(val_X, maxlen=maxlen)\n    test_X = pad_sequences(test_X, maxlen=maxlen)\n\n    ## Get the target values\n    train_y = train_df['target'].values\n    val_y = val_df['target'].values\n    \n    return train_X , train_y, val_X, val_y, test_X, tokenizer.index_word","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ada9d281f4c5dec79fc9c81c3af62e556a918e71"},"cell_type":"code","source":"train_X , train_y, val_X, val_y, test_X, word_index = load_preprocess_data()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"740f823aa38a848949c2f83a767dabaf513e6646"},"cell_type":"code","source":"def build_model( max_features = MAX_FEATURES, maxlen = MAX_LEN, embed_size = EMBED_SIZE,):\n    \n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size)(inp)\n    x = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\n    x = GlobalMaxPool1D()(x)\n    x = Dense(16, activation=\"relu\")(x)\n    x = Dropout(0.1)(x)\n    x = Dense(1, activation=\"sigmoid\")(x)\n    model = Model(inputs=inp, outputs=x)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    model.summary()\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"98d8c9ad0615caf30d9b8a9a58649e0d80144a85"},"cell_type":"code","source":"test_model = build_model()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"653fdac87cf5bb8dcec42290d32146dbc50f1843"},"cell_type":"code","source":"# weight_path=\"{}_weights.best.hdf5\".format('model')\n# checkpoint = ModelCheckpoint(weight_path, monitor='val_loss', verbose=1,\n#                              save_best_only=True, mode='max', save_weights_only = True)\n# callbacks_list = [checkpoint]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d085a281851d0329751f77d31b025bf4171ec806"},"cell_type":"code","source":"def get_embedding_matrix(max_features = MAX_FEATURES):\n    \n    EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\n    def get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n    embeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\n    all_embs = np.stack(embeddings_index.values())\n    emb_mean,emb_std = all_embs.mean(), all_embs.std()\n    embed_size = all_embs.shape[1]\n\n    nb_words = min(max_features, len(word_index))\n    embedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n    for word, i in word_index.items():\n        if i >= max_features: continue\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n    return embedding_matrix","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6754e75d46a441b0fa04fe4ad89ebab6e7e2b182"},"cell_type":"code","source":"embedding_matrix = get_embedding_matrix()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"635eeae0539278a4dd6d77e092d26eddc29c50e5"},"cell_type":"code","source":"## Train the model \nhistory = History()\n#history_log = [test_model.fit(train_X, train_y, batch_size=512, epochs=2, callbacks=callbacks_list, validation_data=(val_X, val_y))]\nhistory_log = [test_model.fit(train_X, train_y, batch_size=512, epochs=1, callbacks=[history], validation_data=(val_X, val_y),weights=[embedding_matrix])]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c1804b04af38c12e05468fb84324ea99866fb397"},"cell_type":"code","source":"# test_model.load_weights(weight_path)\n# test_model.save(\"../input/best_model.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf0fae59b83c310b5a3801e8dfefd244d7f9e34b"},"cell_type":"code","source":"pred_noemb_val_y = test_model.predict([val_X], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_noemb_val_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"f2fb35911aa81d6fdadf0e4b93c95ddfd062c91f"},"cell_type":"code","source":"pred_noemb_test_y = test_model.predict([test_X], batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"630dfe47022ec7e076692896c772cc9ec0640e0c"},"cell_type":"code","source":"from keras.wrappers.scikit_learn import KerasClassifier\nfrom sklearn.metrics import roc_curve","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"94e62fd267e18a7ac4f2081e4a762534ed539581"},"cell_type":"code","source":"y_pred_keras = test_model.predict([val_X], batch_size=1024, verbose=1).ravel()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1d3a32fc3015c5c79317bf66750bb4f836198ade"},"cell_type":"code","source":"fpr_keras, tpr_keras, thresholds_keras = roc_curve(val_y, y_pred_keras)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7cec2f7139d90e0e0a379abfad7b82d27f39cf6d"},"cell_type":"code","source":"from sklearn.metrics import auc\nauc_keras = auc(fpr_keras, tpr_keras)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c8751b1a5c5d0c31f63f85828bfc415c1c1e4be7"},"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(1)\nplt.plot([0, 1], [0, 1], 'k--')\nplt.plot(fpr_keras, tpr_keras, label='Keras (area = {:.3f})'.format(auc_keras))\n#plt.plot(fpr_rf, tpr_rf, label='RF (area = {:.3f})'.format(auc_rf))\nplt.xlabel('False positive rate')\nplt.ylabel('True positive rate')\nplt.title('ROC curve')\nplt.legend(loc='best')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7cd4955ee3f31b4c04b18147416c3ba5f0fb120c"},"cell_type":"code","source":"\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e4fe8b36cf0bcde966f4ef444afae1093eb54d34"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"97c90ad22b97c58eaf0c2f4b76609abbcc00aa10"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"}},"nbformat":4,"nbformat_minor":1}