{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nimport gc\nimport matplot.pylot as plt\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom keras.callbacks import ModelCheckpoint, History\nfrom keras.layers import Conv2D, MaxPool2D\nfrom keras.layers import Reshape, Flatten, Concatenate, Dropout, SpatialDropout1D","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"333192afcca6ca91ebce8b54982e5a09cb5486ae"},"cell_type":"code","source":"\n## some config values \nEMBED_SIZE = 300 #how big is each word vector\nMAX_FEATURES= 50000 # how many unique words to use (i.e num rows in embedding vector)\nMAX_LEN = 100 # max number of words in a question to use\nS_DROPOUT = 0.4\nDROPOUT = 0.1\ndef load_preprocess_data(test_size=0.1,  max_features = MAX_FEATURES, maxlen = MAX_LEN ):\n    \n    train_df = pd.read_csv(\"../input/train.csv\")\n    test_df = pd.read_csv(\"../input/test.csv\")\n#     print(\"Train shape : \",train_df.shape)\n#     print(\"Test shape : \",test_df.shape)\n\n    ## split to train and val\n    train_df_1, val_df = train_test_split(train_df, test_size=test_size, random_state=2018)\n    \n    ## fill up the missing values\n    train_X = train_df[\"question_text\"].fillna(\"_na_\").values\n    val_X = val_df[\"question_text\"].fillna(\"_na_\").values\n    test_X = test_df[\"question_text\"].fillna(\"_na_\").values\n\n    ## Tokenize the sentences\n    tokenizer = Tokenizer(num_words=max_features)\n    tokenizer.fit_on_texts(list(train_X))\n    train_X = tokenizer.texts_to_sequences(train_X)\n    val_X = tokenizer.texts_to_sequences(val_X)\n    test_X = tokenizer.texts_to_sequences(test_X)\n\n    ## Pad the sentences \n    train_X = pad_sequences(train_X, maxlen=maxlen)\n    val_X = pad_sequences(val_X, maxlen=maxlen)\n    test_X = pad_sequences(test_X, maxlen=maxlen)\n\n    ## Get the target values\n    train_y = train_df['target'].values\n    val_y = val_df['target'].values\n    \n    return train_X , train_y, val_X, val_y, test_X, tokenizer.index_word","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ada9d281f4c5dec79fc9c81c3af62e556a918e71"},"cell_type":"code","source":"train_X , train_y, val_X, val_y, test_X, index_word = load_preprocess_data()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba87815d4c7835de8c2e6c3dea5537617354bd62"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = index_word\nnb_words = min(MAX_FEATURES, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor i, word in word_index.items():\n    if i >= MAX_FEATURES: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n\ndel embeddings_index; gc.collect() \nnp.shape(embedding_matrix)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"740f823aa38a848949c2f83a767dabaf513e6646"},"cell_type":"code","source":"def build_model( max_features = MAX_FEATURES, maxlen = MAX_LEN, embed_size = EMBED_SIZE):\n    \n    filter_sizes = [1,2,3,5]\n    num_filters = 36\n\n    inp = Input(shape=(maxlen,))\n    x = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\n    x = SpatialDropout1D(S_DROPOUT)(x)\n    x = Reshape((maxlen, embed_size, 1))(x)\n\n    maxpool_pool = []\n    for i in range(len(filter_sizes)):\n        conv = Conv2D(num_filters, kernel_size=(filter_sizes[i], embed_size),\n                                     kernel_initializer='he_normal', activation='elu')(x)\n        maxpool_pool.append(MaxPool2D(pool_size=(maxlen - filter_sizes[i] + 1, 1))(conv))\n\n    z = Concatenate(axis=1)(maxpool_pool)   \n    z = Flatten()(z)\n    z = Dropout(DROPOUT)(z)\n\n    outp = Dense(1, activation=\"sigmoid\")(z)\n\n    model = Model(inputs=inp, outputs=outp)\n    model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    model.summary()\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"98d8c9ad0615caf30d9b8a9a58649e0d80144a85"},"cell_type":"code","source":"test_model = build_model()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"653fdac87cf5bb8dcec42290d32146dbc50f1843"},"cell_type":"code","source":"# weight_path=\"{}_weights.best.hdf5\".format('model')\n# checkpoint = ModelCheckpoint(weight_path, monitor='val_loss', verbose=1,\n#                              save_best_only=True, mode='max', save_weights_only = True)\n# callbacks_list = [checkpoint]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"635eeae0539278a4dd6d77e092d26eddc29c50e5"},"cell_type":"code","source":"## Train the model \nhistory = History()\n#history_log = [test_model.fit(train_X, train_y, batch_size=512, epochs=2, callbacks=callbacks_list, validation_data=(val_X, val_y))]\nhistory_log = [test_model.fit(train_X, train_y, batch_size=512, epochs=1, callbacks=[history], validation_data=(val_X, val_y))]\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c1804b04af38c12e05468fb84324ea99866fb397"},"cell_type":"code","source":"# test_model.load_weights(weight_path)\n# test_model.save(\"../input/best_model.h5\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf0fae59b83c310b5a3801e8dfefd244d7f9e34b"},"cell_type":"code","source":"pred_val_y = test_model.predict([val_X], batch_size=1024, verbose=1)\nthresholds = []\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    res = metrics.f1_score(val_y, (pred_val_y > thresh).astype(int))\n    thresholds.append([thresh, res])\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, res))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"f2fb35911aa81d6fdadf0e4b93c95ddfd062c91f"},"cell_type":"code","source":"pred_noemb_test_y = test_model.predict([test_X], batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"630dfe47022ec7e076692896c772cc9ec0640e0c"},"cell_type":"code","source":"from keras.wrappers.scikit_learn import KerasClassifier\nfrom sklearn.metrics import roc_curve","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e78bad03f0a3a770fdae98cf17e0996682e1c90b"},"cell_type":"code","source":"y_pred_keras = test_model.predict([val_X], batch_size=1024, verbose=1).ravel()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"81ee5756828c245a430f5278ef138bc389135597"},"cell_type":"code","source":"fpr_keras, tpr_keras, thresholds_keras = roc_curve(val_y, y_pred_keras)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"76e0a5c1b00569aac2d65e097889af05a395db91"},"cell_type":"code","source":"from sklearn.metrics import auc\nauc_keras = auc(fpr_keras, tpr_keras)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b1d62532110f94aadbd258faf77af55c61fa7a18"},"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.figure(1)\nplt.plot([0, 1], [0, 1], 'k--')\nplt.plot(fpr_keras, tpr_keras, label='Keras (area = {:.3f})'.format(auc_keras))\n#plt.plot(fpr_rf, tpr_rf, label='RF (area = {:.3f})'.format(auc_rf))\nplt.xlabel('False positive rate')\nplt.ylabel('True positive rate')\nplt.title('ROC curve')\nplt.legend(loc='best')\nplt.show()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}