{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport time\nimport numpy as np \nimport pandas as pd \nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D,MaxPooling1D,BatchNormalization\nfrom keras.models import Model","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba5a1b8109dee2c9fbc628d5da4a7c3447d42fb8"},"cell_type":"code","source":"train, dev = train_test_split(train, test_size=0.1, random_state=2018)\n\nembed_size = 300 \nmax_features = 50000 \nmaxlen = 60 \n\ntrain_X = train[\"question_text\"].fillna(\"_na_\").values\ndev_X = dev[\"question_text\"].fillna(\"_na_\").values\ntest_X = test[\"question_text\"].fillna(\"_na_\").values\n\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\ndev_X = tokenizer.texts_to_sequences(dev_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\ntrain_X = pad_sequences(train_X, maxlen=maxlen)\ndev_X = pad_sequences(dev_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\ntrain_y = train['target'].values\ndev_y = dev['target'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3e49bb1d91b84278f9de0392b16c3f453cf5475b"},"cell_type":"code","source":"model=None","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3cfab26c6cced33ef7ab84f0d36997113131d530"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4d603907622fe9f9fa6626915c7fe3d97b15ef92"},"cell_type":"code","source":"from keras import backend as K\n\ndef f1(y_true, y_pred):\n    def recall(y_true, y_pred):\n        \"\"\"Recall metric.\n\n        Only computes a batch-wise average of recall.\n\n        Computes the recall, a metric for multi-label classification of\n        how many relevant items are selected.\n        \"\"\"\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n        recall = true_positives / (possible_positives + K.epsilon())\n        return recall\n\n    def precision(y_true, y_pred):\n        \"\"\"Precision metric.\n\n        Only computes a batch-wise average of precision.\n\n        Computes the precision, a metric for multi-label classification of\n        how many selected items are relevant.\n        \"\"\"\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n        precision = true_positives / (predicted_positives + K.epsilon())\n        return precision\n    precision = precision(y_true, y_pred)\n    recall = recall(y_true, y_pred)\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"47edd83c6dfae9e063f64c4f43c88c04154dcfd6"},"cell_type":"code","source":"def net(input_shape):\n    sentence_indices = Input(input_shape, dtype='int32')\n    X = Embedding(max_features, embed_size,weights=[embedding_matrix])(sentence_indices)\n    X = Bidirectional(CuDNNGRU(64, return_sequences=True))(X)\n    X = GlobalMaxPool1D()(X)\n    X = Dense(16, activation=\"relu\")(X)\n    X = Dropout(0.1)(X)\n    X = Dense(1, activation=\"sigmoid\")(X)\n    model = Model(inputs=sentence_indices, outputs=X)\n    return model\nmodel=net((maxlen,))\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy',f1])\n\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3af5590978531810849719e41a563eb02de84add"},"cell_type":"code","source":"weight={0:0.4,1:1}","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7a6bc4f0d045bc8525f143b06852d1082ed9a8bd"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"47238831a4701c8a67dc7ecb130ac1402baf7bb2"},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=3, validation_data=(dev_X, dev_y),class_weight=weight)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b7ab4100f723ad535528865b1edc7896bce80223"},"cell_type":"code","source":"pred_dev_y = model.predict([dev_X], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.601, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(dev_y, (pred_dev_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3216362afb0f49579d287a06f13adf8cd7d8b0cf"},"cell_type":"code","source":"label = model.predict([test_X], batch_size=1024, verbose=1)\nlabel=(label>0.58).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc188f2787ea7b98d3a40953a95a5fc09ff2764d"},"cell_type":"code","source":"submission=pd.read_csv(\"../input/sample_submission.csv\")\nsubmission[\"prediction\"]=label\nsubmission.head()","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}