{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nfrom tqdm import tqdm\ntqdm.pandas()\nimport numpy as np\nimport operator \nimport re\nimport gc\nimport keras\nimport seaborn as sns\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"547d271dbd2fd534b29f884ae4b3537ef8a88c60"},"cell_type":"code","source":"print(train.shape)\nprint(test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c936688c717755be19d671a29d7aa0a79496b03f"},"cell_type":"code","source":"train['target'].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"12befa8257c58e3f916e1fefd7dc47fd99f936f6"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"07e7217449ddfaa6f5108a5f9a4c69cfdd54538f"},"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n\nmax_features = 50000\ntk = Tokenizer(lower = True, filters='', num_words=max_features)\nfull_text = list(train['question_text'].values) + list(test['question_text'].values)\ntk.fit_on_texts(full_text)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"72aede006cbb0b96c54e993396a8225a7cba17ae"},"cell_type":"code","source":"train_tokenized = tk.texts_to_sequences(train['question_text'].fillna('UNK'))\ntest_tokenized = tk.texts_to_sequences(test['question_text'].fillna('UNK'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9d8425c3984f922ce154cd677841849dfc81c90e"},"cell_type":"code","source":"max_len = 70\nX_train = pad_sequences(train_tokenized, maxlen = max_len)\nX_test = pad_sequences(test_tokenized, maxlen = max_len)\nY = train['target'].values\nsub = test[['qid']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5a598374974a5dca3a3ced3fe7306a2af1affb8f"},"cell_type":"code","source":"path = \"../input/embeddings/glove.840B.300d/glove.840B.300d.txt\"\nemb_size = 300","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"853b4d929e87da6c23dcede1b8f892cb0b967671"},"cell_type":"code","source":"def get_coef(word,*arr): \n    return word, np.asarray(arr, dtype='float32')\nembedding_index = dict(get_coef(*o.strip().split(\" \")) for o in open(path, encoding='utf-8', errors='ignore'))\n\nword_index = tk.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.zeros((nb_words + 1, emb_size))\nfor word, i in word_index.items():\n    if i >= max_features: \n        continue\n    embedding_vector = embedding_index.get(word)\n    if embedding_vector is not None: \n        embedding_matrix[i] = embedding_vector","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"11e465ec74888b42cff0e872d72d22282005f845"},"cell_type":"code","source":"\ndel train_tokenized,test_tokenized,train,test\ngc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f344f4f686bcf2d3c539f7f8596691aab2be7870"},"cell_type":"code","source":"from keras.models import Sequential\nfrom keras.layers import CuDNNLSTM, Dense, Bidirectional,Input,Dropout,Flatten,Embedding,BatchNormalization\nfrom keras.models import Model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"897ae3b80341384cdb53990455a01db23217532e"},"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"67b41ce0459264d8d931ef3f7582f90bcee9cad9"},"cell_type":"code","source":"def lstm_model():\n    keras.backend.clear_session()       \n    inp = Input(shape=(70,))\n    x = Embedding(max_features+1, emb_size, weights=[embedding_matrix], trainable=False)(inp)\n    x = Bidirectional(CuDNNLSTM(128, return_sequences=True))(x)\n\n    x = Bidirectional(CuDNNLSTM(64, return_sequences=True))(x)\n    x = Bidirectional(CuDNNLSTM(32, return_sequences=True))(x)\n    x = Flatten()(x)\n\n    x = Dense(100, activation=\"relu\")(x)\n    x = Dropout(0.12)(x)\n    x = BatchNormalization()(x)\n\n    x = Dense(1, activation=\"sigmoid\")(x)\n    model = Model(inputs=inp, outputs=x)\n    model.compile(loss='binary_crossentropy', optimizer='adam')\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8a2553577e54902effa8573eb0d1cec2b0e4d34a","scrolled":true},"cell_type":"code","source":"model = lstm_model()\nmodel.summary()\nfrom sklearn.metrics import f1_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aba2b729064b9b3e14b794fdd27bb1b8befe6b12","scrolled":false},"cell_type":"code","source":"from keras.callbacks import ModelCheckpoint\n\nkfold = StratifiedKFold(n_splits=5,random_state=5, shuffle=True)\nscores = []\n\nfor i, (train, valid) in enumerate(kfold.split(X_train, Y)):\n    X_Train, X_val, Y_train, Y_val = X_train[train], X_train[valid], Y[train], Y[valid]\n    filepath=\"weights_best.h5\"\n    checkpoint = ModelCheckpoint(filepath, monitor='val_loss', verbose=2, save_best_only=True, mode='min')\n    callbacks = [checkpoint]\n    model = lstm_model()\n    model.fit(X_Train, Y_train, batch_size=512, epochs=6, validation_data=(X_val, Y_val), verbose=2, callbacks=callbacks, )\n    model.load_weights(filepath)\n    y_pred = model.predict([X_val], batch_size=1024, verbose=2)\n    y_pred = (y_pred>0.5) + 0\n    \n    score = f1_score(Y_val,y_pred)\n    scores.append(score)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5443d2ecfe96643a06cd213b60a96cb300086324"},"cell_type":"code","source":"print(scores)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"218b8a22dd28b8b8f723c92b3a0e596a4bc95a41"},"cell_type":"code","source":"y_test = model.predict([X_test], batch_size=1024, verbose=2)\ny_test = (y_test>0.5)+0\ny_test = y_test.reshape((-1, 1))\nsub['prediction']=y_test\nsub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"24a0f4b197fcf7ff0eeb06f977b297cd4c8c4b83"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}