{"cells":[{"metadata":{"_uuid":"297256e0e1bb2eea93d76119b18425b1a8fcaac9"},"cell_type":"markdown","source":" ** 引用函式庫 **"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D\nfrom keras.layers import Bidirectional, GlobalMaxPool1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0c798a8376b95063d8ef2ac8c25b53889d3bb7d4"},"cell_type":"markdown","source":"**讀取訓練資料與測試資料**"},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\ntest_df = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",train_df.shape)\nprint(\"Test shape : \",test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ba5a1b8109dee2c9fbc628d5da4a7c3447d42fb8"},"cell_type":"code","source":"## 把訓練資料集切割為train 和 val 樣本\ntrain_df, val_df = train_test_split(train_df, test_size=0.1, random_state=2018)\n\n## 參數設定\nembed_size = 300 # how big is each word vector\nmax_features = 50000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 100 # max number of words in a question to use\n\n## 把缺失的值 用 '_na_' 補上\ntrain_X = train_df[\"question_text\"].fillna(\"_na_\").values\nval_X = val_df[\"question_text\"].fillna(\"_na_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_na_\").values\n\n## Tokenize(斷詞) the sentences(句子)\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\n## Pad the sentences 截長補短 \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\n## Get the target values 取得target\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"2a5f324273d8e4726a6f0f9206170845d5ead890"},"cell_type":"markdown","source":"**前置步驟(預處理)完成，首先訓練雙向GRU model，先不使用任何預訓練的embedding看看效果如何。**"},{"metadata":{"trusted":true,"_uuid":"3cfab26c6cced33ef7ab84f0d36997113131d530"},"cell_type":"code","source":"inp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size)(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ef1e1015e7c3ab5bc5d9774e49820c4b286d7847","scrolled":true},"cell_type":"code","source":"## Train the model \nmodel.fit(train_X, train_y, batch_size=512, epochs=4, validation_data=(val_X, val_y))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a72ba82481de9f96c62c334cc40bd1d134d38e2d"},"cell_type":"markdown","source":"**計算各門檻值所得到的F1 score**"},{"metadata":{"trusted":true,"_uuid":"47b63dca0247a08a808db7ae6eea33065c554948"},"cell_type":"code","source":"pred_noemb_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_noemb_val_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"383e51177bf33da0b8fee42dd6a093908b808f64"},"cell_type":"markdown","source":"**把測試資料預測的結果存入變數中**"},{"metadata":{"trusted":true,"_uuid":"a88df747f43259bab84447b50e45aa9e978f2cee"},"cell_type":"code","source":"pred_noemb_test_y = model.predict([test_X], batch_size=1024, verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"f3831ee610fee119c9851b5ca29b2d80b102ae6a"},"cell_type":"markdown","source":"**釋放記憶體**"},{"metadata":{"trusted":true,"_uuid":"a36a071fb50f6c120e099b5fe27ad6ac977f1125"},"cell_type":"code","source":"del model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"717f7dcd5ccf71e83d0f062e221c46db39845e4d"},"cell_type":"markdown","source":"**查看embedding**\n"},{"metadata":{"trusted":true,"_uuid":"b9d263852f653e466e24f9827548d7d1a7ee7262"},"cell_type":"code","source":"!ls ../input/embeddings/","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7f6893c994ec70de530a648f9ee93e6f3a8cc4d7"},"cell_type":"markdown","source":"**Glove Embeddings:**"},{"metadata":{"trusted":true,"_uuid":"df376806e4842ad576f2043760258c95f0d8e6e9"},"cell_type":"code","source":"## 讀取embedding \nEMBEDDING_FILE = '../input/embeddings/glove.840B.300d/glove.840B.300d.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\n## 轉換成字典形式 \"字\": [ 在各維度的向量 ] ex: {',': array([-0.082752 ,  0.67204  ... ) ,  '.': array([ 0.012001 ... ])}\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE))\n\n## 各維度的向量\nall_embs = np.stack(embeddings_index.values())\n## 計算平均值與標準差\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\n## 維度 = 300\nembed_size = all_embs.shape[1]\n\n## 字的index ex: {'the': 1, 'what': 2, ... }\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\n## 高斯隨機 numpy.random.normal(平均值,標準差,(輸出的shape))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\n## 把出現過的字與該字在embedding中的向量結合在一起\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n\n## 把詞向量也放入model中        \ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a560ab0dbab9cf6fdbdae6721ec030e300f19d78"},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=2, validation_data=(val_X, val_y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ff43855164472de035a5a1d80b3db4838684701a"},"cell_type":"code","source":"pred_glove_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_glove_val_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d51ff8ed6a87b488fec3ac84ca50df661d7c8193"},"cell_type":"code","source":"pred_glove_test_y = model.predict([test_X], batch_size=1024, verbose=1)\n## 投票\npred_glove_test_y = (pred_glove_test_y>0.35).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"39d4fedab4ac170863a0ee1ca3aa9be1ee58fe02"},"cell_type":"code","source":"del word_index, embeddings_index, all_embs, embedding_matrix, model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bc6bab22dd12a09378f4b8b159cb7a5d88a3e7c0"},"cell_type":"markdown","source":"**Wiki News FastText Embeddings:**"},{"metadata":{"trusted":true,"_uuid":"6f3d0fd28dd2b04eaccb732b96b872e5a223d962"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE) if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n        \ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"47238831a4701c8a67dc7ecb130ac1402baf7bb2"},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=2, validation_data=(val_X, val_y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b7ab4100f723ad535528865b1edc7896bce80223"},"cell_type":"code","source":"pred_fasttext_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_fasttext_val_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3216362afb0f49579d287a06f13adf8cd7d8b0cf"},"cell_type":"code","source":"pred_fasttext_test_y = model.predict([test_X], batch_size=1024, verbose=1)\n## 投票\npred_fasttext_test_y = (pred_fasttext_test_y>0.33).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f24f9753ff1d933fa4f75a0ba34df305632d6e93"},"cell_type":"code","source":"del word_index, embeddings_index, all_embs, embedding_matrix, model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4ca44ac68bf404b9c26e07fbcc9c8ac793e04510"},"cell_type":"markdown","source":"**Paragram Embeddings:**"},{"metadata":{"trusted":true,"_uuid":"25ec1aac4aedbf431a2d30de64030ce8e3203c18"},"cell_type":"code","source":"EMBEDDING_FILE = '../input/embeddings/paragram_300_sl999/paragram_300_sl999.txt'\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in open(EMBEDDING_FILE, encoding=\"utf8\", errors='ignore') if len(o)>100)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector\n        \ninp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size, weights=[embedding_matrix])(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc188f2787ea7b98d3a40953a95a5fc09ff2764d"},"cell_type":"code","source":"model.fit(train_X, train_y, batch_size=512, epochs=2, validation_data=(val_X, val_y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9abdfd1cf15257f2c0c2181a13327796e8d4584e"},"cell_type":"code","source":"pred_paragram_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_paragram_val_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"99cb9f6145da909bd7436e46d47547efc097499d"},"cell_type":"code","source":"pred_paragram_test_y = model.predict([test_X], batch_size=1024, verbose=1)\n## 投票\npred_paragram_test_y = (pred_paragram_test_y>0.34).astype(int)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"af087d21bdb4358701e31aded6b522accd5a8a64"},"cell_type":"code","source":"del word_index, embeddings_index, all_embs, embedding_matrix, model, inp, x\nimport gc; gc.collect()\ntime.sleep(10)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"e1312b7a4c3b67ca4ebd26fb083dbac3b6635dc2"},"cell_type":"markdown","source":"** 有使用embedding的結果顯然比沒有使用好，各個embedding結果差異不大，因此各取1/3混和在一起。**"},{"metadata":{"trusted":true,"_uuid":"449bc59fdc9a719aa0759ac51a4481df113604ca"},"cell_type":"code","source":"# pred_val_y = 0.33*pred_glove_val_y + 0.33*pred_fasttext_val_y + 0.34*pred_paragram_val_y\npred_val_y = 1.5*pred_glove_val_y + 1.3*pred_fasttext_val_y + 0.9*pred_paragram_val_y \npred_val_y = (pred_val_y>=2).astype(int)\nprint(metrics.f1_score(val_y, (pred_val_y)))\n#for thresh in np.arange(0.1, 0.501, 0.01):\n#    thresh = np.round(thresh, 2)\n#    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(val_y, (pred_val_y>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"4fdbeffc0f84643d2832eec49234bd9d6c6e216b"},"cell_type":"markdown","source":"** 混和的結果分數比只使用各個embedding還要高，準備submission。 **"},{"metadata":{"trusted":true,"_uuid":"c90fb4a4ef1b3b2ea06563a6901deac1b38822f3"},"cell_type":"code","source":"#pred_test_y = 0.33*pred_glove_test_y + 0.33*pred_fasttext_test_y + 0.34*pred_paragram_test_y\n#pred_test_y = (pred_test_y>0.35).astype(int)\npred_test_y = 1.5*pred_glove_test_y + 1.3*pred_fasttext_test_y + 0.9*pred_paragram_test_y\npred_test_y = (pred_test_y>=2).astype(int)\nout_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}