{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D, CuDNNLSTM\nfrom keras.layers import Bidirectional, GlobalMaxPool1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers","execution_count":1,"outputs":[{"output_type":"stream","text":"Using TensorFlow backend.\n","name":"stderr"}]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\ntest_df = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",train_df.shape)\nprint(\"Test shape : \",test_df.shape)","execution_count":2,"outputs":[{"output_type":"stream","text":"Train shape :  (1306122, 3)\nTest shape :  (375806, 2)\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"## split to train and val\ntrain_df, val_df = train_test_split(train_df, test_size=0.1, random_state=2018)\n\n## some config values \nembed_size = 300 # how big is each word vector\nmax_features = 50000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 100 # max number of words in a question to use\n\n## fill up the missing values\ntrain_X = train_df[\"question_text\"].fillna(\"_na_\").values\nval_X = val_df[\"question_text\"].fillna(\"_na_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_na_\").values\n\n## Tokenize the sentences\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\n## Pad the sentences \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\n## Get the target values\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values","execution_count":3,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(len(train_df))\nprint(len(val_df))","execution_count":4,"outputs":[{"output_type":"stream","text":"1175509\n130613\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"inp = Input(shape=(maxlen,))\nx = Embedding(max_features, embed_size)(inp)\nx = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(1, activation=\"sigmoid\")(x)\nmodel = Model(inputs=inp, outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\nprint(model.summary())","execution_count":5,"outputs":[{"output_type":"stream","text":"WARNING:tensorflow:From /opt/conda/lib/python3.6/site-packages/tensorflow/python/framework/op_def_library.py:263: colocate_with (from tensorflow.python.framework.ops) is deprecated and will be removed in a future version.\nInstructions for updating:\nColocations handled automatically by placer.\nWARNING:tensorflow:From /opt/conda/lib/python3.6/site-packages/keras/backend/tensorflow_backend.py:3445: calling dropout (from tensorflow.python.ops.nn_ops) with keep_prob is deprecated and will be removed in a future version.\nInstructions for updating:\nPlease use `rate` instead of `keep_prob`. Rate should be set to `rate = 1 - keep_prob`.\n_________________________________________________________________\nLayer (type)                 Output Shape              Param #   \n=================================================================\ninput_1 (InputLayer)         (None, 100)               0         \n_________________________________________________________________\nembedding_1 (Embedding)      (None, 100, 300)          15000000  \n_________________________________________________________________\nbidirectional_1 (Bidirection (None, 100, 128)          140544    \n_________________________________________________________________\nglobal_max_pooling1d_1 (Glob (None, 128)               0         \n_________________________________________________________________\ndense_1 (Dense)              (None, 16)                2064      \n_________________________________________________________________\ndropout_1 (Dropout)          (None, 16)                0         \n_________________________________________________________________\ndense_2 (Dense)              (None, 1)                 17        \n=================================================================\nTotal params: 15,142,625\nTrainable params: 15,142,625\nNon-trainable params: 0\n_________________________________________________________________\nNone\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"## Train the model \nmodel.fit(train_X, train_y, batch_size=512, epochs=2, validation_data=(val_X, val_y))","execution_count":6,"outputs":[{"output_type":"stream","text":"WARNING:tensorflow:From /opt/conda/lib/python3.6/site-packages/tensorflow/python/ops/math_ops.py:3066: to_int32 (from tensorflow.python.ops.math_ops) is deprecated and will be removed in a future version.\nInstructions for updating:\nUse tf.cast instead.\nWARNING:tensorflow:From /opt/conda/lib/python3.6/site-packages/tensorflow/python/ops/math_grad.py:102: div (from tensorflow.python.ops.math_ops) is deprecated and will be removed in a future version.\nInstructions for updating:\nDeprecated in favor of operator or tf.math.divide.\nTrain on 1175509 samples, validate on 130613 samples\nEpoch 1/2\n1175509/1175509 [==============================] - 89s 76us/step - loss: 0.1223 - acc: 0.9534 - val_loss: 0.1080 - val_acc: 0.9562\nEpoch 2/2\n1175509/1175509 [==============================] - 86s 73us/step - loss: 0.0981 - acc: 0.9605 - val_loss: 0.1079 - val_acc: 0.9574\n","name":"stdout"},{"output_type":"execute_result","execution_count":6,"data":{"text/plain":"<keras.callbacks.History at 0x7fd210d42710>"},"metadata":{}}]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_noemb_val_y = model.predict([val_X], batch_size=1024, verbose=1)\nbest_score = 0\nbest_threshold = None\nfor thresh in np.arange(0.1, 0.501, 0.01):\n    thresh = np.round(thresh, 2)\n    score = metrics.f1_score(val_y, (pred_noemb_val_y>thresh))\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, score))\n    if score > best_score:\n        best_score = score\n        best_threshold = thresh\n    \nprint('Best score: {0}, best threshold: {1}'.format(best_score, best_threshold))","execution_count":12,"outputs":[{"output_type":"stream","text":"130613/130613 [==============================] - 2s 17us/step\nF1 score at threshold 0.1 is 0.5981284229591617\nF1 score at threshold 0.11 is 0.6070876914250408\nF1 score at threshold 0.12 is 0.6146689549272875\nF1 score at threshold 0.13 is 0.6206165614487812\nF1 score at threshold 0.14 is 0.6261752538548326\nF1 score at threshold 0.15 is 0.6313624186758515\nF1 score at threshold 0.16 is 0.6348402182385035\nF1 score at threshold 0.17 is 0.6380088254251575\nF1 score at threshold 0.18 is 0.6406045340050378\nF1 score at threshold 0.19 is 0.6440886069473577\nF1 score at threshold 0.2 is 0.6469489414694894\nF1 score at threshold 0.21 is 0.648384040425308\nF1 score at threshold 0.22 is 0.6505149687816852\nF1 score at threshold 0.23 is 0.6520917678812416\nF1 score at threshold 0.24 is 0.6534740064505548\nF1 score at threshold 0.25 is 0.6542056074766356\nF1 score at threshold 0.26 is 0.6555431131019037\nF1 score at threshold 0.27 is 0.6554279572325621\nF1 score at threshold 0.28 is 0.6568941823179112\nF1 score at threshold 0.29 is 0.6561342592592593\nF1 score at threshold 0.3 is 0.6553052410412046\nF1 score at threshold 0.31 is 0.6542531181651593\nF1 score at threshold 0.32 is 0.6527246653919694\nF1 score at threshold 0.33 is 0.6511852343325893\nF1 score at threshold 0.34 is 0.6496257075041081\nF1 score at threshold 0.35 is 0.6479911537043863\nF1 score at threshold 0.36 is 0.6454534176901129\nF1 score at threshold 0.37 is 0.6435011269722013\nF1 score at threshold 0.38 is 0.6427261806916609\nF1 score at threshold 0.39 is 0.6402349486049926\nF1 score at threshold 0.4 is 0.6381744343453877\nF1 score at threshold 0.41 is 0.6369493069564652\nF1 score at threshold 0.42 is 0.6350019708316911\nF1 score at threshold 0.43 is 0.6322777667131382\nF1 score at threshold 0.44 is 0.6292722155207077\nF1 score at threshold 0.45 is 0.6254481499019143\nF1 score at threshold 0.46 is 0.6221676221676222\nF1 score at threshold 0.47 is 0.6187357827255806\nF1 score at threshold 0.48 is 0.6152882205513783\nF1 score at threshold 0.49 is 0.610095613048369\nF1 score at threshold 0.5 is 0.6034188034188034\nBest score: 0.6568941823179112, best threshold: 0.28\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_noemb_test_y = model.predict([test_X], batch_size=1024, verbose=1)","execution_count":13,"outputs":[{"output_type":"stream","text":"375806/375806 [==============================] - 6s 15us/step\n","name":"stdout"}]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred_test_y = (pred_noemb_test_y > best_threshold).astype(int)\nout_df = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\nout_df['prediction'] = pred_test_y\nout_df.to_csv(\"submission.csv\", index=False)","execution_count":14,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.4","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}