{"cells":[{"metadata":{"_uuid":"b81918dc6eb9a324113f33d300a2fc855a9b7df0"},"cell_type":"markdown","source":"Simple text classification, following example https://www.tensorflow.org/tutorials/keras/basic_text_classification\nThe idea is to train an embedding layer, and use average1D to average the latent embedding vectors of the words in each sentence. \n\nAlso borrowed code for checking f1 score from https://www.kaggle.com/sudalairajkumar/a-look-at-different-embeddings\n\nPotential improving point.\n\n1. Dropout\n2. Regularization\n3. More complex structure\n4. GRU\n5. Capsule\n6. Attention\n7. Stacking"},{"metadata":{"_uuid":"35e3c4bef61ae0248f3296b5f2234c97af6cee58"},"cell_type":"markdown","source":"| | f1 (validation) | f1 (public LB) |\n|---|---|---|\n|without any RNN layer |0.626|0.613|\n|one Bidirectional LSTM |0.633|0.61|\n|one Bi-LSTM w dropout |0.62|0.61|\n|one Bi-LSTM w dropout, regularization  |0.62|0.61|"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"scrolled":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport itertools\nimport tensorflow as tf\nfrom tensorflow import keras\nimport numpy as np\n\nprint(tf.__version__)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f99383dca73574445da4a7a7c349abf651ba3504"},"cell_type":"code","source":"STRING_LENGTH_MAX =256","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"scrolled":true},"cell_type":"code","source":"train_df = pd.read_csv(\"../input/train.csv\")\ntest_df = pd.read_csv(\"../input/test.csv\")\nprint(\"Train shape : \",train_df.shape)\nprint(\"Test shape : \",test_df.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d13d1ee8b5eab4b04a8999509d6e28423d630374"},"cell_type":"code","source":"train_df['question_text'].apply(len).max()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"474cfc5b8bbc32501c6f8899b6510972529d02e9"},"cell_type":"code","source":"train_df.head(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"db267c4582a2c8fe743878a8532058271cf101e3"},"cell_type":"code","source":"# train = train_df.sample(10000)\ntrain = train_df.sample(1000000)\nx_train = train['question_text']\ny_train = train['target']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c7774e3d5a457903cd0c8093a7eb6df5ca67720e"},"cell_type":"code","source":"import re\n\nx_list_words = [re.findall(r'\\w+',x) for x in x_train.values]\nvocab = set( itertools.chain.from_iterable(x_list_words) )\nword_to_index = {w:(i+3) for i,w in enumerate(vocab)}\nword_to_index[\"<PAD>\"] = 0\nword_to_index[\"<START>\"] = 1\nword_to_index[\"<UNK>\"] = 2  # unknown\nword_to_index[\"<UNUSED>\"] = 3\nindex_to_word = {v:k for k,v in word_to_index.items()}\nlen(index_to_word)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"89a927dd247fc14112b3e0a1c232a50c3874cbed"},"cell_type":"code","source":"x_train_int = [\n    [word_to_index[w] for w in words]\n    for words in x_list_words\n]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8c96433fbe2931df833339747cb48f36f1782605"},"cell_type":"code","source":"validate = train_df.loc[~train_df.index.isin(train.index)]\n# validate = train_df.loc[~train_df.index.isin(train.index)].sample(10000)\nx_val = validate['question_text']\ny_val = validate['target']\n\ndef encode_x(x):\n    x_list_words = [re.findall(r'\\w+',x) for x in x.values]\n    x_list_int = [\n        [word_to_index.get(w,2) for w in words]\n        for words in x_list_words\n    ]\n    return x_list_int\n\nx_val_int = encode_x(x_val)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e0cce9e482ec279aebe12c54d479dace08e6f52e"},"cell_type":"code","source":"def decode_ints(list_ints):\n    return ' '.join([index_to_word.get(i, '?') for i in list_ints])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8b7ec78b0c9287399f64f39383af8ad1e3825f57"},"cell_type":"code","source":"decode_ints(x_val_int[0])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b9c10a7b7c3858b3b8212f8cff542f6efa6f3c86"},"cell_type":"code","source":"x_train.apply(len).max()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5b8eaffe59dc2aff033171d687a89ffcbe611130"},"cell_type":"code","source":"x_val.apply(len).max()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"668cba950a94f585517f7583202a2819cba6f436"},"cell_type":"markdown","source":"Simple text classification, following example https://www.tensorflow.org/tutorials/keras/basic_text_classification\nThe idea is to train an embedding layer, and use average1D"},{"metadata":{"trusted":true,"_uuid":"ed5a3c756c4870d5336b34842a2e313ccf25603b"},"cell_type":"code","source":"train_data = keras.preprocessing.sequence.pad_sequences(x_train_int,\n                                                        value=word_to_index[\"<PAD>\"],\n                                                        padding='post',\n                                                        maxlen=STRING_LENGTH_MAX)\n\ntest_data = keras.preprocessing.sequence.pad_sequences(x_val_int,\n                                                       value=word_to_index[\"<PAD>\"],\n                                                       padding='post',\n                                                       maxlen=STRING_LENGTH_MAX)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6c2514d66270dfc276de6d2d1f38dc8376f6a69c","scrolled":true},"cell_type":"code","source":"test_data.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"42f94d5849f5924a3659aa77393773d8ee6b3e43"},"cell_type":"code","source":"# input shape is the vocabulary count used for the movie reviews (10,000 words)\nvocab_size = len(word_to_index)+1\n\nmodel = keras.Sequential()\nmodel.add(keras.layers.Embedding(vocab_size, 64, input_length=STRING_LENGTH_MAX))\nmodel.add(keras.layers.Bidirectional(keras.layers.CuDNNLSTM(64)))\nmodel.add(keras.layers.Dropout(0.3))\n# model.add(keras.layers.GlobalAveragePooling1D())\nmodel.add(keras.layers.Dense(16, activation=tf.nn.relu))\nmodel.add(keras.layers.Dense(1, activation=tf.nn.sigmoid))\nmodel.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"832b2a418363733a5cb0a9309475adf7642b5bd6"},"cell_type":"code","source":"model.compile(optimizer=tf.train.AdamOptimizer(),\n              loss='binary_crossentropy',\n              metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bb3d03f526f8c00bbf9f12da449cd17533ccae81"},"cell_type":"markdown","source":"To save time on commit, only took a small sample. If using all training data, the best f1 score is 0.626. "},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"65d0667487153dcd9401420bd7e1fb5c379790b1"},"cell_type":"code","source":"early_stop = keras.callbacks.EarlyStopping(monitor='val_loss',\n                              min_delta=0,\n                              patience=0,\n                              verbose=0, mode='auto')\n\nhistory = model.fit(train_data,\n                y_train,\n                epochs=100,\n                batch_size=1024,\n                validation_data=(test_data, y_val),\n                verbose=1,\n                callbacks=[early_stop,])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9575b08bf35afc2918aa227310035fc983f0310b","scrolled":false},"cell_type":"code","source":"from sklearn import metrics\ny_pred = model.predict(test_data, batch_size=2048, verbose=1)\nfor thresh in np.arange(0.1, 0.9, 0.01):\n    thresh = np.round(thresh, 2)\n    print(\"F1 score at threshold {0} is {1}\".format(thresh, metrics.f1_score(y_val, (y_pred>thresh).astype(int))))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"1ed0c97d4ba4be0afeb086cfc0ef74af227f4fce"},"cell_type":"markdown","source":"Make predictions"},{"metadata":{"trusted":true,"_uuid":"c968aa31b753d223b22a7a7c20e941ce52b8e87e"},"cell_type":"code","source":"x_test = test_df['question_text']\nx_test_int = encode_x(x_test)\nsub_data = keras.preprocessing.sequence.pad_sequences(x_test_int,\n                                                       value=word_to_index[\"<PAD>\"],\n                                                       padding='post',\n                                                       maxlen=STRING_LENGTH_MAX)\npred_val_y = model.predict([sub_data], batch_size=1024, verbose=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c06d74e87bb45e9a770ecce381a7dbf649d1471c"},"cell_type":"code","source":"sub = pd.read_csv('../input/sample_submission.csv')\nsub.prediction = pred_val_y > 0.245\nsub.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7399be1c82646142cd859d8b484c74533e99c60d"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}