{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\".\"))\nprint(\"hello\")\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom sklearn.model_selection import train_test_split\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing import sequence\nfrom keras.models import Model\nfrom keras.layers import LSTM, Activation, Dense, Dropout, Input, Embedding\nfrom keras.optimizers import RMSprop\nfrom keras.callbacks import EarlyStopping\nimport numpy as np\nimport nltk as nl\n\n\n%matplotlib inline\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7f028548f8fa017beec5918f8d0e34e6051b8b0b"},"cell_type":"code","source":"data = pd.read_csv(\"../input/train.csv\")\nprint(data.info())\nprint(data.target.value_counts())\ntrain_text = data['question_text'].values[:50]\ntrain_target = data['target'].values[:50]\ntest_text = data['question_text'].values[50:60]\ntest_target = data['target'].values[50:60]\n\n# here I am just taking small chunk of data to commit fast","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6c4198a5eb8c8f94643b5a86a35ed0d249845c26"},"cell_type":"code","source":"def get_one_hot_y(target):\n    y =[]\n    for t in target:\n        z = np.zeros(2, dtype=float)\n        z[t] = 1\n        y.append(z)\n    y = np.array(y)\n    return y\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cf5c24a411e0b7f8d549b9286cfd565fa5f65251"},"cell_type":"code","source":"from gensim.models import KeyedVectors\ntext_model = KeyedVectors.load_word2vec_format('../input/embeddings/wiki-news-300d-1M/wiki-news-300d-1M.vec')\nprint(\"done loading word vec model \")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"424e64c059912b44bc20d6246e418de74e78124c"},"cell_type":"code","source":"print (text_model.most_similar('desk'))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"988c47ae7b7aa8a84acf3c40531042047d534fab"},"cell_type":"code","source":"def sent_vectorizer(sentence, text_model):\n    words = nl.word_tokenize(sentence)\n    sent_vec = np.mean([text_model[w] for w in words if w in text_model]\n                    or [np.zeros(300)], axis=0)\n    return sent_vec\n    ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cedc551b5370b8051b91c528159b67e2d084915e"},"cell_type":"code","source":"def text_vectorizer(text, text_model):\n    X = np.zeros((len(text),300),dtype=float)\n    for i,s in enumerate(text):\n        if i% 1000 == 0:\n            pass\n            #print (i)\n        vec = sent_vectorizer(s, text_model)\n        #print (vec.shape)\n        X[i] = vec\n    return X","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b088174b0a9f9d5609dbf2de9896c0171db2937a"},"cell_type":"code","source":"max_words = 1000\nmax_len = 300\nX = text_vectorizer(train_text, text_model)\nprint( X.shape)\n\nprint(\"done....\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b118f8654f68613b52d422fdb29c607614dd31bb"},"cell_type":"code","source":"y = get_one_hot_y(train_target)\nprint(\"done..\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a52ab63532cb8b68e61663c1c773172b32bde024"},"cell_type":"code","source":"def RNN():\n    inputs = Input(name='inputs',shape=[max_len])\n    layer = Embedding(max_words,50,input_length=max_len)(inputs)\n    layer = LSTM(64)(layer)\n    layer = Dense(256,name='FC1')(layer)\n    layer = Activation('relu')(layer)\n    layer = Dropout(0.5)(layer)\n    layer = Dense(2,name='out_layer')(layer)\n    layer = Activation('softmax')(layer)\n    model = Model(inputs=inputs,outputs=layer)\n    return model\nprint(\"done ......\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"04ee4940314f43cbc2800a95fd383c49ac061406"},"cell_type":"code","source":"model = RNN()\nmodel.summary()\nmodel.compile(loss='binary_crossentropy',optimizer=RMSprop(),metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0861fb80e21ac4b5242e06fecedc13d2b64275f4"},"cell_type":"code","source":"model.fit(X,y,batch_size=128,epochs=10,\n          validation_split=0.2,callbacks=[EarlyStopping(monitor='val_loss',min_delta=0.0001)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"428d0b6da7117dc1878fbb37536819a52a4cac8c"},"cell_type":"code","source":"X_test = text_vectorizer(test_text,text_model)\ny_test = get_one_hot_y(test_target)\naccr = model.evaluate(X_test,y_test)\nprint('Test set\\n  Loss: {:0.3f}\\n  Accuracy: {:0.3f}'.format(accr[0],accr[1]))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc43a3a338a5a13e21f51e9cfc656f74b3c5814a"},"cell_type":"code","source":"submit_data = pd.read_csv(\"../input/test.csv\")\nsubmit_data.info()\nsubmit_text = submit_data[\"question_text\"]\nX_submit = text_vectorizer(submit_text,text_model)\nprint(\"done ....\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b50ff8b9416669cbb65c14a15810dd93e97e51bb"},"cell_type":"code","source":"prediction = model.predict(X_submit)\nprint(\"done ..\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dbeccd926977764018e9ae0e99dfb4824f169ac8"},"cell_type":"code","source":"p = np.argmax(prediction, axis =1)\nprint (p)\nmy_submission = pd.DataFrame( {'qid': submit_data.qid , 'prediction': p } )\nmy_submission.to_csv('submission.csv', index=False)\nprint(\"done....\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b37b9ea0102a443cc5632c5ad62644aca5f6e9c4"},"cell_type":"code","source":"print(os.listdir(\".\"))\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ae7879087207fe856f3a2c946a8ddad08c4c7a13"},"cell_type":"code","source":"l = [line for line in open('submission.csv')]\n","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}