{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"train_data = pd.read_csv('../input/train.csv')\ntest_data = pd.read_csv('../input/test.csv')\nprint('The shape of train data is:', train_data.shape)\nprint('The shape of test data is:', test_data.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"88fdedacdec3293ec7f7473be8b5cae0ca8b9879","scrolled":false},"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\n\ntfidf_vect = TfidfVectorizer(ngram_range = (1,3) ,stop_words = 'english')\n\ntfidf_vect.fit_transform(train_data.question_text.tolist() + test_data.question_text.tolist())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e0c49fbd8824e0a56a078523a333fa9c997bc268"},"cell_type":"code","source":"train_tfidf = tfidf_vect.transform(train_data.question_text.values.tolist())\ntest_tfidf = tfidf_vect.transform(test_data.question_text.values.tolist())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"84f096ca8f707a4e2e66979c748d8112d5673bf8"},"cell_type":"code","source":"train_y = train_data.target.values\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b1afff2546617d571dffd793c3b4c0c49fcdd34b"},"cell_type":"code","source":"#average question text length\nques_text = train_data.question_text.values.tolist() + test_data.question_text.values.tolist()\nj = 0\nfor i in ques_text:\n    j += len(i)\nprint(int(j/len(ques_text)))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"97a8121bc89d9be24c19a4950b8647ca5639b080"},"cell_type":"code","source":"import keras\nfrom keras.layers import Dense, Input, Embedding, LSTM, Dropout, Activation\nfrom keras.models import Model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0d4caaeda88653589556f6895c4670615c094070"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nfrom keras.preprocessing.text import Tokenizer, text_to_word_sequence\nfrom keras.layers import Bidirectional\ntest_df = test_data\n\ntrain_df, val_df = train_test_split(train_data, test_size=0.1, random_state=2018)\n\nembed_dim = 300 #no. of dimensions in word vector\nmax_features = 50000 #no. of unique words to use\nmaxlen = 100 #max number of words in a question to use\n\n## fill up the missing values\ntrain_X = train_df[\"question_text\"].fillna(\"_na_\").values\nval_X = val_df[\"question_text\"].fillna(\"_na_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_na_\").values\n\n## Tokenize the sentences\ntokenizer = Tokenizer(num_words=max_features, filters='!\"#$%&()*+,-./:;<=>?@[\\]^_`{|}~')\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\n# we now have sequences instead of text. Whatever we use ahead i.e. test_X, train_x, val_X are all sequences\n\n\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"937f6ebf12358c1e6f8c1ec460583875b97636a1"},"cell_type":"code","source":"from keras.preprocessing.sequence import pad_sequences\n##Padding the sentences(sequence)\ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"546c80da29f76edc467625acb7e3b1585ae93b60"},"cell_type":"code","source":"train_X.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59d9681b1d373f54e18966fda79414c97f3f4100"},"cell_type":"code","source":"## Get the target values\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e47d000f1db44cf852cb6d0b24be32b77f25b373"},"cell_type":"code","source":"from keras.layers import Dense, GlobalMaxPool1D\nfrom keras.models import Model\ninp = Input(shape=(maxlen,))\n#print(inp.shape)\nx = Embedding(max_features,embed_dim)(inp)\n#print(x.shape)\nx = LSTM(embed_dim, activation='relu')(x)\n#print(x.shape)\nx = Dense(1, activation='sigmoid', input_shape=(embed_dim,))(x)\n#print(x.shape)\n\nmodel = Model(inputs=inp,outputs=x)\nmodel.compile(loss='binary_crossentropy', optimizer='adam',metrics=['accuracy'])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"84467756ea83641407df71d7b351f6cf05933d04"},"cell_type":"code","source":"model.summary()\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c51be31744c37a21bed4ada741fadbfd7ed632aa"},"cell_type":"code","source":"model.summary()\nmodel.fit(train_X,train_y, epochs=2, validation_data=(val_X, val_y))","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}