{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \nimport keras as k\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom keras.layers import LSTM\nfrom keras.layers.embeddings import Embedding\nfrom keras.preprocessing import sequence\nfrom keras.preprocessing.text import Tokenizer\nfrom tqdm import tqdm\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\nprint(os.listdir(\"../input/embeddings\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"df = pd.read_csv('../input/train.csv')\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"90310069442f11be21565675e3e5e667770d1a62"},"cell_type":"code","source":"df.tail()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"34580409f346e683f5f45eb0981b4818d6606d3f"},"cell_type":"code","source":"df['question_text'][1306117]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0c4e16fcb2fafcb77b1ebee907fb497b099e31b0"},"cell_type":"code","source":"df.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8fbc8fd6f1334552d41e0820ff2fb02be51f0e7c"},"cell_type":"code","source":"df[df['question_text'].isnull()]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b7fe44c273f4bf3974c34feabf58a1e2f8552242"},"cell_type":"code","source":"df[df['target']==1].shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a4742003e8e0c8a16a0a97dcfbad8c8f79fb406b"},"cell_type":"code","source":"df[df['target']==0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dac71d505946d6fcd76fe3ec494fd2292adcd90e"},"cell_type":"code","source":"df['question_text'].str.split().str.len().max()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"99ca2cb501cdd87fd137c4c1a4bb9e6470f7a70d"},"cell_type":"code","source":"max_features = 5000\ntokenizer = Tokenizer(num_words=max_features,split=' ')\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c828ba9c0f644016672dc4c543335e880d16b211"},"cell_type":"code","source":"tokenizer.fit_on_texts(df['question_text'].values)\nx = tokenizer.texts_to_sequences(df['question_text'].values)\nx","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3cbb8f0c6dff08f38231ee4312c721f3791d5942"},"cell_type":"code","source":"len(x)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c58725b9beb7811b29d5550f243d5343f78d7659"},"cell_type":"code","source":"X_train = sequence.pad_sequences(x)\nX_train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9eed8728b17b786a1d308b248f6b9a83084952c3"},"cell_type":"code","source":"y_train = df['target'].values\ny_train.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b2a4dec79245a667e637ea1ffd12e4f149ef7cb"},"cell_type":"code","source":"model = Sequential()\nmodel.add(Embedding(max_features,128,input_length=X_train.shape[1]))\nmodel.add(LSTM(256,recurrent_dropout=0.25))\nmodel.add(Dense(1,activation='sigmoid'))\nmodel.compile(loss='binary_crossentropy',optimizer='adam',metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"853202c83696a7a2e3d335f9d228b2c67b6727e6"},"cell_type":"code","source":"model.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9fc813f71ac00fcf04a11d6a9ae51005df6a5c48"},"cell_type":"code","source":"model.fit(X_train,y_train,batch_size=1500,epochs=5,validation_split=0.40,verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2477d45f58600a35438d48867cffb26e58abd61b"},"cell_type":"code","source":"test_df = pd.read_csv('../input/test.csv')\ntest_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"28bc0260b8fee565b6ca768e0545307df8503476"},"cell_type":"code","source":"testx = tokenizer.texts_to_sequences(test_df['question_text'].values)\nX_test = sequence.pad_sequences(testx,maxlen=135)\n\nall_preds = []\ny_pred = model.predict(X_test,batch_size=128,verbose=1)\nall_preds.append(y_pred.flatten())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"de636a88804bd3cb023cf210d5b97ef03e025def"},"cell_type":"code","source":"print(y_pred.shape)\ny_te = (np.array(y_pred.flatten()) > 0.35).astype(np.int)\n\nsubmit_df = pd.DataFrame({\"qid\": test_df[\"qid\"], \"prediction\": y_te})\nsubmit_df.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}