{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport re\nimport string\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\nimport nltk\nimport os\nimport gc\nfrom keras.preprocessing import sequence,text\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.models import Sequential\nfrom keras.layers import Input, Dense,Dropout,Embedding,LSTM, CuDNNGRU, Conv1D,GlobalMaxPooling1D,Flatten,MaxPooling1D,GRU,GlobalMaxPool1D,SpatialDropout1D,Bidirectional\nfrom keras.callbacks import EarlyStopping\nfrom keras.utils import to_categorical\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers\nfrom keras.losses import categorical_crossentropy\nfrom keras.optimizers import Adam\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import accuracy_score,confusion_matrix,classification_report,f1_score\nimport matplotlib.pyplot as plt\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n\ntest = pd.read_csv(\"../input/test.csv\")\ntrain = pd.read_csv(\"../input/train.csv\")","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"from nltk.corpus import stopwords\nimport string\npunctuations = string.punctuation\nstopword = stopwords.words(\"english\")\ndef clean(text):\n    \n    lower_text = text.lower()\n    \n    text = \"\".join(w for w in lower_text if w not in punctuations)\n    \n    words = text.split()\n    words = [w for w in words if w not in stopword]\n    res = \" \".join(words)\n    return res\nclean(\"this is a test!\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f87634c749965850acb9418c7c4b159f044c7e7a"},"cell_type":"code","source":"train['cleaned'] = train['question_text'].apply(clean)\ntest['cleaned'] = test['question_text'].apply(clean)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"59b786168e7134b41d23800aca8591ee2372dca6"},"cell_type":"code","source":"target = to_categorical(train['target']) \n#target = train['label']\nx_train, x_val, y_train, y_val  = train_test_split(train['cleaned'], target, test_size=0.2, random_state=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"61c39d1b71d26f44c5aefb8e7aa55ded6367df12"},"cell_type":"code","source":"words = ' '.join(x_train)\nwords = nltk.word_tokenize(words)\ndist = nltk.FreqDist(words)\nnum_unique_words = len(dist)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b2e7097a820b04b804809c07626563a2e5434c6d"},"cell_type":"code","source":"r_len = []\nfor w in x_train:\n    word=nltk.word_tokenize(w)\n    l=len(word)\n    r_len.append(l)\nmax_len = np.max(r_len)\nmax_len","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"501608f89ba411453d5635ee5e8eb32705c22225"},"cell_type":"code","source":"max_features = num_unique_words\nmax_words = max_len\nbatch_size = 128\nembed_dim = 300","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"286889f8252f5a9b2dab589f3a05e395776dd4ba"},"cell_type":"code","source":"tokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(x_train))\nx_train = tokenizer.texts_to_sequences(x_train)\nx_val = tokenizer.texts_to_sequences(x_val)\nx_test = tokenizer.texts_to_sequences(test['cleaned'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e6c9beb6a0340dbcbb5b878f960117d35aaaf0b7"},"cell_type":"code","source":"x_train = sequence.pad_sequences(x_train, maxlen=max_words)\nx_val = sequence.pad_sequences(x_val, maxlen=max_words)\nx_test = sequence.pad_sequences(x_test, maxlen=max_words)\nprint(x_train.shape,x_val.shape,x_test.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"87a82e37ef5e891fb43f59196eb9227059313dba"},"cell_type":"code","source":"from imblearn.pipeline import make_pipeline\nfrom imblearn.over_sampling import ADASYN, SMOTE, RandomOverSampler\nfrom imblearn.under_sampling import RandomUnderSampler\nros = RandomUnderSampler(random_state=777)\nX_ROS, y_ROS = ros.fit_sample(x_train, y_train)\n#x_train = X_ROS\n#y_train = y_ROS","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a1d0f809d031d93f7db036f13bab3e9230087d9d"},"cell_type":"code","source":"EMBEDDING_FILE =open(\"../input/embeddings/glove.840B.300d/glove.840B.300d.txt\", encoding=\"utf8\") \n\ndef get_coefs(word,*arr): return word, np.asarray(arr, dtype='float32')\nembeddings_index = dict(get_coefs(*o.split(\" \")) for o in EMBEDDING_FILE)\n\nall_embs = np.stack(embeddings_index.values())\nemb_mean,emb_std = all_embs.mean(), all_embs.std()\nembed_size = all_embs.shape[1]\n\nword_index = tokenizer.word_index\nnb_words = min(max_features, len(word_index))\nembedding_matrix = np.random.normal(emb_mean, emb_std, (nb_words, embed_size))\nfor word, i in word_index.items():\n    if i >= max_features: continue\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None: embedding_matrix[i] = embedding_vector","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d60e86a17d7a374a59a50aedb36259e00018aa6d"},"cell_type":"code","source":"inp = Input(shape=(max_words,))\nx = Embedding(max_features, embed_dim, weights=[embedding_matrix])(inp)\nx = Bidirectional(GRU(64,return_sequences=True))(x)\nx = GlobalMaxPool1D()(x)\nx = Dense(16, activation=\"relu\")(x)\nx = Dropout(0.1)(x)\nx = Dense(2, activation=\"softmax\")(x)\nmodel_2 = Model(inputs=inp, outputs=x)\nmodel_2.compile(loss='categorical_crossentropy', optimizer='adam', metrics=['accuracy'])\nprint(model_2.summary())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"82b211cf284553060a2bbe7b727367320f650f1b"},"cell_type":"code","source":"model_2.fit(x_train, y_train, batch_size=512, epochs=2, validation_data=(x_val, y_val))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fda1c26fbc0e1d693c0907c339af836ca8bbcf0b"},"cell_type":"code","source":"pred=np.round(np.clip(model_2.predict(x_val), 0, 1))\nprint(f1_score(y_val, pred, average = None))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"594e0aa1222b6bea7d8abe0b61c7503b7ebf5e31"},"cell_type":"code","source":"pred_2=np.round(np.clip(model_2.predict(x_test), 0, 1)).astype(int)\npred_2 = pd.DataFrame(pred_2)\npred_2 = pred_2.idxmax(axis=1)\nsubmission = pd.DataFrame({'qid':test['qid'], 'prediction':pred_2})\nsubmission.to_csv(\"sub_pred_2.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"51c2535e71837b136d742e55ae188026fa90b522"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"376d4f6787d160dda19ec6bebba1491eff6a76d2"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b2639bbee81da1411f49a5676503438a4b7ef6a"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}