{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# !pip install tensorflow -q --upgrade\n# !pip install numpy -q --upgrade\n# !pip install pandas -q --upgrade\n# !pip install sklearn -q --upgrade","metadata":{"execution":{"iopub.status.busy":"2021-10-11T20:40:54.487854Z","iopub.execute_input":"2021-10-11T20:40:54.488676Z","iopub.status.idle":"2021-10-11T20:41:35.962574Z","shell.execute_reply.started":"2021-10-11T20:40:54.488623Z","shell.execute_reply":"2021-10-11T20:41:35.961292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport nltk\nimport string\nimport numpy as np\nimport pandas as pd\nfrom nltk.corpus import stopwords\nfrom nltk.stem.snowball import SnowballStemmer\nfrom sklearn.manifold import TSNE\nimport re","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-10-11T22:21:42.469584Z","iopub.execute_input":"2021-10-11T22:21:42.469976Z","iopub.status.idle":"2021-10-11T22:21:49.164018Z","shell.execute_reply.started":"2021-10-11T22:21:42.469868Z","shell.execute_reply":"2021-10-11T22:21:49.163087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/sample_submission.csv\")\ntest = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/test.csv\")\ntrain = pd.read_csv(\"/kaggle/input/quora-insincere-questions-classification/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2021-10-11T20:41:35.980095Z","iopub.execute_input":"2021-10-11T20:41:35.980479Z","iopub.status.idle":"2021-10-11T20:41:40.912544Z","shell.execute_reply.started":"2021-10-11T20:41:35.980433Z","shell.execute_reply":"2021-10-11T20:41:40.911428Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Pre processing**","metadata":{}},{"cell_type":"code","source":"def clean_text(text):\n    text = text.translate(string.punctuation)\n    \n    text = text.lower().split()\n    \n    stops = set(stopwords.words(\"english\"))\n    text = [w for w in text if not w in stops and len(w) >= 3]\n    \n    text = \" \".join(text)\n    text = re.sub(r\"[^A-Za-z0-9^,!.\\/'+-=]\", \" \", text)\n    text = re.sub(r\"what's\", \"what is \", text)\n    text = re.sub(r\"\\'s\", \" \", text)\n    text = re.sub(r\"\\'ve\", \" have \", text)\n    text = re.sub(r\"n't\", \" not \", text)\n    text = re.sub(r\"i'm\", \"i am \", text)\n    text = re.sub(r\"\\'re\", \" are \", text)\n    text = re.sub(r\"\\'d\", \" would \", text)\n    text = re.sub(r\"\\'ll\", \" will \", text)\n    text = re.sub(r\",\", \" \", text)\n    text = re.sub(r\"\\.\", \" \", text)\n    text = re.sub(r\"!\", \" ! \", text)\n    text = re.sub(r\"\\/\", \" \", text)\n    text = re.sub(r\"\\^\", \" ^ \", text)\n    text = re.sub(r\"\\+\", \" + \", text)\n    text = re.sub(r\"\\-\", \" - \", text)\n    text = re.sub(r\"\\=\", \" = \", text)\n    text = re.sub(r\"'\", \" \", text)\n    text = re.sub(r\"(\\d+)(k)\", r\"\\g<1>000\", text)\n    text = re.sub(r\":\", \" : \", text)\n    text = re.sub(r\" e g \", \" eg \", text)\n    text = re.sub(r\" b g \", \" bg \", text)\n    text = re.sub(r\" u s \", \" american \", text)\n    text = re.sub(r\"\\0s\", \"0\", text)\n    text = re.sub(r\" 9 11 \", \"911\", text)\n    text = re.sub(r\"e - mail\", \"email\", text)\n    text = re.sub(r\"j k\", \"jk\", text)\n    text = re.sub(r\"\\s{2,}\", \" \", text)\n    \n    text = text.split()\n    stemmer = SnowballStemmer('english')\n    stemmed_words = [stemmer.stem(word) for word in text]\n    text = \" \".join(stemmed_words)\n    return text\n\ntrain['question_text'] = train['question_text'].map(lambda x: clean_text(x))","metadata":{"execution":{"iopub.status.busy":"2021-10-11T20:41:40.929642Z","iopub.execute_input":"2021-10-11T20:41:40.929877Z","iopub.status.idle":"2021-10-11T20:49:29.995387Z","shell.execute_reply.started":"2021-10-11T20:41:40.929841Z","shell.execute_reply":"2021-10-11T20:49:29.994289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocabulary_size = 20000\ntokenizer = tf.keras.preprocessing.text.Tokenizer(num_words= vocabulary_size)\ntokenizer.fit_on_texts(train['question_text'])\nsequences = tokenizer.texts_to_sequences(train['question_text'])\ndata = tf.keras.preprocessing.sequence.pad_sequences(sequences, maxlen=50)","metadata":{"execution":{"iopub.status.busy":"2021-10-11T20:52:47.361959Z","iopub.execute_input":"2021-10-11T20:52:47.362328Z","iopub.status.idle":"2021-10-11T20:53:32.089651Z","shell.execute_reply.started":"2021-10-11T20:52:47.362295Z","shell.execute_reply":"2021-10-11T20:53:32.088942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Embeddings**","metadata":{}},{"cell_type":"code","source":"import zipfile\nwith zipfile.ZipFile(\"/kaggle/input/quora-insincere-questions-classification/embeddings.zip\", 'r') as zip_ref:\n    zip_ref.extractall(\".\")","metadata":{"execution":{"iopub.status.busy":"2021-10-11T20:54:40.694Z","iopub.execute_input":"2021-10-11T20:54:40.694961Z","iopub.status.idle":"2021-10-11T20:58:45.004145Z","shell.execute_reply.started":"2021-10-11T20:54:40.694921Z","shell.execute_reply":"2021-10-11T20:58:45.001014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings_index = {}\nf = open('/kaggle/working/glove.840B.300d/glove.840B.300d.txt', encoding='utf8')\nfor line in f:\n    values = line.split()\n    word = ''.join(values[:-300])\n    coefs = np.asarray(values[-300:], dtype='float32')\n    embeddings_index[word] = coefs\nf.close()","metadata":{"execution":{"iopub.status.busy":"2021-10-11T20:58:45.008782Z","iopub.execute_input":"2021-10-11T20:58:45.009565Z","iopub.status.idle":"2021-10-11T21:02:13.639123Z","shell.execute_reply.started":"2021-10-11T20:58:45.00952Z","shell.execute_reply":"2021-10-11T21:02:13.637989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embedding_matrix = np.zeros((vocabulary_size, 300))\nfor word, index in tokenizer.word_index.items():\n    if index > vocabulary_size - 1:\n        break\n    else:\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None:\n            embedding_matrix[index] = embedding_vector","metadata":{"execution":{"iopub.status.busy":"2021-10-11T21:03:04.371776Z","iopub.execute_input":"2021-10-11T21:03:04.372774Z","iopub.status.idle":"2021-10-11T21:03:04.473331Z","shell.execute_reply.started":"2021-10-11T21:03:04.372732Z","shell.execute_reply":"2021-10-11T21:03:04.472389Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_glove = tf.keras.models.Sequential()\nmodel_glove.add(tf.keras.layers.Embedding(vocabulary_size, \n                                          300, \n                                          input_length=50, \n                                          weights=[embedding_matrix], \n                                          trainable=False))\nmodel_glove.add(tf.keras.layers.Dropout(0.2))\nmodel_glove.add(tf.keras.layers.Conv1D(64, \n                                       5, \n                                       activation='relu'))\nmodel_glove.add(tf.keras.layers.MaxPooling1D(pool_size=4))\nmodel_glove.add(tf.keras.layers.LSTM(100))\nmodel_glove.add(tf.keras.layers.Dense(1, \n                                      activation='sigmoid'))\nmodel_glove.compile(loss='binary_crossentropy', \n                    optimizer='adam', \n                    metrics=['accuracy'])\n## Fit train data\nmodel_glove.fit(data, \n                np.array(train[\"target\"]), \n                validation_split=0.4, \n                epochs = 1)","metadata":{"execution":{"iopub.status.busy":"2021-10-11T21:07:29.845974Z","iopub.execute_input":"2021-10-11T21:07:29.846322Z","iopub.status.idle":"2021-10-11T21:46:59.636414Z","shell.execute_reply.started":"2021-10-11T21:07:29.846291Z","shell.execute_reply":"2021-10-11T21:46:59.635061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Predicting**","metadata":{}},{"cell_type":"code","source":"test['question_text'] = test['question_text'].map(lambda x: clean_text(x))\nsequences = tokenizer.texts_to_sequences(test['question_text'])\ntest_data = tf.keras.preprocessing.sequence.pad_sequences(sequences, maxlen=50)","metadata":{"execution":{"iopub.status.busy":"2021-10-11T21:51:58.348242Z","iopub.execute_input":"2021-10-11T21:51:58.349163Z","iopub.status.idle":"2021-10-11T21:54:20.910684Z","shell.execute_reply.started":"2021-10-11T21:51:58.349116Z","shell.execute_reply":"2021-10-11T21:54:20.90969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = (model_glove.predict(test_data) > 0.35).astype(\"int32\")","metadata":{"execution":{"iopub.status.busy":"2021-10-11T22:16:14.248452Z","iopub.execute_input":"2021-10-11T22:16:14.248783Z","iopub.status.idle":"2021-10-11T22:17:43.643544Z","shell.execute_reply.started":"2021-10-11T22:16:14.248753Z","shell.execute_reply":"2021-10-11T22:17:43.642751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds_df = pd.DataFrame({\"qid\": test.qid, \"prediction\": preds.flatten()})\npreds_df.to_csv(\"submission.csv\",index=False)","metadata":{"execution":{"iopub.status.busy":"2021-10-11T22:18:41.943885Z","iopub.execute_input":"2021-10-11T22:18:41.944266Z","iopub.status.idle":"2021-10-11T22:18:41.964544Z","shell.execute_reply.started":"2021-10-11T22:18:41.9442Z","shell.execute_reply":"2021-10-11T22:18:41.96388Z"},"trusted":true},"execution_count":null,"outputs":[]}]}