{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true,"scrolled":true},"cell_type":"code","source":"df = pd.read_csv('../input/quora-insincere-questions-classification/train.csv')\ndf2 = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"20a4c2f595f35fa443628f1741a247b35679a128"},"cell_type":"code","source":"# extracting the number of examples of each class\nsincere_questions = df[df['target'] == 0].shape[0]\ninsincere_questions = df[df['target'] == 1].shape[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"54260f567a812927957d2adb5ae84780aa2813ad"},"cell_type":"code","source":"import matplotlib\nfrom matplotlib import pyplot as plt\n# import seaborn as sns\n%matplotlib inline\n# bar plot of the 3 classes\nplt.bar(10,sincere_questions,3, label=\"sincere\")\nplt.bar(15,insincere_questions,3, label=\"insincere\")\nplt.legend()\nplt.ylabel('Number of examples')\nplt.title('Proportion of examples')\nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dd6ab8121a96b4e1585115d2c1aa6b2712accb98"},"cell_type":"code","source":"print(\"The number of sincere questions is: \", sincere_questions)\nprint(\"The number of insincere questions is: \", insincere_questions)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2d9b8b1da51d4e0503c2ebd0129c551f12cf1495"},"cell_type":"code","source":"import numpy as np\nimport re\nimport nltk\n#nltk.download('stopwords')\nwpt = nltk.WordPunctTokenizer()\nstop_words = nltk.corpus.stopwords.words('english')\n\ndef normalize_document(doc):\n    # lower case and remove special characters\\whitespaces\n    doc = re.sub(r'[^a-zA-Z\\s]', '', doc, re.I|re.A)\n    doc = doc.lower()\n    doc = doc.strip()\n    # tokenize document\n    tokens = wpt.tokenize(doc)\n    # filter stopwords out of document\n    filtered_tokens = [token for token in tokens if token not in stop_words]\n    # re-create document from filtered tokens\n    doc = ' '.join(filtered_tokens)\n    return doc\n\nnormalize_corpus = np.vectorize(normalize_document)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"69fba2a6505667bd250ec833fcc34ccc695233a1"},"cell_type":"code","source":"df[\"question_text\"] = normalize_corpus(df[\"question_text\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3dbaed7b932dfc8aeb2905af54e0b3887786d891"},"cell_type":"code","source":"df2[\"question_text\"] = normalize_corpus(df2[\"question_text\"])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9d5a63ef23ae64a126a78cd8be6342beb3691e0d"},"cell_type":"code","source":"#nltk.download('punkt')\ndf[\"question_text\"] = df[\"question_text\"].apply(nltk.word_tokenize)\nprint (\"series.apply to train questions\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a6af580f28d6026d4c1dc2d5aee51b28e707dbd7"},"cell_type":"code","source":"df2[\"question_text\"] = df2[\"question_text\"].apply(nltk.word_tokenize)\nprint (\"series.apply to test questions\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e83480d3276b49fbb00ecbf24c37d0e28d8d20a4"},"cell_type":"code","source":"from nltk.stem import PorterStemmer, WordNetLemmatizer\nporter_stemmer = PorterStemmer()\ndf['question_text_tokenized_stemmed']=df['question_text'].apply(lambda x : [porter_stemmer.stem(y) for y in x])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b48c4191c3292b186baab28ed8206c5362973334"},"cell_type":"code","source":"df2['question_text_tokenized_stemmed']=df2['question_text'].apply(lambda x : [porter_stemmer.stem(y) for y in x])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7c10fa3dc4a76d743dd23f9674b831fec975e2bf"},"cell_type":"code","source":"#nltk.download('wordnet')\ndf['question_text_tokenized_lemmatized']=df['question_text_tokenized_stemmed'].apply(lambda x : [WordNetLemmatizer().lemmatize(y) for y in x])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"720e2429f2dbbfeca7e953e54c036ae1817b8b6a"},"cell_type":"code","source":"df2['question_text_tokenized_lemmatized']=df2['question_text_tokenized_stemmed'].apply(lambda x : [WordNetLemmatizer().lemmatize(y) for y in x])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6a62bfdcb5ee6f72bad0d13656b610e34b8be7fd"},"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\n#import numpy as np\nmaxlen = 12\ntraining_samples = 848980 #65%\nvalidation_samples = 457142\nmax_words = 10000\ntokenizer = Tokenizer(num_words=max_words)\ntokenizer.fit_on_texts(df.question_text_tokenized_lemmatized.values)\nsequences = tokenizer.texts_to_sequences(df.question_text_tokenized_lemmatized.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0633b53415343662f54c133f4e4bd2053d3b4fba"},"cell_type":"code","source":"tokenizer.fit_on_texts(df2.question_text_tokenized_lemmatized.values)\ntest_sequences = tokenizer.texts_to_sequences(df2.question_text_tokenized_lemmatized.values)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e86ad1af8e123ca092788c7e63416a9638f5f1ed"},"cell_type":"code","source":"word_index = tokenizer.word_index\nprint('Found %s unique tokens.' % len(word_index))\ndata = pad_sequences(sequences, maxlen=maxlen)\nlabels = np.asarray(df.target)\nprint('Shape of data tensor:', data.shape)\nprint('Shape of label tensor:', labels.shape)\nindices = np.arange(data.shape[0])\nnp.random.shuffle(indices)\ndata = data[indices]\nlabels = labels[indices]\nx_train = data[:training_samples]\ny_train = labels[:training_samples]\nx_val = data[training_samples: training_samples + validation_samples]\ny_val = labels[training_samples: training_samples + validation_samples]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ffd21c5ce559586320ddd1df758659d1b8c500f5"},"cell_type":"code","source":"x_test = pad_sequences(test_sequences, maxlen=maxlen)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5b0efb70fa45f7cf508d49e8889c1f1c228f73c0"},"cell_type":"code","source":"import os\nglove_dir = '../input/glove6b100dtxt'\nembeddings_index = {}\nf = open(os.path.join(glove_dir, 'glove.6B.100d.txt'))\nfor line in f:\n    values = line.split()\n    word = values[0]\n    coefs = np.asarray(values[1:], dtype='float32')\n    embeddings_index[word] = coefs\nf.close()\nprint('Found %s word vectors.' % len(embeddings_index))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"826cf31047aca30bfb5e75b2c1f0c2bec27373c2"},"cell_type":"code","source":"embedding_dim = 100\nembedding_matrix = np.zeros((max_words, embedding_dim))\nfor word, i in word_index.items():\n    if i < max_words:\n        embedding_vector = embeddings_index.get(word)\n        if embedding_vector is not None:\n            embedding_matrix[i] = embedding_vector","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aaf959a502466aa294a5325cca99cfca14c06e38"},"cell_type":"code","source":"from keras.layers import LSTM\nfrom keras.models import Sequential\nfrom keras.layers import Embedding, Flatten, Dense\nmodel1 = Sequential()\nmodel1.add(Embedding(max_words, embedding_dim, input_length = maxlen))\nmodel1.add(LSTM(25))\nmodel1.add(Dense(1, activation='sigmoid'))\n\nmodel1.layers[0].set_weights([embedding_matrix])\nmodel1.layers[0].trainable = False\n\nmodel1.summary()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"4a6056c23064021a705640141f5c42c0badcca84"},"cell_type":"code","source":"from keras import backend as K\ndef recall(y_true, y_pred):\n        \"\"\"Recall metric.\n\n        Only computes a batch-wise average of recall.\n\n        Computes the recall, a metric for multi-label classification of\n        how many relevant items are selected.\n        \"\"\"\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n        recall = true_positives / (possible_positives + K.epsilon())\n        return recall\n    \ndef precision(y_true, y_pred):\n        \"\"\"Precision metric.\n\n        Only computes a batch-wise average of precision.\n\n        Computes the precision, a metric for multi-label classification of\n        how many selected items are relevant.\n        \"\"\"\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n        precision = true_positives / (predicted_positives + K.epsilon())\n        return precision","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8f08a8862edf76dfc5dcd1f9d649dcc4df3c3f18"},"cell_type":"code","source":"def f1(y_true, y_pred):\n    def recall(y_true, y_pred):\n        \"\"\"Recall metric.\n\n        Only computes a batch-wise average of recall.\n\n        Computes the recall, a metric for multi-label classification of\n        how many relevant items are selected.\n        \"\"\"\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        possible_positives = K.sum(K.round(K.clip(y_true, 0, 1)))\n        recall = true_positives / (possible_positives + K.epsilon())\n        return recall\n    def precision(y_true, y_pred):\n        \"\"\"Precision metric.\n\n        Only computes a batch-wise average of precision.\n\n        Computes the precision, a metric for multi-label classification of\n        how many selected items are relevant.\n        \"\"\"\n        true_positives = K.sum(K.round(K.clip(y_true * y_pred, 0, 1)))\n        predicted_positives = K.sum(K.round(K.clip(y_pred, 0, 1)))\n        precision = true_positives / (predicted_positives + K.epsilon())\n        return precision\n    \n    precision = precision(y_true, y_pred)\n    recall = recall(y_true, y_pred)\n    return 2*((precision*recall)/(precision+recall+K.epsilon()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"scrolled":true,"_uuid":"895915e47c6dbaa641ae8802ea0b866579a20968"},"cell_type":"code","source":"model1.compile(optimizer='rmsprop',\nloss='binary_crossentropy',\nmetrics=['acc', f1, precision, recall])\nhistory = model1.fit(x_train, y_train,\nepochs=1,\nbatch_size=100,\nvalidation_data=(x_val, y_val))\nmodel1.save_weights('processed_and_trained1.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b109823917d02dbc8c7888fac5d745c1d51f066"},"cell_type":"code","source":"y_predicted = model1.predict_classes(x_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"97242e1734fce92496263ed25c14e3c560ef3efc"},"cell_type":"code","source":"print(y_predicted)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"188713d54b00cdbca77d3e799fa63c4a71c718b2"},"cell_type":"code","source":"submit = pd.read_csv('../input/quora-insincere-questions-classification/test.csv')\n#del submit['question_text']\nsubmit.rename(index=str, columns={\"question_text\": \"target\"})\nsubmit.question_text = y_predicted\nsubmit.to_csv(\"sample_submission.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cc36ad332697278af4cd97fd7c425b577fb7379e"},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fb118e275e147d979865f92fe54ae4685fee9d4e"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}