{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":true},"cell_type":"code","source":"import string\nimport re\nfrom os import listdir\nfrom numpy import array\nfrom nltk.corpus import stopwords\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.utils.vis_utils import plot_model\nfrom keras.models import Sequential\nfrom keras.layers import Dense\nfrom collections import Counter\n# Scikit Learn\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\nimport pandas as pd\nimport numpy as np\n\nimport matplotlib as mpl\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\nfrom subprocess import check_output\nfrom wordcloud import WordCloud, STOPWORDS\n\nfrom sklearn.linear_model import LogisticRegression\npd.set_option('display.max_colwidth', -1)\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import classification_report, confusion_matrix, accuracy_score, precision_score, recall_score,roc_auc_score,roc_curve\nimport numpy as np\nfrom sklearn import model_selection, preprocessing, metrics, ensemble, naive_bayes, linear_model\nfrom keras.preprocessing.sequence import pad_sequences\n\nimport os\nimport time\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom tqdm import tqdm\nimport math\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import metrics\n\nfrom keras.preprocessing.text import Tokenizer\nfrom keras.preprocessing.sequence import pad_sequences\nfrom keras.layers import Dense, Input, LSTM, Embedding, Dropout, Activation, CuDNNGRU, Conv1D, Flatten\nfrom keras.layers import Bidirectional, GlobalMaxPool1D\nfrom keras.models import Model\nfrom keras import initializers, regularizers, constraints, optimizers, layers\n\n\nprint(\"Libraries loaded\")","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"a06d6485c4fb4b16cde71699486723c3f0035241"},"cell_type":"markdown","source":"**Read the data******"},{"metadata":{"trusted":true,"_uuid":"719db1055eea23f98cbda39c6bf227bd7e6b0645"},"cell_type":"code","source":"train = pd.read_csv(\"../input/train.csv\")\ntest = pd.read_csv(\"../input/test.csv\")\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Train_1 = train[train['target']==1]\n# Train_0 = train[train['target']==0]\n# train =  pd.concat([Train_1,Train_0.head(len(Train_1))], axis=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# train = train.head(10)?\ntest_df = test","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"50d7f1c0fdf4aea206afdacb9db5007a2c62c2ee"},"cell_type":"markdown","source":"**Check the target column break down**"},{"metadata":{"trusted":true,"_uuid":"2afd3095b6715f7704a4848a9d067a3bc129316a"},"cell_type":"code","source":"train[\"target\"].value_counts()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"bf9b2a4e4282411d96dc6e3602f15e0a14393ef2"},"cell_type":"markdown","source":"The data is not balanced"},{"metadata":{"_uuid":"4bcd51b27e18083305894ab7ad4d0512ff6acc6d"},"cell_type":"markdown","source":"**Word Cloud**"},{"metadata":{"_uuid":"f0d7f44deca7e7b8b962238a59f0df4cbb9e00d9"},"cell_type":"markdown","source":"**1) Word cloud of Sincere questions**"},{"metadata":{"trusted":true,"_uuid":"f3b0eeaf667f088235e9f618f7b4a75c588e24f2"},"cell_type":"code","source":"stopwords = set(STOPWORDS)\n\n\nwordcloud = WordCloud(\n                          background_color='white',\n                          stopwords=stopwords,\n                          max_words=200,\n                          max_font_size=40, \n                          random_state=42\n                         ).generate(str(train[train.target==0]['question_text']))\n\nprint(wordcloud)\nfig = plt.figure(1)\nplt.imshow(wordcloud)\nplt.axis('off')\nplt.show()\nfig.savefig(\"word1.png\", dpi=900)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"64ed082363b637ed1d8cce75fb2678dc20776780"},"cell_type":"markdown","source":"**2) Word cloud of In-Sincere questions**"},{"metadata":{"trusted":true,"_uuid":"68a5c228ccc86c6f914f8e7cd4f540c44bd2da4f"},"cell_type":"code","source":"stopwords = set(STOPWORDS)\n\nwordcloud = WordCloud(\n                          background_color='white',\n                          stopwords=stopwords,\n                          max_words=200,\n                          max_font_size=40, \n                          random_state=42\n                         ).generate(str(train[train.target==1]['question_text']))\n\nprint(wordcloud)\nfig = plt.figure(1)\nplt.imshow(wordcloud)\nplt.axis('off')\nplt.show()\nfig.savefig(\"word1.png\", dpi=900)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a6d9634e5004e9ffd5854938d3b9ed631b20b831"},"cell_type":"code","source":"## split to train and val\ntrain_df, val_df = train_test_split(train, test_size=0.1, random_state=2018)\n\n## some config values \nembed_size = 100 # how big is each word vector\nmax_features = 50000 # how many unique words to use (i.e num rows in embedding vector)\nmaxlen = 100 # max number of words in a question to use\n\n## fill up the missing values\ntrain_X = train_df[\"question_text\"].fillna(\"_na_\").values\nval_X = val_df[\"question_text\"].fillna(\"_na_\").values\ntest_X = test_df[\"question_text\"].fillna(\"_na_\").values\n\n## Tokenize the sentences\ntokenizer = Tokenizer(num_words=max_features)\ntokenizer.fit_on_texts(list(train_X))\ntrain_X = tokenizer.texts_to_sequences(train_X)\nval_X = tokenizer.texts_to_sequences(val_X)\ntest_X = tokenizer.texts_to_sequences(test_X)\n\n## Pad the sentences \ntrain_X = pad_sequences(train_X, maxlen=maxlen)\nval_X = pad_sequences(val_X, maxlen=maxlen)\ntest_X = pad_sequences(test_X, maxlen=maxlen)\n\n## Get the target values\ntrain_y = train_df['target'].values\nval_y = val_df['target'].values\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"717fb2b8fc1283bb8c2db5f90a8125a35694ebfb"},"cell_type":"code","source":"\n# inp = Input(shape=(maxlen,))\n# x = Embedding(max_features, embed_size)(inp)\n# x = Bidirectional(CuDNNGRU(64, return_sequences=True))(x)\n# x = GlobalMaxPool1D()(x)\n# x = Dense(16, activation=\"relu\")(x)\n# x = Dropout(0.1)(x)\n# x = Dense(1, activation=\"sigmoid\")(x)\n# model = Model(inputs=inp, outputs=x)\n# model.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n\n# print(model.summary())\n# model.fit(train_X, train_y, batch_size=512, epochs=2, validation_data=(val_X, val_y))\n\n\n# define the model\nmodel = Sequential()\nmodel.add(Embedding(max_features, embed_size, input_length=maxlen))\nmodel.add(Flatten())\nmodel.add(Dense(1, activation='sigmoid'))\n# compile the model\nmodel.compile(optimizer='adam', loss='binary_crossentropy', metrics=['acc'])\n# summarize the model\nprint(model.summary())\nmodel.fit(train_X, train_y, batch_size=512, epochs=5, validation_data=(val_X, val_y))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.metrics import classification_report, confusion_matrix, accuracy_score, precision_score, recall_score,roc_auc_score,roc_curve, auc,  f1_score\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\n\nfrom sklearn.metrics import f1_score\n# Making the Confusion Matrix\ndef get_metrics(y_test, y_pred):\n    cm = confusion_matrix(y_test, y_pred)\n\n    class_names=[0,1] # name  of classes\n    fig, ax = plt.subplots()\n    tick_marks = np.arange(len(class_names))\n    plt.xticks(tick_marks, class_names)\n    plt.yticks(tick_marks, class_names)\n    # create heatmap\n    sns.heatmap(pd.DataFrame(cm), annot=True, cmap=\"YlGnBu\" ,fmt='g')\n    ax.xaxis.set_label_position(\"top\")\n    plt.tight_layout()\n    plt.title('Confusion matrix', y=1.1)\n    plt.ylabel('Actual label')\n    plt.xlabel('Predicted label')\n\n\n    print(\"Accuracy:\",metrics.accuracy_score(y_test, y_pred))\n\n    # Model Precision: what percentage of positive tuples are labeled as such?\n    print(\"Precision:\",metrics.precision_score(y_test, y_pred))\n\n    # Model Recall: what percentage of positive tuples are labelled as such?\n    print(\" True positive rate or (Recall or Sensitivity) :\",metrics.recall_score(y_test, y_pred))\n\n    tn, fp, fn, tp = metrics.confusion_matrix(y_test, y_pred).ravel()\n    specificity = tn / (tn+fp)\n\n    #Specitivity. or True negative rate\n    print(\" True Negative rate or Specitivity :\",specificity)\n\n    false_negative = fn / (fn+tp)\n\n    #False negative rate\n    print(\" False Negative rate :\",false_negative)\n\n    #False positive rate\n    print(\" False positive rate (Type 1 error) :\",1 - specificity)\n    \n    print('F Score', f1_score(y_test, y_pred))\n    print(cm)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"val_pred_y = model.predict_classes([val_X], batch_size=1024, verbose=1)\nget_metrics(val_y,val_pred_y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_y = model.predict_classes([test_X], batch_size=1024, verbose=1)\noutput = pd.DataFrame({\"qid\":test_df[\"qid\"].values})\noutput['prediction'] = test_y\noutput.to_csv(\"submission.csv\", index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}