{"cells":[{"metadata":{"_uuid":"f2ebcaa613bd07ab260ccc21ae9e5484b2f05915"},"cell_type":"markdown","source":"#  Quora Insincere Questions Classification\n---"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load in \n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\nfrom imblearn.pipeline import make_pipeline\nfrom imblearn.over_sampling import SMOTE\nimport tensorflow as tf\n\nimport sklearn.pipeline \nfrom sklearn.pipeline import Pipeline\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfTransformer\nfrom sklearn.model_selection import StratifiedKFold\nfrom sklearn.metrics import precision_score, recall_score, f1_score, roc_auc_score, auc\n\n# Input data files are available in the \"../input/\" directory.\n# For example, running this (by clicking run or pressing Shift+Enter) will list the files in the input directory\n\nimport os\nprint(os.listdir(\"../input\"))\n\n# Any results you write to the current directory are saved as output.","execution_count":null,"outputs":[]},{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"## 1. Loading data "},{"metadata":{"trusted":true,"_uuid":"28800162c77077f21a052e8d6f7e117b35fb7918"},"cell_type":"code","source":"dftrain=pd.read_csv(\"../input/train.csv\")\ndftest=pd.read_csv(\"../input/test.csv\")\ndftrain.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"cad78e1e65b86097be1ab1f68d13902bbf74c593"},"cell_type":"markdown","source":"## 2. Check the type of features "},{"metadata":{"trusted":true,"_uuid":"2691a89ab04333d231118608c9bab1bae710ea3f"},"cell_type":"code","source":"dftrain.dtypes","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"94fe632a1823f72d01468b63bc982d10322e8b34"},"cell_type":"markdown","source":"`This Quora Insincere Questions Classification is a classificaiton problem. So, we need to check the distribution of target feautre`"},{"metadata":{"trusted":true,"_uuid":"6b3971864cc9a75cce682ea60a5ab935aaa22b05"},"cell_type":"code","source":"df_Insincere=dftrain[dftrain.target==0]\nprint(\"Total Samples : {}\". format(dftrain.shape[0]))\nprint(\"No Of Insincere Samples : {}\". format(df_Insincere.shape[0]))\nprint(\"No Of sincere Samples : {}\". format(dftrain.shape[0]- df_Insincere.shape[0]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5c015001f822b1f660defc9b0dd3cf143b9601dd"},"cell_type":"code","source":"dftrain.dropna(axis=1)\ndftrain.drop_duplicates(inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a5dbdcf93f28f5d2545b7ee9869cc1ce28693453"},"cell_type":"code","source":"dftrain.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"babc4b1102abb2f9a10e21d0c43eab400c203b27"},"cell_type":"code","source":"# Check data is Balance \ndfgroup=dftrain.groupby(['target']).agg(['count'])\ndfgroup.columns=['COUNT_PER_CLASS', 'COUNT_PER_CLASS_TEXT']\ndfgroup['COUNT_PER_CLASS_%']=dfgroup['COUNT_PER_CLASS'].map(lambda x: (x/dftrain.shape[0])*100)\ndfgroup","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"768961b2c80120368035bd855ec5f7b3f0bf2b9c"},"cell_type":"markdown","source":"## 3. Visualization "},{"metadata":{"trusted":true,"_uuid":"69ec717038cfff8ae31f697865122b90502b4120"},"cell_type":"code","source":"target_visual={1:'YES',0:'NO'}\ndftrain_visual=dftrain\ndftrain_visual['target']=dftrain_visual['target'].map( lambda x : 'YES' if x>0 else 'NO') ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"44d4b9b37e5149a1db2bef4614118565312d697d"},"cell_type":"code","source":"font={'size':16}\nfig, ax=plt.subplots(figsize=(10,5))\n# Sample Per class\ndf_sample_count=dftrain_visual['target'].groupby(dftrain_visual['target']).count()\nx=df_sample_count.index.values\nax.bar(x,df_sample_count,align='center', label=['On-Time', 'Delayed Flight']) \nax.set_ylabel('Number of Samples')\nax.set_xlabel('Types of Class')\nax.set_xticks(x)\nax.set_xticklabels(x, rotation = 45) \nplt.show()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"478a1203372fd453172738f7c0bcf7c86f6b3e03"},"cell_type":"markdown","source":"# This dataset is imblanced. We need to balance it. "},{"metadata":{"trusted":true,"_uuid":"fd3c9d6214c34b6ada8e564702cda241d38a167f"},"cell_type":"code","source":"countV=CountVectorizer(stop_words='english')\ntfIdf=TfidfTransformer() ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"44b294f98a0e8fb80ed53227a8a9de9a468106f6"},"cell_type":"code","source":"X=dftrain.question_text\nX=X.str.lower().str.strip()\nY= dftrain.target\nY=pd.get_dummies(Y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"02b2af5f05509fd235b872ea7f54e2c978ad128d"},"cell_type":"code","source":"X=countV.fit_transform(X)\nX=tfIdf.fit_transform(X)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"90ac308244ee2c49cc62de1014370b69473c6004"},"cell_type":"code","source":"x_train,x_test, y_train,y_test=train_test_split(X,Y, test_size=0.25,stratify=Y)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"47b4abb6088c4fc573c7a6b423ac65b8a8926947"},"cell_type":"code","source":"def nn_layers(df,weights,biases, keep_prob):\n    l1=tf.add(tf.matmul(df,weights['h1']), biases['b1'])\n    l1=tf.nn.relu(l1)\n    l1=tf.nn.dropout(l1,keep_prob)\n    l_out=tf.add(tf.matmul(l1,weights['out']), biases['out'])\n    return l_out","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"031f414deb4f190ac4a3f5ef014625848ce23be6"},"cell_type":"code","source":"n_hidden_1=3000\nn_input=x_train.shape[1]\nn_classes=y_train.shape[1]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3412acfed3613cea36408e36b0b039736170618d"},"cell_type":"code","source":"# Weight and Biases for every layer\nweights={\n    'h1':tf.Variable(tf.random_normal([n_input, n_hidden_1])),\n    'out':tf.Variable(tf.random_normal([n_hidden_1, n_classes]))\n}\nbiases={\n    'b1':tf.Variable(tf.random_normal([ n_hidden_1])),\n    'out':tf.Variable(tf.random_normal([ n_classes]))\n}\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"168f09648c30905988b814758c74ee2f37385d82"},"cell_type":"code","source":"keep_prob = tf.placeholder(tf.float32)\ntraining_epochs = 5\ndisplay_step = 1000\nbatch_size = 100000\nx=tf.placeholder(tf.float32, [None,n_input])\ny=tf.placeholder(tf.float32, [None,n_classes])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d189d2f16bb2d387e99ab6fda99f302ab2d7b1d1"},"cell_type":"code","source":"predictions=nn_layers(df=x,weights=weights,biases=biases,keep_prob=keep_prob)\ncost=tf.reduce_mean(tf.nn.softmax_cross_entropy_with_logits_v2(logits=predictions, labels=y))\nlr_rate=0.001\noptimizer=tf.train.AdamOptimizer(learning_rate=lr_rate).minimize(cost)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"06dabaea474bc81808514efc74a5b86228076483","_kg_hide-input":true},"cell_type":"code","source":"with tf.Session() as sess:\n    initializer = tf.global_variables_initializer()\n    sess.run(initializer)\n    for epoch in range(training_epochs):\n        avg_cost = 0.0\n        total_batches = int(len(x_train) / batch_size)\n        x_batches = np.array_split(x_train, total_batches)\n        y_batches = np.array_split(y_train_NN, total_batches)\n        for i in range(total_batches):\n            batch_x, batch_y = x_batches[i], y_batches[i]\n            print(batch_x.shape)\n            print(batch_y.shape)\n            _,co=sess.run([optimizer, cost], feed_dict={x:x_batch, y:y_batch, keep_prob:0.50})\n            avg_cost += c / total_batches\n        if epoch % display_step:\n            print(\"Epoch:\", '%04d' % (epoch + 1), \"cost=\", \"{:.9f}\".format(avg_cost))\n    print('Execution Finished')\n    correct_prediction=tf.equal(tf.argmax(predictions,1), tf.argmaxa(y_train,1))\n    accuracy=tf.reduce_mean(tf.cast(correct_prediction, tf.float64))\n    print(\"Test Accuracy {}\".format(accuracy.eval({x:x_test,y:y_test,keep_prob:1.0})))\n    print(\"Train Accuracy {}\".format(accuracy.eval({x:x_train,y:y_train,keep_prob :1.0})))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3989ef8b72f80b9b126077cef88ac63a6db38344"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}