{"cells":[{"metadata":{"trusted":true,"_uuid":"c33c057cca2b9f1286e5007c12447074894f6af5"},"cell_type":"code","source":"import os\nprint(os.listdir(\"../input\"))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d0073732873aae0b0a7434a38a7092794da1d8f8"},"cell_type":"code","source":"!ls ../input/embeddings","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n%matplotlib inline\n\n\nimport nltk\nfrom sklearn.pipeline import Pipeline\nfrom nltk.corpus import stopwords\nfrom string import punctuation as str_pun\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.feature_extraction.text import CountVectorizer,TfidfVectorizer\nfrom sklearn import model_selection\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.metrics import f1_score","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"256113df692bdbd7d96761e51fcaef8b01919dac"},"cell_type":"code","source":"train = pd.read_csv('../input/train.csv')\nprint(train.shape)\ntrain.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"da5bd86a2469a87cb2a27c571fd1e5414da82b40"},"cell_type":"code","source":"test = pd.read_csv('../input/test.csv')\nprint(test.shape)\ntest.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a741bc903174ffd8dbdbafa93345840030098758"},"cell_type":"code","source":"insincere  = train[train['target']==1]\nprint(\"length of the INSINCERE : \",len(insincere))\nsincere = train[train['target']==0]\nprint(\"length of the SINCERE :\",len(sincere))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"dd2f36c1959745cd0190c7db1a79afa7d57674ff"},"cell_type":"code","source":"sns.countplot(data=train,hue=train['target'],x=train['target'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"169e222dd050b897dbeac46a7f10964255048b07"},"cell_type":"markdown","source":"### Number of words in the text\n"},{"metadata":{"trusted":true,"_uuid":"57545b260baa3ff648dda9e0b465b4b0cc336e3c"},"cell_type":"code","source":"train[\"num_words\"] = train[\"question_text\"].apply(lambda x: len(str(x).split()))\ntest[\"num_words\"] = test[\"question_text\"].apply(lambda x: len(str(x).split()))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9ae14a94d239e0af77aa0d6b332c290302500422"},"cell_type":"markdown","source":"### Number of unique words in the text \n"},{"metadata":{"trusted":true,"_uuid":"b76a83ccce260c2f73f831edb9c01180ca938198"},"cell_type":"code","source":"train[\"num_unique_words\"] = train[\"question_text\"].apply(lambda x: len(set(str(x).split())))\ntest[\"num_unique_words\"] = test[\"question_text\"].apply(lambda x: len(set(str(x).split())))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2e6ca88e00950d92e25b210feff57c94b8241d13"},"cell_type":"markdown","source":"### Number of characters in the text \n"},{"metadata":{"trusted":true,"_uuid":"cd629eb5860d8469c975f20949160c9622faee2a"},"cell_type":"code","source":"train[\"num_chars\"] = train[\"question_text\"].apply(lambda x: len(str(x)))\ntest[\"num_chars\"] = test[\"question_text\"].apply(lambda x: len(str(x)))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"b97364a1822be5815c1c39edad1e65895eb9cf0d"},"cell_type":"markdown","source":"### Number of Stopwords in text"},{"metadata":{"trusted":true,"_uuid":"441ecf2b68c74eca42e762ec7b3efaefcdd3aa53"},"cell_type":"code","source":"w = stopwords.words('english')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"9d41ced6e46322873a51da40b38b618a85a79c71"},"cell_type":"code","source":"train['num_stopwords'] = train['question_text'].apply(lambda x : len([nw for nw in str(x).split() if nw.lower() in w]))\ntest['num_stopwords'] = test['question_text'].apply(lambda x : len([nw for nw in str(x).split() if nw.lower() in w]))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"7d4cfd0f4cb5d71d2785bf1f3907562d77e848fe"},"cell_type":"markdown","source":"### Number of punctuations in text"},{"metadata":{"trusted":true,"_uuid":"6a3153951a564b88f184815efcfd10c3b57557cd"},"cell_type":"code","source":"train['num_punctuation'] = train['question_text'].apply(lambda x : len([np for np in str(x) if np in str_pun]))\ntest['num_punctuation'] = test['question_text'].apply(lambda x : len([np for np in str(x) if np in str_pun]))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"de51e8312676839b82fde4545c112fe5f76977ae"},"cell_type":"markdown","source":"### Number of Upper case and Lower case in text"},{"metadata":{"trusted":true,"_uuid":"07ad1d9805b1c3567335467c0fd9c5a6a6f98c16"},"cell_type":"code","source":"train['num_uppercase'] = train['question_text'].apply(lambda x : len([nu for nu in str(x).split() if nu.isupper()]))\ntest['num_uppercase'] = test['question_text'].apply(lambda x : len([nu for nu in str(x).split() if nu.isupper()]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ce04afd9026389da73f9736072952e9ad9827a6f"},"cell_type":"code","source":"train['num_lowercase'] = train['question_text'].apply(lambda x : len([nl for nl in str(x).split() if nl.islower()]))\ntest['num_lowercase'] = test['question_text'].apply(lambda x : len([nl for nl in str(x).split() if nl.islower()]))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0e5e02c021f4dd2d3766bea2f51c3ca06efb8511"},"cell_type":"markdown","source":"### Number of title in text"},{"metadata":{"trusted":true,"_uuid":"6388684964d98994ed300554b3e7f445965f362f"},"cell_type":"code","source":"train['num_title'] = train['question_text'].apply(lambda x : len([nl for nl in str(x).split() if nl.istitle()]))\ntest['num_title'] = test['question_text'].apply(lambda x : len([nl for nl in str(x).split() if nl.istitle()]))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"89cfc4aa9b304cb3f88edd16779d5f58b143eac4"},"cell_type":"markdown","source":"### Get the Count , Mean,Min,Max of the train target"},{"metadata":{"trusted":true,"_uuid":"5f9f133774433b1aa13c86d0c2ee4fbdc61e886d"},"cell_type":"code","source":"train[train['target']==1].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a0dc93b7e6d44c685bb7c8d83ca2016627503537"},"cell_type":"code","source":"train[train['target']==0].describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8374ae57679fa6da5668d00a460589be3f3e0eee"},"cell_type":"code","source":"sns.violinplot(x=train['target'],y=train['num_chars'],data=train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8817d32b9eec0388d798c3786c3abc68a0ce85c2"},"cell_type":"code","source":"sns.violinplot(x=train['target'],y=train['num_words'],data=train,split=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"18d4385b215251ba180d9ec851eb45a3058f640d"},"cell_type":"code","source":"sns.violinplot(x='target',y='num_unique_words',data=train,split=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8f409136a149979cda5abf7f9f0dda81dddb620f"},"cell_type":"code","source":"plt.figure(figsize=(20,15))\nsns.stripplot(x='num_words',y='num_unique_words',data=train, hue='target',jitter=False)#, split=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6b90220e694aef0df0ff148714dc8d0ac2269d7f"},"cell_type":"code","source":"sns.stripplot(x='target',y='num_stopwords',data=train, jitter=False)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"ae49832887b41f6bccdc534f92c8f73e43aa81b7"},"cell_type":"markdown","source":"## Remove Stopwords and Punctuation"},{"metadata":{"trusted":true,"_uuid":"632a90ad01c775cf6d5212d1c07bbc67cb6446f1"},"cell_type":"code","source":"def text_process(question):\n    nopunc = [char for char in question if char not in str_pun]\n    nopunc = \"\".join(nopunc)\n    meaning = [word for word in nopunc.split() if word.lower() not in stopwords.words('english')]\n    return( \" \".join( meaning )) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"028e3a2dc7eda6153d3d36a90343ac9497d6256c"},"cell_type":"code","source":"print(\"Processing ...\")\ntrain['question_text'].apply(text_process)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"76406308bf18f46de6b90a40c3776010f79b46dc"},"cell_type":"code","source":"print(\"processing ...\")\ntest['question_text'].apply(text_process)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ee378d71e500c8c32948259c29c2cb29c950814c"},"cell_type":"code","source":"train.head()","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"0e35fb1929a785ca347aa4fa03da45e4b85cc597"},"cell_type":"markdown","source":"## Pipeline Model\n> ### Logistic Regression and CountVectorizer using Hyperparameter Tuning"},{"metadata":{"trusted":true,"_uuid":"2f8ba4e3f94096fb34fae7d4cdcebb45227d77c3"},"cell_type":"code","source":"pipeline = Pipeline([('cv',CountVectorizer(analyzer='word',ngram_range=(1,4),max_df=0.9)),\n                     ('clf',LogisticRegression(solver='saga',class_weight='balanced',C=0.45,max_iter=250, verbose=1))])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0928965bde655e967bcc3a7041124776ab24c77b"},"cell_type":"code","source":"pipeline.get_params().keys()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"27f62a14dfaa080a84f36f376440e3a8b6835ecd"},"cell_type":"code","source":"X_train = train['question_text'].values\ny_train = train['target']\nX_test = test['question_text'].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"8e4d10e610c59795b813c7ab11fefe88c9975601"},"cell_type":"code","source":"pipeline.fit(X_train,y_train)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"18b0a03348624a313572802a5603c3a6a8ef2d3e"},"cell_type":"markdown","source":">> #### Predict the Model"},{"metadata":{"trusted":true,"_uuid":"748c7fa091eae3ea62a3365ea8edb4d1be8d55fb"},"cell_type":"code","source":"prediction = pipeline.predict(X_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ffc7613dfd46c02d4c4e81995718ab666892d605"},"cell_type":"code","source":"insincere = prediction[prediction == 1]\nprint(\"Length of INSINCERE after Prediction : \",len(insincere))\nsincere = prediction[prediction == 0]\nprint(\"Length of SINCERE after Prediction : \",len(sincere))","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"29af2c52bb5671dd82faf0f51bccc46cd6e1608e"},"cell_type":"markdown","source":"### Submit the File"},{"metadata":{"trusted":true,"_uuid":"449ea9943eae32755d6b5a3d056a47376fd8496e"},"cell_type":"code","source":"submit = pd.DataFrame({'qid':test['qid'],'prediction':prediction})\nsubmit.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"24329b8d1bb6f954e7859f071faa3afb6278df01"},"cell_type":"code","source":"submit.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1b6090c97eae64477a8f8b7fa33cef268bb9ce8e"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}