{"cells":[{"metadata":{"_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","collapsed":true,"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","trusted":false},"cell_type":"markdown","source":"**The problem at hand is a supervised binary classification learning problem. As the data is in text form, we could try few algorithms which are known to work well in such settings**"},{"metadata":{"trusted":true,"_uuid":"cd12f00704af064a18aa9d83364ebbaaeaf0978e"},"cell_type":"code","source":"# Importing the requisite libraries\nimport os\nimport pandas as pd\nimport json\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nimport sklearn.metrics\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import SGDClassifier\nfrom sklearn.naive_bayes import BernoulliNB\nfrom sklearn.naive_bayes import MultinomialNB\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn import linear_model\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.ensemble import VotingClassifier\nfrom xgboost import XGBClassifier\nfrom sklearn.svm import SVC\nimport matplotlib.pyplot as plt","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"cd57ee3f128520c9f7c5f2d7bd97a7fbf96580ca"},"cell_type":"code","source":"#Importing input files\ndf_train = pd.read_csv('../input/train.csv')\ndf_test = pd.read_csv('../input/test.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"60706f2b1c3a94da2147010324d4ab18664267d1"},"cell_type":"code","source":"#checking the training data\ndf_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"6330deb5e040ed869befa47d78e833624a1b5d60"},"cell_type":"code","source":"#checking the training data\ndf_test.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"c85e398a88c7bb61d279f3548e8c4009dff8bc17"},"cell_type":"code","source":"# Exploring the training dataset\nprint(\"The data-set has %d rows and %d columns\"%(df_train.shape[0],df_train.shape[1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"fe762989929e98b2a26b666b4451d05182184341"},"cell_type":"code","source":"# Exploring the testing dataset\nprint(\"The data-set has %d rows and %d columns\"%(df_test.shape[0],df_test.shape[1]))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e2c5728581e9ddf74a50590d9893ecf95c6f95a6"},"cell_type":"code","source":"# Checking the number of categories for target in the tranining data\ncategory_counter={x:0 for x in set(df_train['target'])}\nfor each_cat in df_train['target']:\n    category_counter[each_cat]+=1\n\nprint(category_counter)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"03ba866351beaec436c66a978b50acef576ab393"},"cell_type":"markdown","source":"Looks like a unbalanced class classification problem\nTarget category = 1 - (80,810 examples - 6.2%)\nTarget category = 0 - (1,225,312 examples - 93.8%)"},{"metadata":{"trusted":true,"_uuid":"39ecb82bcbfc3724387f753b8d6d29652a3c7fc1"},"cell_type":"code","source":"#corpus means collection of text. For this particular data-set, in our case it is Review_text\ncorpus=df_train.question_text\ncorpus","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"a9e01b94964ee289f92cf49a9fa87c3d0ccd9322"},"cell_type":"code","source":"#Initializing TFIDF vectorizer to conver the raw corpus to a matrix of TFIDF features \n#and also enabling the removal of stopwords.\nno_features = 500\nvectorizer = TfidfVectorizer(max_df=0.70, min_df=0.001, max_features=no_features, stop_words='english',ngram_range=(1,2))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"3646b1ae41ba90e6f15971017d0f1081f3ce3f61"},"cell_type":"code","source":"#creating TFIDF features sparse matrix by fitting it on the specified corpus\ntfidf_matrix=vectorizer.fit_transform(corpus).todense()\n#grabbing the name of the features.\ntfidf_names=vectorizer.get_feature_names()\n\nprint(\"Number of TFIDF Features: %d\"%len(tfidf_names)) #same info can be gathered by using tfidf_matrix.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"221d6baa30b1562678591ffa26f55a33f09cde04"},"cell_type":"code","source":"# Training data split into training and test data set using 60-40% ratio \n\n#considering the TFIDF features as independent variables to be input to the classifier\n\nvariables = tfidf_matrix\n\n#considering the category values as the class labels for the classifier.\n\nlabels = df_train.target\n\n#splitting the data into random training and test sets for both independent variables and labels.\n\nvariables_train, variables_test, labels_train, labels_test  =   train_test_split(variables, labels, test_size=.2)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"7afd9fa58227597e7147c953fa5736881b407e31"},"cell_type":"code","source":"#analyzing the shape of the training and test data-set:\nprint('Shape of Training Data: '+str(variables_train.shape))\nprint('Shape of Test Data: '+str(variables_test.shape))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2b877fba918f158e28ef9364ed5051b4578669ee"},"cell_type":"code","source":"#Applying Logistic Regression\n\n#initializing the object\nLogreg_classifier= LogisticRegression(random_state=0)\n\n#fitting the classifier or training the classifier on the training data\nLogreg_classifier=Logreg_classifier.fit(variables_train, labels_train)\n\n#after the model has been trained, we proceed to test its performance on the test data\nLogreg_predictions=Logreg_classifier.predict(variables_test)\n\n#the trained classifier has been used to make predictions on the test data-set. To evaluate the performance of the model,\n#there are a number of metrics that can be used as follows:\n\nLogreg_ascore=sklearn.metrics.accuracy_score(labels_test, Logreg_predictions)\n\nprint (\"Accuracy Score of Logistic Regression Classifier: %f\" %(Logreg_ascore))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"97ca8bcf6c2cee47c54250b735821cbf00f82044"},"cell_type":"code","source":"#Predicting on the test data \n\n#Preparing the TF-IDF out of the test data\n\ncorpus1=df_test['question_text']\n\ntfidf_matrix1=vectorizer.transform(corpus1).todense()\n\nvariables1 = tfidf_matrix1\n\nLogreg_Test_predictions = Logreg_classifier.predict(variables1)\n\ntest_id = df_test['qid']\n\nsub_file = pd.DataFrame({'qid': test_id, 'prediction': Logreg_Test_predictions}, columns=['qid', 'prediction'])\n\nsub_file.to_csv('submission.csv', index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}