{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv\nimport re\nimport nltk","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-11T16:19:58.079217Z","iopub.execute_input":"2022-08-11T16:19:58.080557Z","iopub.status.idle":"2022-08-11T16:19:58.086187Z","shell.execute_reply.started":"2022-08-11T16:19:58.080393Z","shell.execute_reply":"2022-08-11T16:19:58.085112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train = pd.read_csv(\"../input/nlp-getting-started/train.csv\")\ndata_test = pd.read_csv(\"../input/nlp-getting-started/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:19:58.119201Z","iopub.execute_input":"2022-08-11T16:19:58.119912Z","iopub.status.idle":"2022-08-11T16:19:58.181098Z","shell.execute_reply.started":"2022-08-11T16:19:58.119874Z","shell.execute_reply":"2022-08-11T16:19:58.180003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train.head(15)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:19:58.182971Z","iopub.execute_input":"2022-08-11T16:19:58.183618Z","iopub.status.idle":"2022-08-11T16:19:58.202849Z","shell.execute_reply.started":"2022-08-11T16:19:58.183578Z","shell.execute_reply":"2022-08-11T16:19:58.201608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:19:58.204404Z","iopub.execute_input":"2022-08-11T16:19:58.205058Z","iopub.status.idle":"2022-08-11T16:19:58.218918Z","shell.execute_reply.started":"2022-08-11T16:19:58.205007Z","shell.execute_reply":"2022-08-11T16:19:58.217927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:19:58.221332Z","iopub.execute_input":"2022-08-11T16:19:58.222382Z","iopub.status.idle":"2022-08-11T16:19:58.230874Z","shell.execute_reply.started":"2022-08-11T16:19:58.222331Z","shell.execute_reply":"2022-08-11T16:19:58.229210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:19:58.237557Z","iopub.execute_input":"2022-08-11T16:19:58.238871Z","iopub.status.idle":"2022-08-11T16:19:58.254361Z","shell.execute_reply.started":"2022-08-11T16:19:58.238814Z","shell.execute_reply":"2022-08-11T16:19:58.251792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:19:58.266247Z","iopub.execute_input":"2022-08-11T16:19:58.266998Z","iopub.status.idle":"2022-08-11T16:19:58.289617Z","shell.execute_reply.started":"2022-08-11T16:19:58.266953Z","shell.execute_reply":"2022-08-11T16:19:58.287549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:19:58.296710Z","iopub.execute_input":"2022-08-11T16:19:58.297200Z","iopub.status.idle":"2022-08-11T16:19:58.322346Z","shell.execute_reply.started":"2022-08-11T16:19:58.297161Z","shell.execute_reply":"2022-08-11T16:19:58.321177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:19:58.325639Z","iopub.execute_input":"2022-08-11T16:19:58.326856Z","iopub.status.idle":"2022-08-11T16:19:58.339663Z","shell.execute_reply.started":"2022-08-11T16:19:58.326811Z","shell.execute_reply":"2022-08-11T16:19:58.338533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_train['keyword'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:19:58.355582Z","iopub.execute_input":"2022-08-11T16:19:58.356105Z","iopub.status.idle":"2022-08-11T16:19:58.367464Z","shell.execute_reply.started":"2022-08-11T16:19:58.356067Z","shell.execute_reply":"2022-08-11T16:19:58.365833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Let's Train our Model Based on Training Data we had**","metadata":{}},{"cell_type":"code","source":"nltk.download('stopwords')\nfrom nltk.corpus import stopwords\nfrom nltk.stem.porter import PorterStemmer\ncorpus = []\nfor i in range(0, 7613):\n  text = re.sub('[^a-zA-Z]', ' ', data_train['text'][i])\n  #text = text.lower()\n  #text = text.split()\n  ps = PorterStemmer()\n  all_stopwords = stopwords.words('english')\n  all_stopwords.remove('not')\n  review = [ps.stem(word) for word in text if not word in set(all_stopwords)]\n  review = ' '.join(text)\n  corpus.append(text)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:19:58.385952Z","iopub.execute_input":"2022-08-11T16:19:58.386819Z","iopub.status.idle":"2022-08-11T16:20:04.651636Z","shell.execute_reply.started":"2022-08-11T16:19:58.386767Z","shell.execute_reply":"2022-08-11T16:20:04.650559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\ncv = CountVectorizer(max_features = 7613)\nX = cv.fit_transform(corpus).toarray()\ny = data_train.iloc[:, [4]].values","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:04.653640Z","iopub.execute_input":"2022-08-11T16:20:04.654059Z","iopub.status.idle":"2022-08-11T16:20:05.424724Z","shell.execute_reply.started":"2022-08-11T16:20:04.654025Z","shell.execute_reply":"2022-08-11T16:20:05.423418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size = 0.20, random_state = 0)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:05.426301Z","iopub.execute_input":"2022-08-11T16:20:05.427091Z","iopub.status.idle":"2022-08-11T16:20:05.960164Z","shell.execute_reply.started":"2022-08-11T16:20:05.427050Z","shell.execute_reply":"2022-08-11T16:20:05.958768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**first, let's try Naive Bayes classification**","metadata":{}},{"cell_type":"code","source":"from sklearn.naive_bayes import GaussianNB\nclassifier = GaussianNB()\nclassifier.fit(X_train, np.ravel(y_train))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:05.962868Z","iopub.execute_input":"2022-08-11T16:20:05.963593Z","iopub.status.idle":"2022-08-11T16:20:07.191348Z","shell.execute_reply.started":"2022-08-11T16:20:05.963545Z","shell.execute_reply":"2022-08-11T16:20:07.190480Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = classifier.predict(X_test)\nprint(np.concatenate((y_pred.reshape(len(y_pred),1), y_test.reshape(len(y_test),1)),1))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:07.193008Z","iopub.execute_input":"2022-08-11T16:20:07.193824Z","iopub.status.idle":"2022-08-11T16:20:07.478402Z","shell.execute_reply.started":"2022-08-11T16:20:07.193780Z","shell.execute_reply":"2022-08-11T16:20:07.477201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#let's see the accuration\nfrom sklearn.metrics import confusion_matrix, accuracy_score\ncm = confusion_matrix(y_test, y_pred)\nprint(cm)\naccuracy_score(y_test, y_pred)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:07.479965Z","iopub.execute_input":"2022-08-11T16:20:07.481184Z","iopub.status.idle":"2022-08-11T16:20:07.496076Z","shell.execute_reply.started":"2022-08-11T16:20:07.481131Z","shell.execute_reply":"2022-08-11T16:20:07.494517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"well, not bad, it gave 0.63 for accuration","metadata":{}},{"cell_type":"markdown","source":"**let's see how about applying the KNN algorithm**","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nclassifier_k = KNeighborsClassifier(n_neighbors = 5, metric = 'minkowski', p = 2)\nclassifier_k.fit(X_train, np.ravel(y_train))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:07.498353Z","iopub.execute_input":"2022-08-11T16:20:07.498838Z","iopub.status.idle":"2022-08-11T16:20:07.508499Z","shell.execute_reply.started":"2022-08-11T16:20:07.498797Z","shell.execute_reply":"2022-08-11T16:20:07.507609Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_prediction_knn= classifier_k.predict(X_test)\ny_prediction_knn","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:07.510045Z","iopub.execute_input":"2022-08-11T16:20:07.510813Z","iopub.status.idle":"2022-08-11T16:20:10.358072Z","shell.execute_reply.started":"2022-08-11T16:20:07.510769Z","shell.execute_reply":"2022-08-11T16:20:10.357011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cm_knn = confusion_matrix(y_test, y_prediction_knn)\nprint(cm_knn)\naccuracy_score(y_test, y_prediction_knn)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:10.359659Z","iopub.execute_input":"2022-08-11T16:20:10.360597Z","iopub.status.idle":"2022-08-11T16:20:10.375050Z","shell.execute_reply.started":"2022-08-11T16:20:10.360550Z","shell.execute_reply":"2022-08-11T16:20:10.373643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"well, the KNN classification algorithm gave better accuracy, so let's use it to predict the test data","metadata":{}},{"cell_type":"markdown","source":"# **let's predict the test data using KNN **","metadata":{}},{"cell_type":"code","source":"corpus_train = []\nfor i in range(0, 3263): #I want to adjust the input dimension from the data train with the test data\n  text = re.sub('[^a-zA-Z]', ' ', data_train['text'][i])\n  text = text.lower()\n  #text = text.split()\n  ps = PorterStemmer()\n  all_stopwords = stopwords.words('english')\n  all_stopwords.remove('not')\n  review = [ps.stem(word) for word in text if not word in set(all_stopwords)]\n  review = ' '.join(text)\n  corpus_train.append(text)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:10.378470Z","iopub.execute_input":"2022-08-11T16:20:10.379111Z","iopub.status.idle":"2022-08-11T16:20:13.045645Z","shell.execute_reply.started":"2022-08-11T16:20:10.378906Z","shell.execute_reply":"2022-08-11T16:20:13.044187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#check the missing value and information of test data\ndata_test.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:13.047401Z","iopub.execute_input":"2022-08-11T16:20:13.048375Z","iopub.status.idle":"2022-08-11T16:20:13.064478Z","shell.execute_reply.started":"2022-08-11T16:20:13.048324Z","shell.execute_reply":"2022-08-11T16:20:13.062814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:13.067060Z","iopub.execute_input":"2022-08-11T16:20:13.067535Z","iopub.status.idle":"2022-08-11T16:20:13.090934Z","shell.execute_reply.started":"2022-08-11T16:20:13.067485Z","shell.execute_reply":"2022-08-11T16:20:13.089621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:13.092943Z","iopub.execute_input":"2022-08-11T16:20:13.093548Z","iopub.status.idle":"2022-08-11T16:20:13.117985Z","shell.execute_reply.started":"2022-08-11T16:20:13.093490Z","shell.execute_reply":"2022-08-11T16:20:13.116595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:13.119869Z","iopub.execute_input":"2022-08-11T16:20:13.121239Z","iopub.status.idle":"2022-08-11T16:20:13.137085Z","shell.execute_reply.started":"2022-08-11T16:20:13.121186Z","shell.execute_reply":"2022-08-11T16:20:13.135832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpus_ = []\nfor i in range(0, 3263):\n  text = re.sub('[^a-zA-Z]', ' ', data_test['text'][i])\n  text = text.lower()\n  #text = text.split()\n  ps = PorterStemmer()\n  all_stopwords = stopwords.words('english')\n  all_stopwords.remove('not')\n  review = [ps.stem(word) for word in text if not word in set(all_stopwords)]\n  review = ' '.join(text)\n  corpus_.append(text)","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:13.138745Z","iopub.execute_input":"2022-08-11T16:20:13.139344Z","iopub.status.idle":"2022-08-11T16:20:16.057843Z","shell.execute_reply.started":"2022-08-11T16:20:13.139294Z","shell.execute_reply":"2022-08-11T16:20:16.056330Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_train = CountVectorizer(max_features = 3263)\nX_train_new = cv_train.fit_transform(corpus_train).toarray()\ny_train_new = data_train.iloc[:, [4]].values","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:16.059824Z","iopub.execute_input":"2022-08-11T16:20:16.060633Z","iopub.status.idle":"2022-08-11T16:20:16.316701Z","shell.execute_reply.started":"2022-08-11T16:20:16.060583Z","shell.execute_reply":"2022-08-11T16:20:16.315177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_test = CountVectorizer(max_features = 3263)\nX_ = cv_test.fit_transform(corpus_).toarray()","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:16.318767Z","iopub.execute_input":"2022-08-11T16:20:16.319218Z","iopub.status.idle":"2022-08-11T16:20:16.586135Z","shell.execute_reply.started":"2022-08-11T16:20:16.319185Z","shell.execute_reply":"2022-08-11T16:20:16.584837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\nclassifier_pr = KNeighborsClassifier(n_neighbors = 5, metric = 'minkowski', p = 2)\nclassifier_pr.fit(X_train_new, np.ravel(y_train_new[0:3263]))","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:16.587718Z","iopub.execute_input":"2022-08-11T16:20:16.588492Z","iopub.status.idle":"2022-08-11T16:20:16.600047Z","shell.execute_reply.started":"2022-08-11T16:20:16.588413Z","shell.execute_reply":"2022-08-11T16:20:16.598329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_prediction=classifier_pr.predict(X_)\ny_prediction","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:16.602126Z","iopub.execute_input":"2022-08-11T16:20:16.602693Z","iopub.status.idle":"2022-08-11T16:20:18.238874Z","shell.execute_reply.started":"2022-08-11T16:20:16.602620Z","shell.execute_reply":"2022-08-11T16:20:18.237612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission=pd.DataFrame({'id': data_test['id'], 'target' : y_prediction.ravel()})\nsubmission['target'] = submission['target']\nsubmission.to_csv('submission.csv', index=False)\nsubmission.tail()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-11T16:20:18.241034Z","iopub.execute_input":"2022-08-11T16:20:18.241621Z","iopub.status.idle":"2022-08-11T16:20:18.266693Z","shell.execute_reply.started":"2022-08-11T16:20:18.241576Z","shell.execute_reply":"2022-08-11T16:20:18.265530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **PS: actually I don't apply EDA so much to this code for this section :D, but I will aplly it next, so before we predict the tweet disaster classification using ML algorithm, at least we can gain insights from the available data first :) **","metadata":{}}]}