{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**This is a work in progress. Please share your feedback :)**","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"markdown","source":"## Introduction\n\nI'm very excited to participate in kaggle's first NLP Competition. The goal is to predict whether a given tweet is about a real disaster or not. If Disaster then predict 1 esle predict 0.\n\nBelow are the action items:\n\n- Tokenization of text using NLTK\n- Sklearn TfidfVectorizer\n- Computing score on different metrics (F1-score is used in competition) but it is fun to look other metrics.","metadata":{}},{"cell_type":"markdown","source":"# Importing Packages","metadata":{}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn import model_selection\nimport nltk\nfrom nltk.tokenize import word_tokenize\nfrom sklearn.feature_extraction.text import TfidfVectorizer\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers\n\ntf.__version__","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:13:25.654653Z","iopub.execute_input":"2022-07-10T07:13:25.655137Z","iopub.status.idle":"2022-07-10T07:13:25.665158Z","shell.execute_reply.started":"2022-07-10T07:13:25.655077Z","shell.execute_reply":"2022-07-10T07:13:25.664187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading the Data","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"../input/nlp-getting-started/train.csv\")\ntest_df = pd.read_csv(\"../input/nlp-getting-started/test.csv\")\n\n# Shuffle the training data \ntrain_df = train_df.sample(frac=1).reset_index(drop=True)\n\n# Display top5 rows of train_df\ndisplay(train_df.head())\n\n# Display top5 rows of test_df\ndisplay(test_df.head())","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:13:25.718437Z","iopub.execute_input":"2022-07-10T07:13:25.719584Z","iopub.status.idle":"2022-07-10T07:13:25.783866Z","shell.execute_reply.started":"2022-07-10T07:13:25.719547Z","shell.execute_reply":"2022-07-10T07:13:25.782509Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Preparation (Train & Valid Set)","metadata":{}},{"cell_type":"code","source":"# initiating the input and labels\nx_train, x_test, y_train, y_test = model_selection.train_test_split(train_df.drop(\"target\", axis=1), \n                                                                    train_df.target, \n                                                                    test_size=0.15, \n                                                                    random_state=42, \n                                                                    stratify=train_df.target) \n\nprint(f\"{type(x_train)} {x_train.shape, x_test.shape}\")\nprint(f\"{type(y_train)} {y_train.shape, y_test.shape}\")\n\n# Checking the stratify split:\nprint(\"\\nChecking the stratify split which maintains the same proportion while splitting\")\nprint(y_train.value_counts(normalize=True))\nprint(y_test.value_counts(normalize=True))","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:13:25.786033Z","iopub.execute_input":"2022-07-10T07:13:25.786412Z","iopub.status.idle":"2022-07-10T07:13:25.808915Z","shell.execute_reply.started":"2022-07-10T07:13:25.786379Z","shell.execute_reply":"2022-07-10T07:13:25.807561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Understanding word_tokenize & TfidfVectorizer\ntext = \"you'll never find a rainbow, if you're looking down.\"\nprint(word_tokenize(text))\n\nt = TfidfVectorizer(tokenizer=word_tokenize, token_pattern=None)\nt.fit([text])\nprint(t.vocabulary_)\nprint(t.transform([text]))","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:13:25.811076Z","iopub.execute_input":"2022-07-10T07:13:25.811459Z","iopub.status.idle":"2022-07-10T07:13:25.822772Z","shell.execute_reply.started":"2022-07-10T07:13:25.811427Z","shell.execute_reply":"2022-07-10T07:13:25.821190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Lets implement the above concept on our training data","metadata":{}},{"cell_type":"code","source":"# initialize TfidfVectorizer with NLTK's word_tokenize function as tokenizer\ntfidf = TfidfVectorizer(tokenizer=word_tokenize, token_pattern=None)\ntfidf.fit(x_train.text)\n\n# transform training and validation data tweets\ntrain_sequences = tfidf.transform(x_train.text)\ntrain_labels = np.array(y_train)\n\ntest_squences = tfidf.transform(x_test.text)\ntest_labels = np.array(y_test)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:13:25.835927Z","iopub.execute_input":"2022-07-10T07:13:25.836348Z","iopub.status.idle":"2022-07-10T07:13:30.581983Z","shell.execute_reply.started":"2022-07-10T07:13:25.836313Z","shell.execute_reply":"2022-07-10T07:13:30.580746Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Loading the model","metadata":{}},{"cell_type":"code","source":"## Building the model\nfrom sklearn import naive_bayes\nfrom sklearn import metrics\n\nmodel = naive_bayes.MultinomialNB()\nmodel.fit(train_sequences, train_labels)\n\npreds = model.predict(test_squences)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:13:30.587339Z","iopub.execute_input":"2022-07-10T07:13:30.587710Z","iopub.status.idle":"2022-07-10T07:13:30.598940Z","shell.execute_reply.started":"2022-07-10T07:13:30.587677Z","shell.execute_reply":"2022-07-10T07:13:30.597914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def check_model_performance(preds, test_labels):\n    print(f\"Accuracy score: {round(metrics.accuracy_score(preds, test_labels), 3)}\")\n    print(f\"F1 score: {round(metrics.f1_score(preds, test_labels, average='weighted'), 3)}\")\n    print(f\"Precion score: {round(metrics.precision_score(preds, test_labels, average='weighted'), 3)}\")\n    print(f\"Recall score: {round(metrics.recall_score(preds, test_labels, average='weighted'), 3)}\")\n    \n    print(\"\\nConfusion Matrix\")\n    cm = metrics.confusion_matrix(preds, test_labels)\n    disp = metrics.ConfusionMatrixDisplay(confusion_matrix=cm,\n                                  display_labels=model.classes_)\n\n    disp.plot();\n    \ncheck_model_performance(preds, test_labels)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:13:30.600398Z","iopub.execute_input":"2022-07-10T07:13:30.600925Z","iopub.status.idle":"2022-07-10T07:13:30.887250Z","shell.execute_reply.started":"2022-07-10T07:13:30.600892Z","shell.execute_reply":"2022-07-10T07:13:30.885198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Conclusion:\n- It is nice that with baseline model we're getting f1 score ~80%.\n- We can procced with implementing some text pre-processing (removing special characters, stopwords)\n- Later will try SVM, Neuro Networks (RNN, LSTM, Tensorflow Hub Embeddings)\n\n---","metadata":{}},{"cell_type":"markdown","source":"## Prediction of test set","metadata":{}},{"cell_type":"code","source":"## prediction on Test Data\ntest_df_sequences = tfidf.transform(test_df.text)\npreds = model.predict(test_df_sequences)","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:13:30.891196Z","iopub.execute_input":"2022-07-10T07:13:30.891788Z","iopub.status.idle":"2022-07-10T07:13:32.000456Z","shell.execute_reply.started":"2022-07-10T07:13:30.891730Z","shell.execute_reply":"2022-07-10T07:13:31.999047Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv(\"../input/nlp-getting-started/sample_submission.csv\")\nsubmission.target = preds","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:13:32.001971Z","iopub.execute_input":"2022-07-10T07:13:32.002327Z","iopub.status.idle":"2022-07-10T07:13:32.013208Z","shell.execute_reply.started":"2022-07-10T07:13:32.002294Z","shell.execute_reply":"2022-07-10T07:13:32.011839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv(\"submission.csv\", index=False)\nprint(\"Submission csv file generated...\")","metadata":{"execution":{"iopub.status.busy":"2022-07-10T07:13:32.014813Z","iopub.execute_input":"2022-07-10T07:13:32.015194Z","iopub.status.idle":"2022-07-10T07:13:32.029758Z","shell.execute_reply.started":"2022-07-10T07:13:32.015160Z","shell.execute_reply":"2022-07-10T07:13:32.028661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}