{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nfrom sklearn import feature_extraction, linear_model, model_selection, preprocessing\nfrom sklearn.neural_network import MLPClassifier\nfrom sklearn.ensemble import GradientBoostingClassifier\nimport re\nfrom sklearn.svm import SVC\nfrom scipy import stats\nfrom sklearn.metrics import make_scorer, roc_auc_score, f1_score\nfrom sklearn.model_selection import RandomizedSearchCV\nfrom sklearn.linear_model import RidgeCV","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-14T19:30:14.187200Z","iopub.execute_input":"2022-07-14T19:30:14.187569Z","iopub.status.idle":"2022-07-14T19:30:15.609056Z","shell.execute_reply.started":"2022-07-14T19:30:14.187489Z","shell.execute_reply":"2022-07-14T19:30:15.607820Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Importing the data from the competition `Natural Language Processing with Disaster Tweets`. \n\nEach sample in the train and test set has the following information:\n- The text of a tweet\n- A keyword from that tweet (although this may be blank!)\n- The location the tweet was sent from (may also be blank)\n\nColumns: \n- `id` - a unique identifier for each tweet\n- `text` - the text of the tweet\n- `location` - the location the tweet was sent from (may be blank)\n- `keyword` - a particular keyword from the tweet (may be blank)\n- `target` - in train.csv only, this denotes whether a tweet is about a real disaster (`1`) or not (`0`)","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/nlp-getting-started/train.csv\")\ntest_df = pd.read_csv(\"/kaggle/input/nlp-getting-started/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:30:45.403319Z","iopub.execute_input":"2022-07-14T19:30:45.403676Z","iopub.status.idle":"2022-07-14T19:30:45.467273Z","shell.execute_reply.started":"2022-07-14T19:30:45.403645Z","shell.execute_reply":"2022-07-14T19:30:45.466626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets see an example of each type of target as well as keyword and location:","metadata":{}},{"cell_type":"code","source":"print(f\"This is a non-disaster tweet: {train_df[train_df['target'] == 0]['text'].values[1]}\")\nprint(f\"This is a disaster tweet: {train_df[train_df['target'] == 1]['text'].values[1]}\")\n\nprint(f\"\\n\")\n\nprint(f'A tweet, keyword, and the target - 1 means disaster and 0 means non-disaster:')\nprint(train_df[(train_df['target'] == 0) & (pd.isnull(train_df['keyword']) == 0)][['text','keyword', 'target']].values[1])\nprint(train_df[(train_df['target'] == 1) & (pd.isnull(train_df['keyword']) == 0)][['text','keyword', 'target']].values[1])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:30:46.375434Z","iopub.execute_input":"2022-07-14T19:30:46.376555Z","iopub.status.idle":"2022-07-14T19:30:46.410394Z","shell.execute_reply.started":"2022-07-14T19:30:46.376510Z","shell.execute_reply":"2022-07-14T19:30:46.409377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Before we start working with the data, it needs some preprocessing. \n\nWe will do the following\n\n1. Remove @tags from all tweets\n2. Remove numbers from all tweets\n3. Remove #hashtags\n4. Remove characters (e.g. / - + =)\n5. Check for emails\n6. Check for websites ( http...)\n7. Normalize (lower words)\n8. Remove stopwords (e.g. it, or, was, you)\n9. Lemmatization - dictionary form of all words (e.g. children -> child, walked -> walk)","metadata":{}},{"cell_type":"code","source":"import nltk \nfrom nltk.corpus import stopwords\nstop_words=stopwords.words('english')\n\nfrom nltk.corpus import wordnet\nfrom nltk.stem import WordNetLemmatizer\n\nlemmatizer = WordNetLemmatizer()","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:30:48.533086Z","iopub.execute_input":"2022-07-14T19:30:48.533626Z","iopub.status.idle":"2022-07-14T19:30:48.971090Z","shell.execute_reply.started":"2022-07-14T19:30:48.533596Z","shell.execute_reply":"2022-07-14T19:30:48.970034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocessPipline(data, labelName):\n    \n    dataTemp = data.copy()\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: re.sub('(\\s*)@\\w+(\\s*)','', x)) # Remove tags\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: re.sub('\\d+','', x)) # Remove numbers\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: re.sub('#','', x)) # Remove hashtags\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: re.sub('https?://\\S+|www\\.\\S+','',x)) # Remove websites\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: re.sub('[^A-Za-z]',' ',x)) # Remove characters\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: x.lower()) # Lower letters only\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x : [word for word in x.split()  if word not in stop_words]) # Remove stopwords\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: [lemmatizer.lemmatize(word) for word in x]) # Lemmatization\n    dataTemp[labelName] = dataTemp[labelName].apply(lambda x: ' '.join(x)) # Join to one array\n    return dataTemp","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:30:49.712522Z","iopub.execute_input":"2022-07-14T19:30:49.712841Z","iopub.status.idle":"2022-07-14T19:30:49.722260Z","shell.execute_reply.started":"2022-07-14T19:30:49.712818Z","shell.execute_reply":"2022-07-14T19:30:49.721227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = preprocessPipline(train_df, 'text')\ntest_df = preprocessPipline(test_df, 'text')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:30:50.428010Z","iopub.execute_input":"2022-07-14T19:30:50.428966Z","iopub.status.idle":"2022-07-14T19:30:53.346101Z","shell.execute_reply.started":"2022-07-14T19:30:50.428919Z","shell.execute_reply":"2022-07-14T19:30:53.345132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check for same tweet with different label\n\npositive_tweets = train_df[train_df['target'] == 1]\nnegative_tweets = train_df[train_df['target'] == 0]","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:30:55.193401Z","iopub.execute_input":"2022-07-14T19:30:55.193993Z","iopub.status.idle":"2022-07-14T19:30:55.200495Z","shell.execute_reply.started":"2022-07-14T19:30:55.193960Z","shell.execute_reply":"2022-07-14T19:30:55.199203Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f'Number of positive tweets: {len(positive_tweets[\"text\"].values)}')\nprint(f'Number of negative tweets: {len(negative_tweets[\"text\"].values)}')\nprint(f'Number of overlapping tweets: {len(np.intersect1d(positive_tweets[\"text\"].values, negative_tweets[\"text\"].values))}')\n\noverlap = np.intersect1d(positive_tweets[\"text\"].values, negative_tweets[\"text\"].values)\n\nnew_train = [] \nfor row in train_df.iterrows():\n    if row[1][3] not in overlap:\n        new_train.append(row[1])\n        \ntrain_df = pd.DataFrame(new_train)\n\nprint(f'\\n After removal of overlaps:')\npositive_tweets = train_df[train_df['target'] == 1]\nnegative_tweets = train_df[train_df['target'] == 0]\n\nprint(f'Number of positive tweets: {len(positive_tweets[\"text\"].values)}')\nprint(f'Number of negative tweets: {len(negative_tweets[\"text\"].values)}')\nprint(f'Number of overlapping tweets: {len(np.intersect1d(positive_tweets[\"text\"].values, negative_tweets[\"text\"].values))}')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:31:22.927127Z","iopub.execute_input":"2022-07-14T19:31:22.927538Z","iopub.status.idle":"2022-07-14T19:31:23.766135Z","shell.execute_reply.started":"2022-07-14T19:31:22.927508Z","shell.execute_reply":"2022-07-14T19:31:23.765181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Words aren’t things that computers naturally understand. By encoding them in a numeric form, we can apply mathematical rules and do matrix operations to them.\n\nOne of the most basic ways we can numerically represent words is through the one-hot encoding method (also sometimes called count vectorizing). The idea is super simple. Create a vector that has as many dimensions as your corpora has unique words. Each unique word has a unique dimension and will be represented by a 1 in that dimension with 0s everywhere else.\n\nHowever, this transformation has no relational information. \n\nTF-IDF vectors are related to one-hot encoded vectors. However, instead of just featuring a count, they feature numerical representations where words aren’t just there or not there. Instead, words are represented by their term frequency multiplied by their inverse document frequency. In simpler terms, words that occur a lot but everywhere should be given very little weighting or significance. We can think of this as words like the or and in the English language. They don’t provide a large amount of value.\n\nHowever, if a word appears very little or appears frequently, but only in one or two places, then these are probably more important words and should be weighted as such.\n\nWe will start by ysing the TF-IDF method. ","metadata":{}},{"cell_type":"code","source":"def TFIDF_vectorizer(train_data, test_data):\n    tfidf_vectorizer = feature_extraction.text.TfidfVectorizer()\n    tfidf_train_vectors = tfidf_vectorizer.fit_transform(train_data)\n    tfidf_test_vectors = tfidf_vectorizer.transform(test_data)\n    \n    return tfidf_train_vectors, tfidf_test_vectors\n\nexample_train_vector, _ = TFIDF_vectorizer(train_df[\"text\"][0:5], test_df[\"text\"][0:5])\nprint(example_train_vector[0].todense())","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:31:41.921666Z","iopub.execute_input":"2022-07-14T19:31:41.921992Z","iopub.status.idle":"2022-07-14T19:31:41.938775Z","shell.execute_reply.started":"2022-07-14T19:31:41.921966Z","shell.execute_reply":"2022-07-14T19:31:41.937575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's transform the train and test data:","metadata":{}},{"cell_type":"code","source":"train_vectors_tfidf, test_vectors_tfidf = TFIDF_vectorizer(train_df[\"text\"], test_df[\"text\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:31:43.578282Z","iopub.execute_input":"2022-07-14T19:31:43.579186Z","iopub.status.idle":"2022-07-14T19:31:43.697146Z","shell.execute_reply.started":"2022-07-14T19:31:43.579141Z","shell.execute_reply":"2022-07-14T19:31:43.696213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"First model we will try is the Ridge Regression. Our vectors are really big, so we want to push our model's weights toward 0 without completely discounting different words - ridge regression is a good way to do this.","metadata":{}},{"cell_type":"code","source":"clf_Ridge = linear_model.RidgeClassifier()\nf1_scorer = make_scorer(f1_score, pos_label=1)\n\n\n# RANDOM SEARCH FOR 20 COMBINATIONS OF PARAMETERS\nrand_list = {\"alpha\": [1e-2, 1e-1, 1]}\n              \nrand_search_Ridge = RandomizedSearchCV(clf_Ridge, param_distributions = rand_list, n_iter = 10, n_jobs = 1, cv = 3, scoring = f1_scorer) \nrand_search_Ridge.fit(train_vectors_tfidf, train_df[\"target\"]) \n\n#clf = RidgeCV(alphas=[1e-2, 1e-1, 1]).fit(train_vectors_tfidf, train_df[\"target\"])","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:31:45.353002Z","iopub.execute_input":"2022-07-14T19:31:45.353336Z","iopub.status.idle":"2022-07-14T19:31:45.764830Z","shell.execute_reply.started":"2022-07-14T19:31:45.353313Z","shell.execute_reply":"2022-07-14T19:31:45.764046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Ridge_score = rand_search_Ridge.score(train_vectors_tfidf, train_df[\"target\"])\n\nprint(f'The score of the Ridge regression on the training data is {Ridge_score}')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:31:47.334432Z","iopub.execute_input":"2022-07-14T19:31:47.334736Z","iopub.status.idle":"2022-07-14T19:31:47.343625Z","shell.execute_reply.started":"2022-07-14T19:31:47.334713Z","shell.execute_reply":"2022-07-14T19:31:47.342595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Next model we will try is the Support Vector Classification (SVC), which also often proves as a good model for NLP. ","metadata":{}},{"cell_type":"code","source":"mdl = SVC(probability = True, random_state = 1)\n\n\n# RANDOM SEARCH FOR 20 COMBINATIONS OF PARAMETERS\nrand_list = {\"C\": np.linspace(2,10,5),\n             \"gamma\": np.linspace(0.1,1,5)}\n              \nrand_search_SVC = RandomizedSearchCV(mdl, param_distributions = rand_list, n_iter = 10, n_jobs = 2, cv = 3, scoring = f1_scorer) \nrand_search_SVC.fit(train_vectors_tfidf, train_df[\"target\"]) \n#rand_search.cv_results_","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:31:49.916457Z","iopub.execute_input":"2022-07-14T19:31:49.916766Z","iopub.status.idle":"2022-07-14T19:35:33.464977Z","shell.execute_reply.started":"2022-07-14T19:31:49.916742Z","shell.execute_reply":"2022-07-14T19:35:33.463679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SVC_score = rand_search_SVC.score(train_vectors_tfidf, train_df[\"target\"])\nprint(f'The score of the SVC on the training data is {SVC_score}')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:35:33.467042Z","iopub.execute_input":"2022-07-14T19:35:33.467434Z","iopub.status.idle":"2022-07-14T19:35:36.642560Z","shell.execute_reply.started":"2022-07-14T19:35:33.467394Z","shell.execute_reply":"2022-07-14T19:35:36.641215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets now try a Random Forest Classifier:","metadata":{}},{"cell_type":"code","source":"from sklearn.ensemble import RandomForestClassifier\n\nrfc = RandomForestClassifier(random_state=42)\n\nparam_grid = { \n    'n_estimators': [20, 50, 70, 80, 100, 150, 200],\n    'max_features': ['auto', 'sqrt', 'log2'],\n    'max_depth' : [100, 150, 200, 250, 300, 350, 400, 450, 500],\n}\n\nrand_search_rfc = RandomizedSearchCV(rfc, param_distributions = param_grid, n_iter = 10, n_jobs = 2, cv = 3, scoring = f1_scorer) \nrand_search_rfc.fit(train_vectors_tfidf, train_df[\"target\"]) ","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:35:36.643578Z","iopub.execute_input":"2022-07-14T19:35:36.643899Z","iopub.status.idle":"2022-07-14T19:38:24.274293Z","shell.execute_reply.started":"2022-07-14T19:35:36.643868Z","shell.execute_reply":"2022-07-14T19:38:24.273630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rfc_score = rand_search_rfc.score(train_vectors_tfidf, train_df[\"target\"])\nprint(f'The score of the Random Forest Classifier on the training data is {rfc_score}')\nprint(f'With the best parameters being: {rand_search_rfc.best_params_}')","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:38:24.277324Z","iopub.execute_input":"2022-07-14T19:38:24.277723Z","iopub.status.idle":"2022-07-14T19:38:24.788909Z","shell.execute_reply.started":"2022-07-14T19:38:24.277692Z","shell.execute_reply":"2022-07-14T19:38:24.787211Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will take the model with the highest score and to use in the competition:","metadata":{}},{"cell_type":"code","source":"if Ridge_score > SVC_score and Ridge_score > rfc_score:\n    print(\"Ridge regression won!\")\n    result = rand_search_Ridge.predict(test_vectors_tfidf)\n    result[result >= 0.5] = 1\n    result[result < 0.5] = 0\n\n    sample_submission = pd.read_csv('../input/nlp-getting-started/sample_submission.csv')\n    sample_submission['target'] = result.reshape(1,result.shape[0])[0]\n    sample_submission['target'] = sample_submission['target'].astype(int)\n    #sample_submission.head()\n    sample_submission.to_csv('submission_Ridge_tfidf.csv', index=False)\nelif SVC_score > rfc_score:\n    print(\"SVC won!\")\n    result = rand_search_SVC.predict(test_vectors_tfidf)\n    result[result >= 0.5] = 1\n    result[result < 0.5] = 0\n\n    sample_submission = pd.read_csv('../input/nlp-getting-started/sample_submission.csv')\n    sample_submission['target'] = result.reshape(1,result.shape[0])[0]\n    sample_submission['target'] = sample_submission['target'].astype(int)\n    #sample_submission.head()\n    sample_submission.to_csv('submission_SVC_tfidf.csv', index=False)\nelse:\n    print(\"Random Forest won!\")\n    result = rand_search_rfc.predict(test_vectors_tfidf)\n    result[result >= 0.5] = 1\n    result[result < 0.5] = 0\n\n    sample_submission = pd.read_csv('../input/nlp-getting-started/sample_submission.csv')\n    sample_submission['target'] = result.reshape(1,result.shape[0])[0]\n    sample_submission['target'] = sample_submission['target'].astype(int)\n    #sample_submission.head()\n    sample_submission.to_csv('submission_RFC_tfidf.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-14T19:38:24.790595Z","iopub.execute_input":"2022-07-14T19:38:24.791090Z","iopub.status.idle":"2022-07-14T19:38:24.915535Z","shell.execute_reply.started":"2022-07-14T19:38:24.791059Z","shell.execute_reply":"2022-07-14T19:38:24.914554Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}