{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Disaster Tweets","metadata":{}},{"cell_type":"markdown","source":"### Imports","metadata":{}},{"cell_type":"code","source":"! pip install pycld3 # language detection","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:02:52.079579Z","iopub.execute_input":"2022-07-05T22:02:52.080985Z","iopub.status.idle":"2022-07-05T22:03:05.577100Z","shell.execute_reply.started":"2022-07-05T22:02:52.080808Z","shell.execute_reply":"2022-07-05T22:03:05.575780Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data manipulation\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# plotting\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom matplotlib_venn import venn2 # Venn diagrams\n\n# text analysis\nfrom cld3 import get_language\nfrom wordcloud import WordCloud\n\n# text preprocessing\nimport regex as re\n\n# model building\nfrom sklearn.model_selection import train_test_split\nfrom sklearn import feature_extraction\nfrom sklearn.naive_bayes import CategoricalNB, MultinomialNB\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\nfrom sklearn.linear_model import SGDClassifier, LogisticRegression\nfrom sklearn.pipeline import Pipeline\nfrom sklearn.ensemble import RandomForestClassifier\nfrom scipy.special import softmax\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.metrics import mean_squared_error as mse\n\nrandom_state = 69","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:11:19.930029Z","iopub.execute_input":"2022-07-05T22:11:19.930428Z","iopub.status.idle":"2022-07-05T22:11:19.938565Z","shell.execute_reply.started":"2022-07-05T22:11:19.930396Z","shell.execute_reply":"2022-07-05T22:11:19.937765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data loading","metadata":{}},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-05T22:03:06.683003Z","iopub.execute_input":"2022-07-05T22:03:06.683322Z","iopub.status.idle":"2022-07-05T22:03:06.692070Z","shell.execute_reply.started":"2022-07-05T22:03:06.683295Z","shell.execute_reply":"2022-07-05T22:03:06.690953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data = pd.read_csv('/kaggle/input/nlp-getting-started/train.csv')\ntest_data = pd.read_csv('/kaggle/input/nlp-getting-started/test.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:03:06.694061Z","iopub.execute_input":"2022-07-05T22:03:06.694661Z","iopub.status.idle":"2022-07-05T22:03:06.774156Z","shell.execute_reply.started":"2022-07-05T22:03:06.694604Z","shell.execute_reply":"2022-07-05T22:03:06.773130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"train_data.columns","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:03:06.775363Z","iopub.execute_input":"2022-07-05T22:03:06.775670Z","iopub.status.idle":"2022-07-05T22:03:06.786532Z","shell.execute_reply.started":"2022-07-05T22:03:06.775643Z","shell.execute_reply":"2022-07-05T22:03:06.785550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Location","metadata":{}},{"cell_type":"code","source":"train_locations = set(train_data['location'])\ntest_locations = set(test_data['location'])\nf, ax = plt.subplots(1, 2, figsize=(14, 7))\nvenn2((train_locations, test_locations), ('train', 'test'), ax=ax[0],\n      set_colors=('darkseagreen', 'coral'), alpha=0.8)\nax[0].title.set_text('Unique locations for train/test datasets')\nloc_intersept = set.intersection(train_locations, test_locations)\nax[1].bar(x=['train unique', 'common'],\n        height=[len([x for x in train_data.location if x not in loc_intersept]),\n               len([x for x in train_data.location if x in loc_intersept])],\n          color = 'darkseagreen', alpha=0.8)\n\nax[1].bar(x=['common', 'test unique'],\n        height=[len([x for x in test_data.location if x in loc_intersept]),\n                len([x for x in test_data.location if x not in loc_intersept])],\n        color = 'coral', alpha=0.8,\n        align='edge')\nax[1].title.set_text('Tweet counts with locations common in train-test datasets');","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:03:06.788179Z","iopub.execute_input":"2022-07-05T22:03:06.790202Z","iopub.status.idle":"2022-07-05T22:03:07.113512Z","shell.execute_reply.started":"2022-07-05T22:03:06.790151Z","shell.execute_reply":"2022-07-05T22:03:07.112513Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(1, 2, figsize=(14, 7))\n\ndis_locations = set(train_data[train_data.target == 1].location)\ncas_locations = set(train_data[train_data.target == 0].location)\ndiscas_intersept = set.intersection(dis_locations, cas_locations)\nvenn2((dis_locations, cas_locations), ('disaster', 'casual'), ax=ax[0], set_colors=('tab:blue', 'tab:orange'), alpha=0.8)\nax[0].title.set_text('Unique locations for disaster/casual tweets')\n\nax[1].bar(x=['disaster unique', 'common'],\n        height=[len([x for x in train_data[train_data.target==1].location if x not in discas_intersept]),\n                len([x for x in train_data[train_data.target==1].location if x in discas_intersept])],\n        alpha=0.8)\nax[1].bar(x=['common', 'casual unique'],\n        height=[len([x for x in test_data[train_data.target==0].location if x in discas_intersept]),\n                len([x for x in test_data[train_data.target==0].location if x not in discas_intersept])],\n        align='edge',\n        alpha=0.8)\nax[1].legend(['disaster', 'casual'])\n\nax[1].title.set_text('Tweet counts with locations common in disaster-casual datasets');","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:03:07.114636Z","iopub.execute_input":"2022-07-05T22:03:07.115000Z","iopub.status.idle":"2022-07-05T22:03:07.420564Z","shell.execute_reply.started":"2022-07-05T22:03:07.114969Z","shell.execute_reply":"2022-07-05T22:03:07.419333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We will use this information as a feature! (Look into Categorical NB part feat_extraction function)","metadata":{}},{"cell_type":"code","source":"f, ax = plt.subplots(1, 2, figsize=(24, 6), gridspec_kw={'width_ratios': [1, 6]})\n\ntrain_data['location_specified'] = [1 if type(loc)==str else 0 for loc in train_data.location]\n\nplt.rc('xtick', labelsize=9)\nax[0].hist(train_data.location_specified[train_data.target==1], bins=4, align='left', alpha=0.8)\nax[0].hist(train_data.location_specified[train_data.target==0], bins=4, align='mid', alpha=0.7)\nax[0].legend(['disaster tweet', 'casual tweet'])\nax[0].title.set_text('Is location specified (0 or 1)');\n\nn_locations = 12\n\nLocationCountsDisaster = train_data[train_data.target==1][train_data.location_specified == True].location.value_counts()\nLocationCountsCasual = train_data[train_data.target==0][train_data.location_specified == True].location.value_counts()\nLocationCounts = pd.concat([LocationCountsDisaster, LocationCountsCasual], axis=1)\nLocationCounts.columns = ['disaster', 'casual']\n\nax[1].bar(x=LocationCounts.index[:n_locations], height=LocationCounts.disaster[:n_locations], alpha=0.8)\nax[1].bar(x=LocationCounts.index[:n_locations], height=LocationCounts.casual[:n_locations], alpha=0.6)\nax[1].legend(['disaster', 'casual'])\nax[1].title.set_text(f'Top {n_locations} locations for disaster tweets');","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:03:07.421912Z","iopub.execute_input":"2022-07-05T22:03:07.422275Z","iopub.status.idle":"2022-07-05T22:03:07.915565Z","shell.execute_reply.started":"2022-07-05T22:03:07.422242Z","shell.execute_reply":"2022-07-05T22:03:07.914606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Keyword","metadata":{}},{"cell_type":"markdown","source":"Train and test datasets share same keywords","metadata":{}},{"cell_type":"code","source":"train_keywords = set(list(train_data['keyword']))\ntest_keywords = set(list(test_data['keyword']))\nprint(f'train keywords: {len(train_keywords)}')\nprint(f'test keywords: {len(test_keywords)}')\nprint(f'train-test intersecting keywords: {len(set.intersection(train_keywords, test_keywords))}')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:03:07.917312Z","iopub.execute_input":"2022-07-05T22:03:07.917786Z","iopub.status.idle":"2022-07-05T22:03:07.926292Z","shell.execute_reply.started":"2022-07-05T22:03:07.917741Z","shell.execute_reply":"2022-07-05T22:03:07.925360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's compare disaster tweets rate for most common keywords","metadata":{}},{"cell_type":"code","source":"n_keywords = 8\n\nKeywordCountsDisaster = train_data[train_data.target==1].keyword.value_counts()\nKeywordCountsCasual = train_data[train_data.target==0].keyword.value_counts()\nKeywordCounts = pd.concat([KeywordCountsDisaster, KeywordCountsCasual], axis=1)\nKeywordCounts.columns = ['disaster', 'casual']\n\nf, ax = plt.subplots(1, 2, figsize=(24, 6))\n\nplt.rc('xtick', labelsize=8)\nax[0].bar(x=KeywordCounts.index[:n_keywords], height=KeywordCounts.disaster[:n_keywords], alpha=0.8)\nax[0].bar(x=KeywordCounts.index[:n_keywords], height=KeywordCounts.casual[:n_keywords], alpha=0.7)\nax[0].legend(['disaster', 'casual'])\nax[0].title.set_text(f'Tweet counts for top {n_keywords} keywords for disaster')\n\nKeywordCounts = KeywordCounts.sort_values(by='casual', ascending=False)\nax[1].bar(x=KeywordCounts.index[:n_keywords], height=KeywordCounts.casual[:n_keywords], color='#ff7f0e', alpha=0.6)\nax[1].bar(x=KeywordCounts.index[:n_keywords], height=KeywordCounts.disaster[:n_keywords], color='#1f77b4', alpha=0.8)\nax[1].legend(['casual', 'disaster'])\nax[1].title.set_text(f'Tweet counts for top {n_keywords} keywords for casual');","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:03:07.929238Z","iopub.execute_input":"2022-07-05T22:03:07.929589Z","iopub.status.idle":"2022-07-05T22:03:08.373360Z","shell.execute_reply.started":"2022-07-05T22:03:07.929558Z","shell.execute_reply":"2022-07-05T22:03:08.372247Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_keywords=len(KeywordCounts)\n\nKeywordCounts = KeywordCounts.sort_values(by='casual', ascending=False)\nplt.figure(figsize=(22, 6))\nplt.rc('xtick', labelsize=10)\nplt.bar(x=range(n_keywords), height=KeywordCounts.casual[:n_keywords], color='#ff7f0e', alpha=0.6)\nplt.bar(x=range(n_keywords), height=KeywordCounts.disaster[:n_keywords], color='#1f77b4', alpha=0.8)\nplt.legend(['casual', 'disaster'])\nplt.title(f'Tweet counts for all keywords');","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:03:08.374754Z","iopub.execute_input":"2022-07-05T22:03:08.375219Z","iopub.status.idle":"2022-07-05T22:03:09.804588Z","shell.execute_reply.started":"2022-07-05T22:03:08.375175Z","shell.execute_reply":"2022-07-05T22:03:09.803528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Tweet language","metadata":{}},{"cell_type":"code","source":"train_data['lang'] = [get_language(x).language for x in train_data['text']]\ntest_data['lang'] = [get_language(x).language for x in test_data['text']]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:03:09.806261Z","iopub.execute_input":"2022-07-05T22:03:09.806784Z","iopub.status.idle":"2022-07-05T22:03:12.637224Z","shell.execute_reply.started":"2022-07-05T22:03:09.806747Z","shell.execute_reply":"2022-07-05T22:03:12.636136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(1, 3, figsize=(24, 6), gridspec_kw={'width_ratios': [2, 1, 4]})\n\nn_langs = 15\nax[0].pie([train_data.lang.value_counts()[0], train_data.lang.value_counts()[1:].sum()], labels=['en', 'non_en'], colors=('indianred', 'tan'))\nax[0].title.set_text('English/non-english tweet counts')\n\nLanguageCountsDisaster = train_data[train_data['target']==1].lang.value_counts()\nLanguageCountsCasual = train_data[train_data['target']==0].lang.value_counts()\nLanguageCounts = pd.concat([LanguageCountsDisaster, LanguageCountsCasual], axis=1)\nLanguageCounts.columns = ['disaster', 'casual']\n\nax[1].bar(x=['disaster'], height=LanguageCounts.disaster[:1], alpha=0.8)\nax[1].bar(x=['casual'], height=LanguageCounts.casual[:1], alpha=0.7)\nax[1].title.set_text(f'English tweets distribution')\n          \nax[2].bar(x=LanguageCounts.index[1:n_langs], height=LanguageCounts.disaster[1:n_langs], alpha=0.8)\nax[2].bar(x=LanguageCounts.index[1:n_langs], height=LanguageCounts.casual[1:n_langs], alpha=0.7)\nax[2].title.set_text(f'Top {n_langs} languages by tweet count except for english')\nax[2].legend(['disaster', 'casual']);","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:03:12.638635Z","iopub.execute_input":"2022-07-05T22:03:12.639357Z","iopub.status.idle":"2022-07-05T22:03:13.116476Z","shell.execute_reply.started":"2022-07-05T22:03:12.639310Z","shell.execute_reply":"2022-07-05T22:03:13.115485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Tweet length","metadata":{}},{"cell_type":"code","source":"f, ax = plt.subplots(1, 2, figsize=(24, 6))\n\ntrain_data['text_length'] = [len(text) for text in train_data['text']]\ntest_data['text_length'] = [len(text) for text in test_data['text']]\n\n\nax[0].hist(train_data['text_length'][train_data['target']==1], bins=range(0, 144, 2), alpha=0.8)\nax[0].hist(train_data['text_length'][train_data['target']==0], bins=range(0, 144, 2), alpha=0.8)\nax[0].title.set_text('Text length histogram')\nax[0].legend(['disaster', 'casual'])\n\ntrain_data['text_wordcount'] = [len(text.split()) for text in train_data['text']]\n\nax[1].hist(train_data['text_wordcount'][train_data['target']==1], bins=range(40), alpha=0.8)\nax[1].hist(train_data['text_wordcount'][train_data['target']==0], bins=range(40), alpha=0.8)\nax[1].title.set_text('Word count histogram')\nax[1].legend(['disaster', 'casual']);","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:03:13.117761Z","iopub.execute_input":"2022-07-05T22:03:13.118126Z","iopub.status.idle":"2022-07-05T22:03:13.984094Z","shell.execute_reply.started":"2022-07-05T22:03:13.118093Z","shell.execute_reply":"2022-07-05T22:03:13.982980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Simple text features","metadata":{}},{"cell_type":"code","source":"f, ax = plt.subplots(4, 2, figsize=(24, 18))\ndensity = False\n\ntrain_data['uppercase_count'] = [sum([c.isupper() for c in text]) for text in train_data['text']]\ntest_data['uppercase_count'] = [sum([c.isupper() for c in text]) for text in test_data['text']]\n\n\nax[0][0].hist(train_data[train_data.target==1].uppercase_count, bins=range(0,80,2), alpha=0.8, density=density)\nax[0][0].hist(train_data[train_data.target==0].uppercase_count, bins=range(0,80,2), alpha=0.8, density=density)\nax[0][0].title.set_text('Uppercase letters count histogram')\nax[0][0].legend(['disaster', 'casual'])\n\npunctuation_signs = {'.',',','!',':','\"','?',';','-','(', ')','&'}\ntrain_data['punctuation_count'] = [sum([c in punctuation_signs for c in text]) \n                                   for text in train_data['text']]\n\nax[0][1].hist(train_data[train_data.target==1].punctuation_count, bins=range(16), alpha=0.8, density=density)\nax[0][1].hist(train_data[train_data.target==0].punctuation_count, bins=range(16), alpha=0.7, density=density)\nax[0][1].title.set_text('Punctuation signs count histogram')\nax[0][1].legend(['disaster', 'casual'])\n\ntrain_data['number_count'] = [sum([word.isnumeric() for word in text.split()]) \n                                   for text in train_data['text']]\ntest_data['number_count'] = [sum([word.isnumeric() for word in text.split()]) \n                                  for text in test_data['text']]\nax[1][0].hist(train_data[train_data.target==1].number_count, bins=range(5), alpha=0.8, density=density)\nax[1][0].hist(train_data[train_data.target==0].number_count, bins=range(5), alpha=0.7, density=density)\nax[1][0].title.set_text('Numbers count histogram')\nax[1][0].legend(['disaster', 'casual'])\n\n\ntrain_data['special_count'] = [sum([not c.isalnum() for c in text]) \n                                   for text in train_data['text']]\nax[1][1].hist(train_data[train_data.target==1].special_count, bins=range(0, 64), alpha=0.8, density=density)\nax[1][1].hist(train_data[train_data.target==0].special_count, bins=range(0, 64), alpha=0.7, density=density)\nax[1][1].title.set_text('Special symbols count histogram')\nax[1][1].legend(['disaster', 'casual'])\n\n\ntrain_data['hashtag_count'] = [sum(x == '#' for x in text) for text in train_data['text']]\ntest_data['hashtag_count'] = [sum(x == '#' for x in text) for text in test_data['text']]\nax[2][0].hist(train_data[train_data.target==1].hashtag_count, bins=range(0, 5), alpha=0.8, density=density)\nax[2][0].hist(train_data[train_data.target==0].hashtag_count, bins=range(0, 5), alpha=0.7, density=density)\nax[2][0].title.set_text('Hashtag count histogram')\nax[2][0].legend(['disaster', 'casual'])\n\n\ntrain_data['at_count'] = [sum(x == '@' for x in text) for text in train_data['text']]\nax[2][1].hist(train_data[train_data.target==1].at_count, bins=range(0, 5), alpha=0.8, density=density)\nax[2][1].hist(train_data[train_data.target==0].at_count, bins=range(0, 5), alpha=0.7, density=density)\nax[2][1].title.set_text('At(@) count histogram')\nax[2][1].legend(['disaster', 'casual'])\n\n\ntrain_data['link_count'] = [sum(1 if 'http' in word else 0 for word in text.split()) for text in train_data['text']]\ntest_data['link_count'] = [sum(1 if 'http' in word else 0 for word in text.split()) for text in test_data['text']]\nax[3][0].hist(train_data[train_data.target==1].link_count, bins=range(0, 5), width=0.3, alpha=0.8, density=density)\nax[3][0].hist(train_data[train_data.target==0].link_count, bins=range(0, 5), width=0.3, alpha=0.7, density=density)\nax[3][0].title.set_text('Tweets link(http) count')\nax[3][0].legend(['disaster', 'casual'])\n\ntrain_data['hashtag_numbers_link'] = train_data.hashtag_count + train_data.number_count \\\n                                     + train_data.link_count\ntest_data['hashtag_numbers_link'] = test_data.hashtag_count + test_data.number_count \\\n                                    + test_data.link_count\n\nax[3][1].hist(train_data[train_data.target==1].hashtag_numbers_link, bins=range(0, 12), width=0.3, alpha=0.8, density=density)\nax[3][1].hist(train_data[train_data.target==0].hashtag_numbers_link, bins=range(0, 12), width=0.3, alpha=0.7, density=density)\nax[3][1].title.set_text('hashtag_cnt + number_cnt + has_link')\nax[3][1].legend(['disaster', 'casual']);\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:03:13.985789Z","iopub.execute_input":"2022-07-05T22:03:13.986290Z","iopub.status.idle":"2022-07-05T22:03:16.566623Z","shell.execute_reply.started":"2022-07-05T22:03:13.986242Z","shell.execute_reply":"2022-07-05T22:03:16.565847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Words","metadata":{}},{"cell_type":"code","source":"train_data['tokens'] = [x.split() for x in train_data['text']]\n\ntrain_tweets_combined = train_data.tokens.sum()\ntrain_tweets_dis_comb = train_data[train_data.target == 1].tokens.sum()\ntrain_tweets_cas_comb = train_data[train_data.target == 0].tokens.sum()\n\nwordcloud_dis = WordCloud(background_color='#345E7B').generate(' '.join(train_tweets_dis_comb))\nwordcloud_cas = WordCloud(background_color='#785435').generate(' '.join(train_tweets_cas_comb))\n\nf, ax = plt.subplots(1, 2, figsize=(20, 14))\nax[0].imshow(wordcloud_dis)\nax[0].title.set_text('Disaster tweets')\nax[1].imshow(wordcloud_cas)\nax[1].title.set_text('Casual tweets')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:03:16.567629Z","iopub.execute_input":"2022-07-05T22:03:16.568306Z","iopub.status.idle":"2022-07-05T22:03:22.356812Z","shell.execute_reply.started":"2022-07-05T22:03:16.568268Z","shell.execute_reply":"2022-07-05T22:03:22.355845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"words_n = 30\n\nfrom collections import Counter\ndis_counter = Counter(train_tweets_dis_comb)\ndis_counts = dis_counter.most_common(words_n)\ncas_counter = Counter(train_tweets_cas_comb)\ncas_counts = [(k, cas_counter[k]) for k, _ in dis_counts]\n\nplt.figure(figsize=(20, 6))\nplt.bar(*list(zip(*dis_counts)), alpha=0.7)\nplt.bar(*list(zip(*cas_counts)), alpha=0.7)\nplt.legend(['disaster', 'casual']);","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:03:22.358092Z","iopub.execute_input":"2022-07-05T22:03:22.358431Z","iopub.status.idle":"2022-07-05T22:03:22.801638Z","shell.execute_reply.started":"2022-07-05T22:03:22.358400Z","shell.execute_reply":"2022-07-05T22:03:22.800937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_data['tokens'] = [x.split() for x in train_data['text']]\n\ntrain_tweets_combined = train_data.tokens.sum()\ntrain_tweets_dis_comb = train_data[train_data.target == 1].tokens.sum()\ntrain_tweets_cas_comb = train_data[train_data.target == 0].tokens.sum()\n\nwordcloud_dis = WordCloud(background_color='#345E7B').generate(' '.join(train_tweets_dis_comb))\nwordcloud_cas = WordCloud(background_color='#785435').generate(' '.join(train_tweets_cas_comb))\n\nf, ax = plt.subplots(1, 2, figsize=(20, 14))\nax[0].imshow(wordcloud_dis)\nax[0].title.set_text('Disaster tweets preprocessed')\nax[1].imshow(wordcloud_cas)\nax[1].title.set_text('Casual tweets preprocessed')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:04:11.630762Z","iopub.execute_input":"2022-07-05T22:04:11.631232Z","iopub.status.idle":"2022-07-05T22:04:17.549949Z","shell.execute_reply.started":"2022-07-05T22:04:11.631199Z","shell.execute_reply":"2022-07-05T22:04:17.548894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"words_n = 30\n\nfrom collections import Counter\ndis_counter = Counter(train_tweets_dis_comb)\ndis_counts = dis_counter.most_common(words_n)\ncas_counter = Counter(train_tweets_cas_comb)\ncas_counts = [(k, cas_counter[k]) for k, _ in dis_counts]\n\nplt.figure(figsize=(20, 6))\nplt.bar(*list(zip(*dis_counts)), alpha=0.7)\nplt.bar(*list(zip(*cas_counts)), alpha=0.7)\nplt.legend(['disaster', 'casual']);","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:04:17.551529Z","iopub.execute_input":"2022-07-05T22:04:17.551944Z","iopub.status.idle":"2022-07-05T22:04:18.013876Z","shell.execute_reply.started":"2022-07-05T22:04:17.551910Z","shell.execute_reply":"2022-07-05T22:04:18.013142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Base models","metadata":{}},{"cell_type":"markdown","source":"### Categorical NB on keywords, location and tweet language only","metadata":{}},{"cell_type":"code","source":"# feat_features = ['keyword', 'lang', 'location', 'text_length', 'hashtag_numbers_link']\nfeat_features = ['keyword', 'lang', 'location']\n\n\nX = train_data[feat_features]\nX_test = test_data[feat_features]\n\nmerged_X = pd.concat([X, X_test])\n\ndef feat_extraction(X):\n    X.keyword[X.keyword.isna()] = 'None'\n    X.keyword = X.keyword.astype('category')\n    X.lang = X.lang.astype('category').cat.codes\n    X.location = ['Disaster' if x in dis_locations \n                  else 'Casual' if x in cas_locations \n                  else 'Common' if discas_intersept \n                  else 'NoInfo' for x in X.location]\n    X.location = X.location.astype('category').cat.codes\n    \n#     X['text_brackets'] = X.text_length.apply(lambda x: 0 if x < 30 else 2 if x > 134 else 1).astype('category')\n#     X['no_tag_num_link'] = X.hashtag_numbers_link.apply(lambda x: 0 if x == 0 else 1)\n#     X.drop(['hashtag_numbers_link', 'text_length'], axis=1)\n    \n    X = pd.get_dummies(X)\n    return X\n\nmerged_X = feat_extraction(merged_X)\n\n\nX = merged_X[:len(train_data)]\nX_test = merged_X[len(train_data):]\n\ny = train_data['target']\nX_train, X_dev, y_train, y_dev = train_test_split(X, y, test_size=0.2, random_state=random_state)\n\nnb_cls1 = CategoricalNB()\nnb_cls1.fit(X_train, y_train)\nprint(f'Categorical NB Classificator(keyword, location, language) score: {round(nb_cls1.score(X_dev, y_dev),3)}')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:04:18.014994Z","iopub.execute_input":"2022-07-05T22:04:18.015616Z","iopub.status.idle":"2022-07-05T22:04:18.157967Z","shell.execute_reply.started":"2022-07-05T22:04:18.015579Z","shell.execute_reply":"2022-07-05T22:04:18.156948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Multinomial NB on text","metadata":{}},{"cell_type":"code","source":"X = train_data['text']\ny = train_data['target']\nX_test = test_data['text']\nmerged_X = pd.concat([X, X_test])\n\nvectorizer = CountVectorizer(strip_accents='unicode')\nmerged_X = vectorizer.fit_transform(merged_X)\n\nX = merged_X[:len(train_data)]\nX_test = merged_X[len(train_data):]\nX_train, X_dev, y_train, y_dev = train_test_split(X, y, test_size=0.2, random_state=random_state)\n\nnb_cls2 = MultinomialNB()\nnb_cls2.fit(X_train, y_train)\nprint(f'Multinomial NB Classificator(text) score: {round(nb_cls2.score(X_dev, y_dev),3)}')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:04:18.159759Z","iopub.execute_input":"2022-07-05T22:04:18.160120Z","iopub.status.idle":"2022-07-05T22:04:18.540269Z","shell.execute_reply.started":"2022-07-05T22:04:18.160088Z","shell.execute_reply":"2022-07-05T22:04:18.539116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### TF-IDF with LogisticRegression","metadata":{}},{"cell_type":"code","source":"X = train_data['text']\ny = train_data['target']\nX_test = test_data['text']\nmerged_X = pd.concat([X, X_test])\n\nvectorizer = TfidfVectorizer(strip_accents='unicode')\nmerged_X = vectorizer.fit_transform(merged_X)\n\nX = merged_X[:len(train_data)]\nX_test = merged_X[len(train_data):]\nX_train, X_dev, y_train, y_dev = train_test_split(X, y, test_size=0.2, random_state=random_state)\n\nlr_cls1 = LogisticRegression()\nlr_cls1.fit(X_train, y_train)\nprint(f'TF-IDF transform with logistic classifier: {round(lr_cls1.score(X_dev, y_dev),3)}')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:04:18.920638Z","iopub.execute_input":"2022-07-05T22:04:18.921590Z","iopub.status.idle":"2022-07-05T22:04:20.022574Z","shell.execute_reply.started":"2022-07-05T22:04:18.921551Z","shell.execute_reply":"2022-07-05T22:04:20.021281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### TF-IDF with KNeighborsClassifier","metadata":{}},{"cell_type":"code","source":"from sklearn.neighbors import KNeighborsClassifier\n\nX = train_data['text']\ny = train_data['target']\nX_test = test_data['text']\nmerged_X = pd.concat([X, X_test])\n\nvectorizer = TfidfVectorizer(strip_accents='unicode')\nmerged_X = vectorizer.fit_transform(merged_X)\n\nX = merged_X[:len(train_data)]\nX_test = merged_X[len(train_data):]\nX_train, X_dev, y_train, y_dev = train_test_split(X, y, test_size=0.2, random_state=random_state)\n\nn_neighbors = 100\nkn_cls1 = KNeighborsClassifier(n_neighbors)\nkn_cls1.fit(X_train, y_train)\nprint(f'TF-IDF transform with KNN classifier(neighbors={n_neighbors}): {round(kn_cls1.score(X_dev, y_dev),3)}')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:13:09.929378Z","iopub.execute_input":"2022-07-05T22:13:09.929891Z","iopub.status.idle":"2022-07-05T22:13:10.768062Z","shell.execute_reply.started":"2022-07-05T22:13:09.929831Z","shell.execute_reply":"2022-07-05T22:13:10.767094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Ensembling (three approaches)","metadata":{}},{"cell_type":"code","source":"text_feature = 'text'","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:04:21.958537Z","iopub.execute_input":"2022-07-05T22:04:21.959142Z","iopub.status.idle":"2022-07-05T22:04:21.963521Z","shell.execute_reply.started":"2022-07-05T22:04:21.959102Z","shell.execute_reply":"2022-07-05T22:04:21.962715Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_feature_vectors(dataset, features):\n    count_vectorizer = CountVectorizer(strip_accents='unicode')\n    tfidf_vectorizer = TfidfVectorizer(strip_accents='unicode')\n\n    count_x = count_vectorizer.fit_transform(dataset[text_feature])\n    tfidf_x = tfidf_vectorizer.fit_transform(dataset[text_feature])\n    feat_x = feat_extraction(dataset[features])\n    return count_x, tfidf_x, feat_x","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:04:30.541096Z","iopub.execute_input":"2022-07-05T22:04:30.541829Z","iopub.status.idle":"2022-07-05T22:04:30.547909Z","shell.execute_reply.started":"2022-07-05T22:04:30.541772Z","shell.execute_reply":"2022-07-05T22:04:30.547119Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features = feat_features + [text_feature]","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:04:30.839910Z","iopub.execute_input":"2022-07-05T22:04:30.840832Z","iopub.status.idle":"2022-07-05T22:04:30.844643Z","shell.execute_reply.started":"2022-07-05T22:04:30.840789Z","shell.execute_reply":"2022-07-05T22:04:30.843944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### First approach: \n* Split trainset into Train and Dev\n* Train Base models on train and predict probas on Dev\n* Then split Dev preds probas into lvl2_Train and lvl2_Dev\n* Train Second level model on lvl2_Train and evaluate on lvl2_Dev","metadata":{"execution":{"iopub.status.busy":"2022-06-24T13:44:55.161689Z","iopub.execute_input":"2022-06-24T13:44:55.162561Z","iopub.status.idle":"2022-06-24T13:44:55.168646Z","shell.execute_reply.started":"2022-06-24T13:44:55.162501Z","shell.execute_reply":"2022-06-24T13:44:55.167563Z"}}},{"cell_type":"code","source":"X = train_data[features]\nX_test = test_data[features]\nX_train, X_dev, y_train, y_dev = train_test_split(X, y, test_size=0.3, random_state=random_state)\nmerged_X = pd.concat([X_train, X_dev, X_test])\n\ncount_x, tfidf_x, feat_x = get_feature_vectors(merged_X, feat_features)\n\nnb_cls1 = CategoricalNB()\nnb_cls2 = MultinomialNB()\nlr_cls1 = LogisticRegression()\nkn_cls1 = KNeighborsClassifier(n_neighbors=100)\n\nnb_cls1.fit(feat_x[:len(X_train)], y_train)\nnb_cls2.fit(count_x[:len(X_train)], y_train)\nlr_cls1.fit(tfidf_x[:len(X_train)], y_train)\nkn_cls1.fit(tfidf_x[:len(X_train)], y_train)\n\nnb_cls1_dev_preds = nb_cls1.predict_proba(feat_x[len(X_train): len(X_train)+len(X_dev)])[:,1]\nnb_cls2_dev_preds = nb_cls2.predict_proba(count_x[len(X_train): len(X_train)+len(X_dev)])[:,1]\nlr_cls1_dev_preds = lr_cls1.predict_proba(tfidf_x[len(X_train): len(X_train)+len(X_dev)])[:,1]\nkn_cls1_dev_preds = kn_cls1.predict_proba(tfidf_x[len(X_train): len(X_train)+len(X_dev)])[:,1]\n\nlvl2_X = pd.DataFrame({'nb_cls1': nb_cls1_dev_preds,\n                       'nb_cls2': nb_cls2_dev_preds,\n                       'lr_cls1': lr_cls1_dev_preds,\n                       'kn_cls1': kn_cls1_dev_preds})\nlvl2_y = y_dev\nlvl2_X_train, lvl2_X_dev, lvl2_y_train, lvl2_y_dev = train_test_split(lvl2_X, lvl2_y, test_size=0.3, random_state=random_state)\n\nlvl2_lr_cls1 = LogisticRegression()\nlvl2_lr_cls1.fit(lvl2_X_train, lvl2_y_train)\n    \nprint(f'Ensemble score with linear classificator (First approach): {round(lvl2_lr_cls1.score(lvl2_X_dev, lvl2_y_dev),3)}')\nfirst_approach_preds = lvl2_lr_cls1.predict_proba(lvl2_X_dev)\nfirst_approach_target = lvl2_y_dev\n\nlvl2_lr_cls1.fit(lvl2_X, lvl2_y);","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:04:33.841841Z","iopub.execute_input":"2022-07-05T22:04:33.842292Z","iopub.status.idle":"2022-07-05T22:04:35.966460Z","shell.execute_reply.started":"2022-07-05T22:04:33.842257Z","shell.execute_reply":"2022-07-05T22:04:35.965591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Second approach: \n* Split trainset into Train and Dev\n* Train Base models on Train and predict probas on Train\n* Train Second level model on Train predicted probas\n* Evaluate Base models and Second level models on Dev\n\nWould most certainly overfit (We will see it later in histograms)","metadata":{"execution":{"iopub.status.busy":"2022-06-24T13:44:55.17105Z","iopub.execute_input":"2022-06-24T13:44:55.172914Z","iopub.status.idle":"2022-06-24T13:44:55.182696Z","shell.execute_reply.started":"2022-06-24T13:44:55.17285Z","shell.execute_reply":"2022-06-24T13:44:55.181525Z"}}},{"cell_type":"code","source":"X = train_data[features]\nX_test = test_data[features]\n\nX_train, X_dev, y_train, y_dev = train_test_split(X, y, test_size=0.3, random_state=random_state)\nmerged_X = pd.concat([X_train, X_dev, X_test])\n\n# level 1 models and feature generation\ncount_x, tfidf_x, feat_x = get_feature_vectors(merged_X, feat_features)\n\nnb_cls1 = CategoricalNB()\nnb_cls2 = MultinomialNB()\nlr_cls1 = LogisticRegression()\nkn_cls1 = KNeighborsClassifier(n_neighbors=100)\n\nnb_cls1.fit(feat_x[:len(X_train)], y_train)\nnb_cls2.fit(count_x[:len(X_train)], y_train)\nlr_cls1.fit(tfidf_x[:len(X_train)], y_train)\nkn_cls1.fit(tfidf_x[:len(X_train)], y_train)\n\n# level 2 model\nnb_cls1_train_preds = nb_cls1.predict_proba(feat_x[:len(X_train)])[:,1]\nnb_cls2_train_preds = nb_cls2.predict_proba(count_x[:len(X_train)])[:,1]\nlr_cls1_train_preds = lr_cls1.predict_proba(tfidf_x[:len(X_train)])[:,1]\nkn_cls1_train_preds = kn_cls1.predict_proba(tfidf_x[:len(X_train)])[:,1]\n\nlvl2_X_train = pd.DataFrame({'nb_cls1': nb_cls1_train_preds,\n                             'nb_cls2': nb_cls2_train_preds,\n                             'lr_cls1': lr_cls1_train_preds,\n                             'kn_cls1': kn_cls1_train_preds})\n\nnb_cls1_dev_preds = nb_cls1.predict_proba(feat_x[len(X_train): -len(X_test)])[:,1]\nnb_cls2_dev_preds = nb_cls2.predict_proba(count_x[len(X_train): -len(X_test)])[:,1]\nlr_cls1_dev_preds = lr_cls1.predict_proba(tfidf_x[len(X_train): -len(X_test)])[:,1]\n\nlvl2_X_dev = pd.DataFrame({'nb_cls1': nb_cls1_dev_preds,\n                           'nb_cls2': nb_cls2_dev_preds,\n                           'lr_cls1': lr_cls1_dev_preds,\n                           'kn_cls1': lr_cls1_dev_preds})\n\n\nlvl2_lr_cls2 = LogisticRegression()\nlvl2_lr_cls2.fit(lvl2_X_train, y_train)\nprint(f'Ensemble score with linear classificator (Second approach): {round(lvl2_lr_cls2.score(lvl2_X_dev, y_dev),3)}')\nsecond_approach_preds = lvl2_lr_cls2.predict_proba(lvl2_X_dev)\nsecond_approach_target = y_dev","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:04:48.874780Z","iopub.execute_input":"2022-07-05T22:04:48.875632Z","iopub.status.idle":"2022-07-05T22:04:51.670268Z","shell.execute_reply.started":"2022-07-05T22:04:48.875574Z","shell.execute_reply":"2022-07-05T22:04:51.669169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Third approach (Voting):\n* Split trainset into Train and Dev\n* Train Base models on Train\n* Choose voting strategy on Dev","metadata":{}},{"cell_type":"code","source":"X = train_data[features]\nX_test = test_data[features]\n\nX_train, X_dev, y_train, y_dev = train_test_split(X, y, test_size=0.3, random_state=random_state)\nmerged_X = pd.concat([X_train, X_dev, X_test])\n\n\n# level 1 models and feature generation\ncount_x, tfidf_x, feat_x = get_feature_vectors(merged_X, feat_features)\n\nnb_cls1 = CategoricalNB()\nnb_cls2 = MultinomialNB()\nlr_cls1 = LogisticRegression()\nkn_cls1 = KNeighborsClassifier(n_neighbors=100)\n\nnb_cls1.fit(feat_x[:len(X_train)], y_train)\nnb_cls2.fit(count_x[:len(X_train)], y_train)\nlr_cls1.fit(tfidf_x[:len(X_train)], y_train)\nkn_cls1.fit(tfidf_x[:len(X_train)], y_train)\n\n\n# level 2 model\nnb_cls1_train_preds = nb_cls1.predict_proba(feat_x[:len(X_train)])[:,1]\nnb_cls2_train_preds = nb_cls2.predict_proba(count_x[:len(X_train)])[:,1]\nlr_cls1_train_preds = lr_cls1.predict_proba(tfidf_x[:len(X_train)])[:,1]\n\nlvl2_X_train = pd.DataFrame({'nb_cls1': nb_cls1_train_preds,\n                             'nb_cls2': nb_cls2_train_preds,\n                             'lr_cls1': lr_cls1_train_preds,\n                             'kn_cls1': kn_cls1_train_preds})\n\nnb_cls1_dev_preds = nb_cls1.predict_proba(feat_x[len(X_train): -len(X_test)])[:,1]\nnb_cls2_dev_preds = nb_cls2.predict_proba(count_x[len(X_train): -len(X_test)])[:,1]\nlr_cls1_dev_preds = lr_cls1.predict_proba(tfidf_x[len(X_train): -len(X_test)])[:,1]\n\nlvl2_X_dev = pd.DataFrame({'nb_cls1': nb_cls1_dev_preds,\n                           'nb_cls2': nb_cls2_dev_preds,\n                           'lr_cls1': lr_cls1_dev_preds,\n                           'kn_cls1': kn_cls1_dev_preds})\ncustom_weights = [0.275, 0.225, 0.3, 0.2]\ndev_preds_proba = lvl2_X_dev['nb_cls1']*custom_weights[0] + lvl2_X_dev['nb_cls2']*custom_weights[1] \\\n                  + lvl2_X_dev['lr_cls1']*custom_weights[2] + lvl2_X_dev['kn_cls1']*custom_weights[3]\ndev_preds = round(dev_preds_proba)\nprint(f'Ensemble score with soft voting (Third approach): {round(accuracy_score(dev_preds, y_dev),3)}')\nthird_approach_preds = dev_preds_proba\nthird_approach_target = y_dev","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:04:54.463944Z","iopub.execute_input":"2022-07-05T22:04:54.464377Z","iopub.status.idle":"2022-07-05T22:04:56.120711Z","shell.execute_reply.started":"2022-07-05T22:04:54.464343Z","shell.execute_reply":"2022-07-05T22:04:56.119620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Choosing voting strategy\n* Goal is to evoid overfitting\n* Distribute model weights according to their quality\n* And don't let overconfident models make decisions","metadata":{}},{"cell_type":"markdown","source":"##### Base models predictions ","metadata":{"execution":{"iopub.status.busy":"2022-06-24T15:30:29.087976Z","iopub.execute_input":"2022-06-24T15:30:29.088523Z","iopub.status.idle":"2022-06-24T15:30:29.094031Z","shell.execute_reply.started":"2022-06-24T15:30:29.088471Z","shell.execute_reply":"2022-06-24T15:30:29.092759Z"}}},{"cell_type":"code","source":"bins = 20\nf, ax = plt.subplots(1, 5, figsize=(24, 4))\nax[0].hist(nb_cls1_dev_preds, color='lightseagreen', bins=bins)\nax[0].title.set_text('NB on keyword, location, language etc')\nax[1].hist(nb_cls2_dev_preds, color='lightseagreen', bins=bins)\nax[1].title.set_text('Multinomial NB for text')\nax[2].hist(lr_cls1_dev_preds, color='lightseagreen', bins=bins)\nax[2].title.set_text('TFIDF with KNN')\nax[3].hist(kn_cls1_dev_preds, color='lightseagreen', bins=bins)\nax[3].title.set_text('TFIDF with logreg')\nax[4].hist(y_dev, color='limegreen', bins=bins);\nax[4].title.set_text('True distribution')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:05:03.860922Z","iopub.execute_input":"2022-07-05T22:05:03.861323Z","iopub.status.idle":"2022-07-05T22:05:04.606024Z","shell.execute_reply.started":"2022-07-05T22:05:03.861293Z","shell.execute_reply":"2022-07-05T22:05:04.605009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f, ax = plt.subplots(1, 4, figsize=(18, 4))\nax[0].hist(nb_cls1_dev_preds[round(lvl2_X_dev['nb_cls1']).to_numpy() == y_dev], color='limegreen', bins=bins)\nax[0].hist(nb_cls1_dev_preds[round(lvl2_X_dev['nb_cls1']).to_numpy() != y_dev], alpha=0.8, bins=bins)\nax[0].legend(['true', 'false'])\nax[0].title.set_text('NB on keyword, location, language etc')\n\nax[1].hist(nb_cls2_dev_preds[round(lvl2_X_dev['nb_cls2']).to_numpy() == y_dev], color='limegreen', bins=bins)\nax[1].hist(nb_cls2_dev_preds[round(lvl2_X_dev['nb_cls2']).to_numpy() != y_dev], alpha=0.8, bins=bins)\nax[1].title.set_text('Multinomial NB for text')\n\nax[2].hist(lr_cls1_dev_preds[round(lvl2_X_dev['lr_cls1']).to_numpy() == y_dev], color='limegreen', bins=bins)\nax[2].hist(lr_cls1_dev_preds[round(lvl2_X_dev['lr_cls1']).to_numpy() != y_dev], alpha=0.8, bins=bins)\nax[2].title.set_text('TFIDF with logreg')\n\nax[3].hist(kn_cls1_dev_preds[round(lvl2_X_dev['kn_cls1']).to_numpy() == y_dev], color='limegreen', bins=bins)\nax[3].hist(kn_cls1_dev_preds[round(lvl2_X_dev['kn_cls1']).to_numpy() != y_dev], alpha=0.8, bins=bins)\nax[3].title.set_text('TFIDF with KNN')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:05:06.022924Z","iopub.execute_input":"2022-07-05T22:05:06.023340Z","iopub.status.idle":"2022-07-05T22:05:06.827257Z","shell.execute_reply.started":"2022-07-05T22:05:06.023305Z","shell.execute_reply":"2022-07-05T22:05:06.826056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### Evaluate different voting stategies\nStrategies:\n* Equal weights\n* Weights proportional to accuracy\n* Weights inverse proportional to MSE\n* Custom Weights","metadata":{}},{"cell_type":"code","source":"def temp_softmax(arr, T=1):\n    arr = np.array(arr)\n    arr = arr / T\n    arr = np.e**arr\n    arr = arr / sum(arr)\n    return arr\n\nsoftmax_temperature = 0.1\nbins=40\nf, ax = plt.subplots(1, 4, figsize=(24, 6))\nmodels_dev_preds = lvl2_X_dev\ny_dev = y_dev.reset_index(drop=True)\n\n# WEIGHTS\n# equal_weights\nequal_weights = [0.33, 0.33, 0.33, 0.33]\n\n# weights proportional to accuracy of base models\naccuracy_vec = [accuracy_score(round(models_dev_preds[col_name]), y_dev) \n                for col_name in models_dev_preds.columns]\nacc_weights = temp_softmax(accuracy_vec, softmax_temperature)\n\n# weights inverse proportional to mse of base models\nmse_vec = [mse(models_dev_preds[col_name], y_dev) \n           for col_name in models_dev_preds.columns]\nmse_diff_vec = [max(mse_vec) - elem for elem in mse_vec]\nmse_weights = temp_softmax(mse_diff_vec, softmax_temperature)\n\n\n# soft voting with equal weights\nvoting1 = models_dev_preds['nb_cls1']*equal_weights[0] + models_dev_preds['nb_cls2']*equal_weights[1] \\\n          + models_dev_preds['lr_cls1']*equal_weights[2] + + models_dev_preds['kn_cls1']*equal_weights[3]\n# soft voting with accuracy weights\nvoting2 = models_dev_preds['nb_cls1']*acc_weights[0] + models_dev_preds['nb_cls2']*acc_weights[1] \\\n          + models_dev_preds['lr_cls1']*acc_weights[2] + + models_dev_preds['kn_cls1']*acc_weights[3]\n# soft voting with mse weights\nvoting3 = models_dev_preds['nb_cls1']*mse_weights[0] + models_dev_preds['nb_cls2']*mse_weights[1] \\\n          + models_dev_preds['lr_cls1']*mse_weights[2] + models_dev_preds['kn_cls1']*mse_weights[2]\n# soft voting with custom weights\nvoting4 = models_dev_preds['nb_cls1']*custom_weights[0] + models_dev_preds['nb_cls2']*custom_weights[1] \\\n          + models_dev_preds['lr_cls1']*custom_weights[2] + models_dev_preds['kn_cls1']*custom_weights[3]\n\n\nax[0].hist(voting1[round(voting1).to_numpy() == y_dev], color='limegreen', bins=bins)\nax[0].hist(voting1[round(voting1).to_numpy() != y_dev], alpha=0.8, bins=bins)\nax[0].legend(['true', 'false'])\nax[0].title.set_text(f\"Equal weights ({[str(round(x,2)) for x in equal_weights]}) score: {round(accuracy_score(round(voting1), y_dev),2)}\")\n\nax[1].hist(voting2[round(voting2).to_numpy() == y_dev], color='limegreen', bins=bins)\nax[1].hist(voting2[round(voting2).to_numpy() != y_dev], alpha=0.8, bins=bins)\nax[1].legend(['true', 'false'])\nax[1].title.set_text(f'Accuracy weights ({[str(round(x,2)) for x in acc_weights]}) score: {round(accuracy_score(round(voting2), y_dev),2)}')\n\nax[2].hist(voting3[round(voting3).to_numpy() == y_dev], color='limegreen', bins=bins)\nax[2].hist(voting3[round(voting3).to_numpy() != y_dev], alpha=0.8, bins=bins)\nax[2].legend(['true', 'false'])\nax[2].title.set_text(f'MSE weights ({[str(round(x,2)) for x in mse_weights]}) score: {round(accuracy_score(round(voting3), y_dev),2)}')\n\nax[3].hist(voting4[round(voting4).to_numpy() == y_dev], color='limegreen', bins=bins)\nax[3].hist(voting4[round(voting4).to_numpy() != y_dev], alpha=0.8, bins=bins)\nax[3].legend(['true', 'false'])\nax[3].title.set_text(f'Custom weights ({[str(round(x,2)) for x in custom_weights]}) score: {round(accuracy_score(round(voting4), y_dev),2)}')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:05:09.127169Z","iopub.execute_input":"2022-07-05T22:05:09.127566Z","iopub.status.idle":"2022-07-05T22:05:10.557195Z","shell.execute_reply.started":"2022-07-05T22:05:09.127536Z","shell.execute_reply":"2022-07-05T22:05:10.556149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Evaluate different ensembling approaches","metadata":{}},{"cell_type":"code","source":"try:\n    first_approach_preds = first_approach_preds[:,1]\n    second_approach_preds = second_approach_preds[:,1]\n    third_approach_preds = third_approach_preds\n    third_approach_target = third_approach_target.to_numpy()\nexcept:\n    print('This cell was already executed')","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:05:10.765059Z","iopub.execute_input":"2022-07-05T22:05:10.765444Z","iopub.status.idle":"2022-07-05T22:05:10.771778Z","shell.execute_reply.started":"2022-07-05T22:05:10.765413Z","shell.execute_reply":"2022-07-05T22:05:10.770637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Stacking histograms\n\nf, ax = plt.subplots(1, 3, figsize=(24, 6))\nbins=60\n\nax[0].hist(x = [first_approach_preds[np.round_(first_approach_preds) != first_approach_target], \n                first_approach_preds[np.round_(first_approach_preds) == first_approach_target]], color=['tab:blue','limegreen'], \n           stacked=True, bins=bins)\nax[0].legend(['true', 'false'])\nax[0].title.set_text(f\"First approach (score: {round(accuracy_score(np.round_(first_approach_preds), first_approach_target), 3)})\")\n\nax[1].hist(x = [second_approach_preds[np.round_(second_approach_preds) != second_approach_target], \n                second_approach_preds[np.round_(second_approach_preds) == second_approach_target]], color=['tab:blue','limegreen'], \n           stacked=True, bins=bins)\nax[1].legend(['true', 'false'])\nax[1].title.set_text(f\"Second approach (score: {round(accuracy_score(np.round_(second_approach_preds), second_approach_target), 3)})\")\n\nax[2].hist(x = [third_approach_preds[np.round_(third_approach_preds) != third_approach_target], \n                third_approach_preds[np.round_(third_approach_preds) == third_approach_target]], color=['tab:blue','limegreen'], \n           stacked=True, bins=bins)\nax[2].legend(['true', 'false'])\nax[2].title.set_text(f\"Third approach (score: {round(accuracy_score(np.round_(third_approach_preds), third_approach_target), 3)})\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:05:11.197872Z","iopub.execute_input":"2022-07-05T22:05:11.198282Z","iopub.status.idle":"2022-07-05T22:05:12.433077Z","shell.execute_reply.started":"2022-07-05T22:05:11.198248Z","shell.execute_reply":"2022-07-05T22:05:12.431973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TODO: Evaluate 1st + 3rd approach soft voting solution (robustness(3rd) + fine-tuning(1st))\n#       They have different train/test splits and complicated training process,\n#       so evaluating it is not feasible","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:05:12.571373Z","iopub.execute_input":"2022-07-05T22:05:12.572143Z","iopub.status.idle":"2022-07-05T22:05:12.576706Z","shell.execute_reply.started":"2022-07-05T22:05:12.572106Z","shell.execute_reply":"2022-07-05T22:05:12.575842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Predictions","metadata":{}},{"cell_type":"code","source":"### Soft Voting\n# fit base models on whole train\nX_all = pd.concat([X, X_test])\ncount_x, tfidf_x, feat_x = get_feature_vectors(X_all, feat_features)\n\nnb_cls1.fit(feat_x[:-len(X_test)], y)\nnb_cls2.fit(count_x[:-len(X_test)], y)\nlr_cls1.fit(tfidf_x[:-len(X_test)], y)\nkn_cls1.fit(tfidf_x[:-len(X_test)], y)\n\n\n# test predictions\nmodels_preds = pd.DataFrame({'nb_cls1': nb_cls1.predict_proba(feat_x[-len(X_test):])[:,1],\n                             'nb_cls2': nb_cls2.predict_proba(count_x[-len(X_test):])[:,1],\n                             'lr_cls1': lr_cls1.predict_proba(tfidf_x[-len(X_test):])[:,1],\n                             'kn_cls1': kn_cls1.predict_proba(tfidf_x[-len(X_test):])[:,1]})","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:05:14.511031Z","iopub.execute_input":"2022-07-05T22:05:14.511443Z","iopub.status.idle":"2022-07-05T22:05:16.950618Z","shell.execute_reply.started":"2022-07-05T22:05:14.511412Z","shell.execute_reply":"2022-07-05T22:05:16.949623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Base models","metadata":{}},{"cell_type":"code","source":"f, ax = plt.subplots(1, 4, figsize=(18, 6))\nbins=40\nax[0].hist(models_preds['nb_cls1'], bins=bins, color='lightseagreen', alpha=0.65)\nax[0].title.set_text('nb_cls1')\nax[1].hist(models_preds['nb_cls2'], bins=bins, color='lightseagreen', alpha=0.65)\nax[1].title.set_text('nb_cls2')\nax[2].hist(models_preds['lr_cls1'], bins=bins, color='lightseagreen', alpha=0.65)\nax[2].title.set_text('lr_cls1')\nax[3].hist(models_preds['kn_cls1'], bins=bins, color='lightseagreen', alpha=0.65)\nax[3].title.set_text('kn_cls1');","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:05:16.959915Z","iopub.execute_input":"2022-07-05T22:05:16.960299Z","iopub.status.idle":"2022-07-05T22:05:17.725912Z","shell.execute_reply.started":"2022-07-05T22:05:16.960265Z","shell.execute_reply":"2022-07-05T22:05:17.724965Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Ensembles","metadata":{}},{"cell_type":"code","source":"weights = custom_weights\nvoting_preds_proba = models_preds['nb_cls1']*weights[0] \\\n                     + models_preds['nb_cls2']*weights[1] \\\n                     + models_preds['lr_cls1']*weights[2] \\\n                     + models_preds['kn_cls1']*weights[3]\n\nlogreg_preds_proba = lvl2_lr_cls1.predict_proba(models_preds)[:,1]\nlvl2_voting_preds_proba = 0.2*logreg_preds_proba + 0.8*voting_preds_proba","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:11:00.986394Z","iopub.execute_input":"2022-07-05T22:11:00.986871Z","iopub.status.idle":"2022-07-05T22:11:01.005819Z","shell.execute_reply.started":"2022-07-05T22:11:00.986824Z","shell.execute_reply":"2022-07-05T22:11:01.004317Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bins = 60\nf, ax = plt.subplots(1, 3, figsize=(18, 6))\nax[0].hist(logreg_preds_proba, bins=bins, color='lightseagreen')\nax[0].title.set_text('Test preds with logreg (1st approach)')\n\nax[1].hist(lvl2_voting_preds_proba, bins=bins, color='seagreen')\nax[1].title.set_text('Mean between logreg and voting ensembles');\n\nax[2].hist(voting_preds_proba, bins=bins, color='lightseagreen')\nax[2].title.set_text('Test preds with voting (3rd approach)')\n","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:11:03.330737Z","iopub.execute_input":"2022-07-05T22:11:03.331171Z","iopub.status.idle":"2022-07-05T22:11:04.041214Z","shell.execute_reply.started":"2022-07-05T22:11:03.331137Z","shell.execute_reply":"2022-07-05T22:11:04.039906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = round(lvl2_voting_preds_proba)\n\noutput = pd.DataFrame({'id': test_data['id'], 'target': preds.astype(int)})\noutput.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-05T22:11:08.426419Z","iopub.execute_input":"2022-07-05T22:11:08.426800Z","iopub.status.idle":"2022-07-05T22:11:08.441546Z","shell.execute_reply.started":"2022-07-05T22:11:08.426769Z","shell.execute_reply":"2022-07-05T22:11:08.440535Z"},"trusted":true},"execution_count":null,"outputs":[]}]}