{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from typing import Tuple\n\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\n\nfrom geopy.geocoders import Nominatim\nimport folium\nimport warnings\nimport re\nimport emoji\nfrom tqdm import tqdm\n\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nfrom sklearn.feature_extraction.text import CountVectorizer, TfidfVectorizer\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.decomposition import PCA, TruncatedSVD\n\n\nplt.style.use('ggplot')\n\nTRAIN_PATH = '../input/nlp-getting-started/train.csv'\nTEST_PATH = '../input/nlp-getting-started/test.csv'","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:12:34.333701Z","iopub.execute_input":"2022-08-05T14:12:34.334453Z","iopub.status.idle":"2022-08-05T14:12:34.342289Z","shell.execute_reply.started":"2022-08-05T14:12:34.334416Z","shell.execute_reply":"2022-08-05T14:12:34.341293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"warnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T13:59:59.015561Z","iopub.execute_input":"2022-08-05T13:59:59.016450Z","iopub.status.idle":"2022-08-05T13:59:59.022963Z","shell.execute_reply.started":"2022-08-05T13:59:59.016415Z","shell.execute_reply":"2022-08-05T13:59:59.021830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(TRAIN_PATH)\ndf.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T13:59:59.025120Z","iopub.execute_input":"2022-08-05T13:59:59.025915Z","iopub.status.idle":"2022-08-05T13:59:59.108104Z","shell.execute_reply.started":"2022-08-05T13:59:59.025871Z","shell.execute_reply":"2022-08-05T13:59:59.106980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T13:59:59.110323Z","iopub.execute_input":"2022-08-05T13:59:59.110649Z","iopub.status.idle":"2022-08-05T13:59:59.133530Z","shell.execute_reply.started":"2022-08-05T13:59:59.110620Z","shell.execute_reply":"2022-08-05T13:59:59.132759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T13:59:59.134796Z","iopub.execute_input":"2022-08-05T13:59:59.135255Z","iopub.status.idle":"2022-08-05T13:59:59.156391Z","shell.execute_reply.started":"2022-08-05T13:59:59.135209Z","shell.execute_reply":"2022-08-05T13:59:59.155675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_style('darkgrid')\nsns.countplot(x='target', data=df, palette='icefire')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T13:59:59.157589Z","iopub.execute_input":"2022-08-05T13:59:59.158105Z","iopub.status.idle":"2022-08-05T13:59:59.367695Z","shell.execute_reply.started":"2022-08-05T13:59:59.158076Z","shell.execute_reply":"2022-08-05T13:59:59.366498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_locations_cat = df.groupby(['location', 'target'])[['id']].count().sort_values(by='id', ascending=False)\ndf_locations_cat","metadata":{"execution":{"iopub.status.busy":"2022-08-05T13:59:59.369247Z","iopub.execute_input":"2022-08-05T13:59:59.370300Z","iopub.status.idle":"2022-08-05T13:59:59.399880Z","shell.execute_reply.started":"2022-08-05T13:59:59.370267Z","shell.execute_reply":"2022-08-05T13:59:59.398594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_locations_total = df_locations_cat.groupby('location').sum().sort_values(by='id', ascending=False)\ndf_locations_total","metadata":{"execution":{"iopub.status.busy":"2022-08-05T13:59:59.401315Z","iopub.execute_input":"2022-08-05T13:59:59.401600Z","iopub.status.idle":"2022-08-05T13:59:59.414925Z","shell.execute_reply.started":"2022-08-05T13:59:59.401577Z","shell.execute_reply":"2022-08-05T13:59:59.414059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from functools import lru_cache\n\n@lru_cache(maxsize=100)\ndef find_long_lat(name: str) -> Tuple[int, int]:\n    geolocator = Nominatim(user_agent='user_agent')\n    location = geolocator.geocode(name)\n    if location:\n        return location.longitude, location.latitude\n    return 0, 0","metadata":{"execution":{"iopub.status.busy":"2022-08-05T13:59:59.418200Z","iopub.execute_input":"2022-08-05T13:59:59.418944Z","iopub.status.idle":"2022-08-05T13:59:59.425867Z","shell.execute_reply.started":"2022-08-05T13:59:59.418914Z","shell.execute_reply":"2022-08-05T13:59:59.425084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(find_long_lat('Kolkata'))\n%time","metadata":{"execution":{"iopub.status.busy":"2022-08-05T13:59:59.429272Z","iopub.execute_input":"2022-08-05T13:59:59.429600Z","iopub.status.idle":"2022-08-05T14:00:00.423777Z","shell.execute_reply.started":"2022-08-05T13:59:59.429571Z","shell.execute_reply":"2022-08-05T14:00:00.422454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_locations_total = df_locations_total.iloc[:100]\ndf_locations_total","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:00:00.425633Z","iopub.execute_input":"2022-08-05T14:00:00.426371Z","iopub.status.idle":"2022-08-05T14:00:00.438975Z","shell.execute_reply.started":"2022-08-05T14:00:00.426329Z","shell.execute_reply":"2022-08-05T14:00:00.438021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_locations_total['long'] = df_locations_total \\\n                            .index              \\\n                            .to_series()        \\\n                            .apply(lambda x: find_long_lat(x)[0])\ndf_locations_total['lat'] = df_locations_total  \\\n                            .index              \\\n                            .to_series()        \\\n                            .apply(lambda x: find_long_lat(x)[1])\n\ndf_locations_total","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:00:00.439920Z","iopub.execute_input":"2022-08-05T14:00:00.440644Z","iopub.status.idle":"2022-08-05T14:01:09.134617Z","shell.execute_reply.started":"2022-08-05T14:00:00.440615Z","shell.execute_reply":"2022-08-05T14:01:09.133312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"map_ = folium.Map(location=[20,0], tiles=\"OpenStreetMap\", zoom_start=2)\n# add marker one by one on the map\nfor i in range(0,len(df_locations_total)):\n    folium.Circle(\n      location=[df_locations_total.iloc[i]['lat'], df_locations_total.iloc[i]['long']],\n      popup=f\"{df_locations_total.index[i]}:{df_locations_total.iloc[i]['id']}\",\n      radius=float(df_locations_total.iloc[i]['id'])*20000,\n      color='crimson',\n      fill=True,\n      fill_color='crimson'\n   ).add_to(map_)\n\n\nmap_","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:09.136269Z","iopub.execute_input":"2022-08-05T14:01:09.136676Z","iopub.status.idle":"2022-08-05T14:01:09.366893Z","shell.execute_reply.started":"2022-08-05T14:01:09.136639Z","shell.execute_reply":"2022-08-05T14:01:09.365812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_loc_cat_new = df_locations_cat.reset_index()\ndf_loc_cat_new['target_str'] = df_loc_cat_new['target'].apply(lambda x: 'Disaster' if x == 1 else 'Not disaster')\nfig = px.bar(df_loc_cat_new[:50], x='location', y='id', color='target_str')\nfig.update_layout(\n    title='class count per location (Top 50)',\n    xaxis_title='Location', \n    yaxis_title = 'Counts', \n)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:09.368675Z","iopub.execute_input":"2022-08-05T14:01:09.369428Z","iopub.status.idle":"2022-08-05T14:01:10.440643Z","shell.execute_reply.started":"2022-08-05T14:01:09.369391Z","shell.execute_reply":"2022-08-05T14:01:10.439607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Tweet length","metadata":{}},{"cell_type":"code","source":"df['tweet_length'] = df['text'].apply(lambda x: len(x))\nplt.figure(figsize=(20, 5))\nbins = 150\nplt.hist(df[df['target'] == 0]['tweet_length'], alpha=0.6, bins=bins, label='Not disaster', color='lightcoral')\nplt.hist(df[df['target'] == 1]['tweet_length'], alpha=0.8, bins=bins, label='Disaster', color='cadetblue')\nplt.xlabel('Tweet length')\nplt.ylabel('# tweets')\nplt.legend(loc='upper right')\nplt.xlim(0,150)\n\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:10.442015Z","iopub.execute_input":"2022-08-05T14:01:10.442381Z","iopub.status.idle":"2022-08-05T14:01:11.186331Z","shell.execute_reply.started":"2022-08-05T14:01:10.442352Z","shell.execute_reply":"2022-08-05T14:01:11.185168Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test = pd.read_csv(TEST_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:11.187941Z","iopub.execute_input":"2022-08-05T14:01:11.189030Z","iopub.status.idle":"2022-08-05T14:01:11.218303Z","shell.execute_reply.started":"2022-08-05T14:01:11.188987Z","shell.execute_reply":"2022-08-05T14:01:11.217497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['word_count'] = df['text'].apply(lambda x: len(x.split()))\ndf_test['word_count'] = df_test['text'].apply(lambda x: len(x.split()))\n\ndf['unique_word_count'] = df['text'].apply(lambda x: len(set(x.split())))\ndf_test['unique_word_count'] = df_test['text'].apply(lambda x: len(set(x.split())))\n\ndf['mean_word_length'] = df['text'].apply(lambda x: np.mean([len(w) for w in x.split()]))\ndf_test['mean_word_length'] = df_test['text'].apply(lambda x: np.mean([len(w) for w in x.split()]))\n\ndf['hashtag_count'] = df['text'].apply(lambda x: len([c for c in x if c == '#']))\ndf_test['hashtag_count'] = df_test['text'].apply(lambda x: len([c for c in x if c == '#']))\n\nFEATURES = ['word_count', 'unique_word_count', 'mean_word_length', 'hashtag_count']\n\nfig, axes = plt.subplots(ncols=2, nrows=2, figsize=(20, 10), dpi=100)\nfor i, feature in enumerate(FEATURES):\n    sns.distplot(df[df['target'] == 0][feature], label='Not Disaster', ax=axes[i // 2][i % 2], color='green')\n    sns.distplot(df[df['target'] == 1][feature], label='Disaster', ax=axes[i // 2][i % 2], color='red')\n    axes[i // 2][i % 2].set_xlabel('')\n    axes[i // 2][i % 2].tick_params(axis='x', labelsize=12)\n    axes[i // 2][i % 2].tick_params(axis='y', labelsize=12)\n    axes[i // 2][i % 2].legend()\n    axes[i // 2][i % 2].set_title(f'{feature} Distribution in Train Set - Disaster / Non-disaster', fontsize=13)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:11.219831Z","iopub.execute_input":"2022-08-05T14:01:11.220490Z","iopub.status.idle":"2022-08-05T14:01:13.282159Z","shell.execute_reply.started":"2022-08-05T14:01:11.220450Z","shell.execute_reply":"2022-08-05T14:01:13.280962Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axes = plt.subplots(ncols=2, nrows=2, figsize=(20, 10), dpi=100)\nfor i, feature in enumerate(FEATURES):\n    sns.distplot(df[feature], label='Train', ax=axes[i // 2][i % 2])\n    sns.distplot(df_test[feature], label='Test', ax=axes[i // 2][i % 2])\n\n    axes[i // 2][i % 2].set_xlabel('')\n    axes[i // 2][i % 2].tick_params(axis='x', labelsize=12)\n    axes[i // 2][i % 2].tick_params(axis='y', labelsize=12)\n    axes[i // 2][i % 2].legend()\n    axes[i // 2][i % 2].set_title(f'{feature} Distribution in Train / Test Set', fontsize=13)\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:13.283679Z","iopub.execute_input":"2022-08-05T14:01:13.284094Z","iopub.status.idle":"2022-08-05T14:01:15.195999Z","shell.execute_reply.started":"2022-08-05T14:01:13.284058Z","shell.execute_reply":"2022-08-05T14:01:15.195227Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(35, 5))\nkw_counts = sns.countplot(x='keyword', data=df, order=df['keyword'].value_counts()[:50].index, palette='flare')\nkw_counts.set_xticklabels(kw_counts.get_xticklabels(), rotation=45, horizontalalignment='right')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:15.197249Z","iopub.execute_input":"2022-08-05T14:01:15.198174Z","iopub.status.idle":"2022-08-05T14:01:16.567728Z","shell.execute_reply.started":"2022-08-05T14:01:15.198143Z","shell.execute_reply":"2022-08-05T14:01:16.566771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Cleaning","metadata":{}},{"cell_type":"code","source":"df['keyword'].notnull().sum()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:16.569347Z","iopub.execute_input":"2022-08-05T14:01:16.569695Z","iopub.status.idle":"2022-08-05T14:01:16.577943Z","shell.execute_reply.started":"2022-08-05T14:01:16.569666Z","shell.execute_reply":"2022-08-05T14:01:16.576794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['keyword'].unique()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:16.579411Z","iopub.execute_input":"2022-08-05T14:01:16.580024Z","iopub.status.idle":"2022-08-05T14:01:16.590394Z","shell.execute_reply.started":"2022-08-05T14:01:16.579965Z","shell.execute_reply":"2022-08-05T14:01:16.589270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Hashtag processing and wordcloud","metadata":{}},{"cell_type":"code","source":"r = re.compile(r'#(\\w+)')\n# r.findall('lorem #ipsum #dolor #sit #amet #089 #007 hello amazing#help')\n\ndf['hashtags'] = df['text'].apply(lambda x: ' '.join(r.findall(x)))\ndf['hashtags'].iloc[:10]","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:16.591853Z","iopub.execute_input":"2022-08-05T14:01:16.592166Z","iopub.status.idle":"2022-08-05T14:01:16.608873Z","shell.execute_reply.started":"2022-08-05T14:01:16.592138Z","shell.execute_reply":"2022-08-05T14:01:16.607775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['n_hashtags'] = df['hashtags'].apply(lambda x: len(x.split()))\nfig = sns.countplot(x='n_hashtags', data=df)\nfig.set_title(\"Number of Hashtags\")","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:16.610605Z","iopub.execute_input":"2022-08-05T14:01:16.610959Z","iopub.status.idle":"2022-08-05T14:01:16.889207Z","shell.execute_reply.started":"2022-08-05T14:01:16.610932Z","shell.execute_reply":"2022-08-05T14:01:16.888099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from wordcloud import WordCloud\n\nplt.figure(figsize=(20, 10), dpi=100)\nhashtag_corpus = []\nfor arr in df['hashtags']:\n    hashtag_corpus.append(arr)\ncorpus = ' '.join(hashtag_corpus)\nwc_hashtags = WordCloud(background_color='black', width=800, height=800).generate(corpus)\nplt.imshow(wc_hashtags)\nplt.axis('off')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:16.890299Z","iopub.execute_input":"2022-08-05T14:01:16.890577Z","iopub.status.idle":"2022-08-05T14:01:18.552257Z","shell.execute_reply.started":"2022-08-05T14:01:16.890552Z","shell.execute_reply":"2022-08-05T14:01:18.551327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_dups = len(df[df.duplicated(subset=['text'])])\nlen_before = len(df)\n\ndf.drop_duplicates(subset=['text'], inplace=True)\nprint(f'{len_before} - {len(df)} = {n_dups} duplicate tweets have been removed')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:18.553571Z","iopub.execute_input":"2022-08-05T14:01:18.554359Z","iopub.status.idle":"2022-08-05T14:01:18.572852Z","shell.execute_reply.started":"2022-08-05T14:01:18.554326Z","shell.execute_reply":"2022-08-05T14:01:18.571777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Cleaning\n\nWe have to do some basic preprocessing to remove unnecessary items such as emojis, and urls from the tweets. Additionally, hashtags would need to have the pound sign removed from them as they do not add any additional information (other than identifying hashtags, which we extracted already)\n\nFor this purpose we are doing all our cleaning on a copy of our dataframe. After performing sufficient analysis and visualization on the cleaned data and additionally extracted data we will build a pipeline for processing the primary dataframe.\n","metadata":{"execution":{"iopub.status.busy":"2022-08-04T15:05:43.631149Z","iopub.execute_input":"2022-08-04T15:05:43.632066Z","iopub.status.idle":"2022-08-04T15:05:43.657803Z","shell.execute_reply.started":"2022-08-04T15:05:43.632023Z","shell.execute_reply":"2022-08-04T15:05:43.656983Z"}}},{"cell_type":"code","source":"df_copy = df.copy()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:18.574252Z","iopub.execute_input":"2022-08-05T14:01:18.574779Z","iopub.status.idle":"2022-08-05T14:01:18.579305Z","shell.execute_reply.started":"2022-08-05T14:01:18.574722Z","shell.execute_reply":"2022-08-05T14:01:18.578556Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Removing Emojis","metadata":{}},{"cell_type":"code","source":"# Reference : https://gist.github.com/slowkow/7a7f61f495e3dbb7e3d767f97bd7304b\ndef remove_emoji(text: str):\n    emoji_pattern = re.compile(\"[\"\n                           u\"\\U0001F600-\\U0001F64F\"  # emoticons\n                           u\"\\U0001F300-\\U0001F5FF\"  # symbols & pictographs\n                           u\"\\U0001F680-\\U0001F6FF\"  # transport & map symbols\n                           u\"\\U0001F1E0-\\U0001F1FF\"  # flags (iOS)\n                           u\"\\U00002702-\\U000027B0\"\n                           u\"\\U000024C2-\\U0001F251\"\n                           \"]+\", flags=re.UNICODE)\n    return emoji_pattern.sub(r'', text)\n\nremove_emoji(\"Omg another Earthquake 😔😔\")","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:18.580685Z","iopub.execute_input":"2022-08-05T14:01:18.581238Z","iopub.status.idle":"2022-08-05T14:01:18.596139Z","shell.execute_reply.started":"2022-08-05T14:01:18.581209Z","shell.execute_reply":"2022-08-05T14:01:18.594948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_copy['text_cleaned'] = df_copy['text'].apply(remove_emoji)\ndf_copy['text_cleaned'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:18.600653Z","iopub.execute_input":"2022-08-05T14:01:18.601325Z","iopub.status.idle":"2022-08-05T14:01:18.649303Z","shell.execute_reply.started":"2022-08-05T14:01:18.601292Z","shell.execute_reply":"2022-08-05T14:01:18.648249Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Removing URLS","metadata":{}},{"cell_type":"code","source":"test_str = 'hello world https://stackoverflow.com/questions/6038061/regular-expression-to-find-urls-within-a-string'\n\ndef remove_urls(text: str):\n    r = re.compile('(?:(?:https?|ftp):\\/\\/)?[\\w/\\-?=%.]+\\.[\\w/\\-&?=%.]+')\n    return r.sub(r'', text)\n\nremove_urls(test_str)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:18.650410Z","iopub.execute_input":"2022-08-05T14:01:18.651161Z","iopub.status.idle":"2022-08-05T14:01:18.658220Z","shell.execute_reply.started":"2022-08-05T14:01:18.651132Z","shell.execute_reply":"2022-08-05T14:01:18.657449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_copy['text_cleaned'] = df_copy['text_cleaned'].apply(remove_urls)\ndf_copy['text_cleaned'].describe()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:18.659203Z","iopub.execute_input":"2022-08-05T14:01:18.660050Z","iopub.status.idle":"2022-08-05T14:01:18.767095Z","shell.execute_reply.started":"2022-08-05T14:01:18.660020Z","shell.execute_reply":"2022-08-05T14:01:18.765966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_copy.drop_duplicates(subset=['text_cleaned'], keep='first', inplace=True)\ndf_copy.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:18.768563Z","iopub.execute_input":"2022-08-05T14:01:18.769733Z","iopub.status.idle":"2022-08-05T14:01:18.790317Z","shell.execute_reply.started":"2022-08-05T14:01:18.769658Z","shell.execute_reply":"2022-08-05T14:01:18.789207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Remove Mentions and Hashtags\nWe already captured hashtags in a seperate column. Mentions begin with `@`, we can similarly collect the mentions in each tweet and remove them subsequently.","metadata":{}},{"cell_type":"code","source":"r = re.compile(r'@(\\w+)')\ndf_copy['mentions'] = df_copy['text_cleaned'].apply(lambda x: ' '.join(r.findall(x)))\ndf_copy['mentions'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:18.792033Z","iopub.execute_input":"2022-08-05T14:01:18.792483Z","iopub.status.idle":"2022-08-05T14:01:18.808889Z","shell.execute_reply.started":"2022-08-05T14:01:18.792446Z","shell.execute_reply":"2022-08-05T14:01:18.807778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Remove Punctuations","metadata":{}},{"cell_type":"code","source":"import string\ndef remove_punct(text: str):\n    return text.translate(str.maketrans('', '', string.punctuation))\n\nremove_punct('hello everyone #hashtag')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:18.810803Z","iopub.execute_input":"2022-08-05T14:01:18.811520Z","iopub.status.idle":"2022-08-05T14:01:18.818571Z","shell.execute_reply.started":"2022-08-05T14:01:18.811479Z","shell.execute_reply":"2022-08-05T14:01:18.817638Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_copy['text_cleaned'] = df_copy['text_cleaned'].apply(remove_punct)\ndf_copy.info()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:01:18.819896Z","iopub.execute_input":"2022-08-05T14:01:18.820508Z","iopub.status.idle":"2022-08-05T14:01:18.877179Z","shell.execute_reply.started":"2022-08-05T14:01:18.820470Z","shell.execute_reply":"2022-08-05T14:01:18.876038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n\ndef load_glove():\n    pkl_path = '../input/glove-twitter-pickles-27b-25d-50d-100d-200d/glove.twitter.27B.100d.pkl'\n    with open(pkl_path, 'rb') as f:\n        glove = pickle.load(f)\n    return glove\n\nglove = load_glove()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:06:31.276829Z","iopub.execute_input":"2022-08-05T14:06:31.277237Z","iopub.status.idle":"2022-08-05T14:06:34.403003Z","shell.execute_reply.started":"2022-08-05T14:06:31.277209Z","shell.execute_reply":"2022-08-05T14:06:34.401795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# turn to lowercase\ndf_copy['text_cleaned'] = df_copy['text_cleaned'].apply(lambda x: x.casefold())\ndf_copy['text_cleaned'].head()","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:11:55.063687Z","iopub.execute_input":"2022-08-05T14:11:55.064121Z","iopub.status.idle":"2022-08-05T14:11:55.077310Z","shell.execute_reply.started":"2022-08-05T14:11:55.064090Z","shell.execute_reply":"2022-08-05T14:11:55.076323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_corpus(df: pd.DataFrame, col: str):\n    corpus_ = {}\n    for text in df[col]:\n        for word in word_tokenize(text):\n            if corpus_.get(word, None) is None:\n                corpus_[word] = 1\n            else:\n                corpus_[word] += 1\n    \n    return corpus_","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:15:21.589829Z","iopub.execute_input":"2022-08-05T14:15:21.590659Z","iopub.status.idle":"2022-08-05T14:15:21.595829Z","shell.execute_reply.started":"2022-08-05T14:15:21.590627Z","shell.execute_reply":"2022-08-05T14:15:21.595143Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"corpus = create_corpus(df=df_copy, col='text_cleaned')\n'usa' in corpus","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:15:51.427798Z","iopub.execute_input":"2022-08-05T14:15:51.428179Z","iopub.status.idle":"2022-08-05T14:15:52.305788Z","shell.execute_reply.started":"2022-08-05T14:15:51.428148Z","shell.execute_reply":"2022-08-05T14:15:52.304604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics.pairwise import cosine_similarity\n\ndef similarity(word1: str, word2: str):\n    vec1, vec2 = glove.get(word1, None), glove.get(word2, None)\n    if vec1 is None or vec2 is None:\n        return np.zeros(1)\n    vec1 = vec1.reshape(1, -1)\n    vec2 = vec2.reshape(1, -1)\n    return cosine_similarity(vec1, vec2)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:22:19.508880Z","iopub.execute_input":"2022-08-05T14:22:19.509321Z","iopub.status.idle":"2022-08-05T14:22:19.517143Z","shell.execute_reply.started":"2022-08-05T14:22:19.509284Z","shell.execute_reply":"2022-08-05T14:22:19.516014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"similarity('cow', 'moose')","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:24:47.011987Z","iopub.execute_input":"2022-08-05T14:24:47.012391Z","iopub.status.idle":"2022-08-05T14:24:47.020107Z","shell.execute_reply.started":"2022-08-05T14:24:47.012356Z","shell.execute_reply":"2022-08-05T14:24:47.019112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:25:00.938075Z","iopub.execute_input":"2022-08-05T14:25:00.938454Z","iopub.status.idle":"2022-08-05T14:25:00.946203Z","shell.execute_reply.started":"2022-08-05T14:25:00.938423Z","shell.execute_reply":"2022-08-05T14:25:00.945115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tweet = df_copy.iloc[42].text_cleaned\nkeyword = df_copy.iloc[42].keyword\n\ndef produce_sim_array(sentence: str, keyword: str):\n    vecs = []\n    vecs = [similarity(w, keyword) for w in word_tokenize(sentence)]\n    return np.vstack(vecs)\n\nproduce_sim_array(tweet, keyword)","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:27:47.654653Z","iopub.execute_input":"2022-08-05T14:27:47.655252Z","iopub.status.idle":"2022-08-05T14:27:47.678909Z","shell.execute_reply.started":"2022-08-05T14:27:47.655209Z","shell.execute_reply":"2022-08-05T14:27:47.677911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-08-05T14:27:17.408729Z","iopub.execute_input":"2022-08-05T14:27:17.409113Z","iopub.status.idle":"2022-08-05T14:27:17.415379Z","shell.execute_reply.started":"2022-08-05T14:27:17.409084Z","shell.execute_reply":"2022-08-05T14:27:17.414633Z"},"trusted":true},"execution_count":null,"outputs":[]}]}