{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport random\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom collections import Counter\nimport re\nimport string\n\nfrom wordcloud import WordCloud, STOPWORDS\nimport nltk\nfrom nltk.corpus import stopwords\nimport spacy\nfrom spacy.util import compounding\nfrom spacy.util import minibatch\nfrom tqdm import tqdm\nimport os\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-12T00:23:03.853437Z","iopub.execute_input":"2022-07-12T00:23:03.853786Z","iopub.status.idle":"2022-07-12T00:23:03.861491Z","shell.execute_reply.started":"2022-07-12T00:23:03.853758Z","shell.execute_reply":"2022-07-12T00:23:03.860234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:03.863913Z","iopub.execute_input":"2022-07-12T00:23:03.864909Z","iopub.status.idle":"2022-07-12T00:23:03.879873Z","shell.execute_reply.started":"2022-07-12T00:23:03.864870Z","shell.execute_reply":"2022-07-12T00:23:03.878881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/tweet-sentiment-extraction/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/tweet-sentiment-extraction/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:03.882557Z","iopub.execute_input":"2022-07-12T00:23:03.883234Z","iopub.status.idle":"2022-07-12T00:23:03.988164Z","shell.execute_reply.started":"2022-07-12T00:23:03.883162Z","shell.execute_reply":"2022-07-12T00:23:03.987198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:03.989385Z","iopub.execute_input":"2022-07-12T00:23:03.990266Z","iopub.status.idle":"2022-07-12T00:23:03.998396Z","shell.execute_reply.started":"2022-07-12T00:23:03.990216Z","shell.execute_reply":"2022-07-12T00:23:03.997242Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.info","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:04.001724Z","iopub.execute_input":"2022-07-12T00:23:04.003264Z","iopub.status.idle":"2022-07-12T00:23:04.020000Z","shell.execute_reply.started":"2022-07-12T00:23:04.003167Z","shell.execute_reply":"2022-07-12T00:23:04.018553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:04.021746Z","iopub.execute_input":"2022-07-12T00:23:04.024550Z","iopub.status.idle":"2022-07-12T00:23:04.039528Z","shell.execute_reply.started":"2022-07-12T00:23:04.024492Z","shell.execute_reply":"2022-07-12T00:23:04.038257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['sentiment'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:04.041027Z","iopub.execute_input":"2022-07-12T00:23:04.041587Z","iopub.status.idle":"2022-07-12T00:23:04.058905Z","shell.execute_reply.started":"2022-07-12T00:23:04.041539Z","shell.execute_reply":"2022-07-12T00:23:04.056278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['sentiment'].value_counts().plot.bar()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:04.060595Z","iopub.execute_input":"2022-07-12T00:23:04.061539Z","iopub.status.idle":"2022-07-12T00:23:04.230455Z","shell.execute_reply.started":"2022-07-12T00:23:04.061484Z","shell.execute_reply":"2022-07-12T00:23:04.229623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:04.231512Z","iopub.execute_input":"2022-07-12T00:23:04.232622Z","iopub.status.idle":"2022-07-12T00:23:04.255196Z","shell.execute_reply.started":"2022-07-12T00:23:04.232578Z","shell.execute_reply":"2022-07-12T00:23:04.253922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.dropna(inplace=True)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:04.256539Z","iopub.execute_input":"2022-07-12T00:23:04.257680Z","iopub.status.idle":"2022-07-12T00:23:04.280780Z","shell.execute_reply.started":"2022-07-12T00:23:04.257642Z","shell.execute_reply":"2022-07-12T00:23:04.279450Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.isna().sum()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:04.284548Z","iopub.execute_input":"2022-07-12T00:23:04.285244Z","iopub.status.idle":"2022-07-12T00:23:04.307355Z","shell.execute_reply.started":"2022-07-12T00:23:04.285206Z","shell.execute_reply":"2022-07-12T00:23:04.306481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['Num_of_words_text'] = df_train['text'].apply(lambda x : len(str(x).split()))\ndf_train['Num_of_words_ST'] = df_train['selected_text'].apply(lambda x : len(str(x).split()))\ndf_train['Difference'] = df_train['Num_of_words_text'] - df_train['Num_of_words_ST']\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:04.308485Z","iopub.execute_input":"2022-07-12T00:23:04.309100Z","iopub.status.idle":"2022-07-12T00:23:04.407491Z","shell.execute_reply.started":"2022-07-12T00:23:04.309065Z","shell.execute_reply":"2022-07-12T00:23:04.406371Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def jaccard_similarity(str1, str2):\n    a = set(str1.lower().split())\n    b = set(str2.lower().split())\n    c = a.intersection(b)\n    return float(len(c)/(len(a)+len(b)-len(c)))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:04.408828Z","iopub.execute_input":"2022-07-12T00:23:04.409162Z","iopub.status.idle":"2022-07-12T00:23:04.414895Z","shell.execute_reply.started":"2022-07-12T00:23:04.409132Z","shell.execute_reply":"2022-07-12T00:23:04.414161Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"jaccard_sim = []\nfor index, rows in df_train.iterrows():\n    st1 = rows.text\n    st2 = rows.selected_text\n    jaccard_sim.append([st1, st2, jaccard_similarity(st1,st2)])\n\ndf_jaccard = pd.DataFrame(jaccard_sim, columns = ['text','selected_text', 'jaccard_similarity'])\ndf_train = df_train.merge(df_jaccard, how='left')\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:04.415983Z","iopub.execute_input":"2022-07-12T00:23:04.416450Z","iopub.status.idle":"2022-07-12T00:23:06.511606Z","shell.execute_reply.started":"2022-07-12T00:23:04.416421Z","shell.execute_reply":"2022-07-12T00:23:06.510860Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16,6))\np1 = sns.kdeplot(df_train[df_train['sentiment']=='positive']['Difference'], shade=True, color='y').set_title(\"Kernel Distribution of Difference in Number of Words(Pos/Neg)\")\np2 = sns.kdeplot(df_train[df_train['sentiment']=='negative']['Difference'], shade=True, color='c')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:06.512804Z","iopub.execute_input":"2022-07-12T00:23:06.513285Z","iopub.status.idle":"2022-07-12T00:23:06.771647Z","shell.execute_reply.started":"2022-07-12T00:23:06.513254Z","shell.execute_reply":"2022-07-12T00:23:06.770776Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16,6))\np3 = sns.kdeplot(df_train[df_train['sentiment']=='neutral']['Difference'], shade=True, color='r').set_title(\"Kernel Distribution of Difference in Number of Words(Neutral)\")","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:06.773165Z","iopub.execute_input":"2022-07-12T00:23:06.773873Z","iopub.status.idle":"2022-07-12T00:23:07.053152Z","shell.execute_reply.started":"2022-07-12T00:23:06.773828Z","shell.execute_reply":"2022-07-12T00:23:07.052052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16,6))\np1 = sns.kdeplot(df_train[df_train['sentiment']=='positive']['jaccard_similarity'], shade=True, color='y').set_title(\"Kernel Distribution of Jaccard Similarity(Pos/Neg)\")\np2 = sns.kdeplot(df_train[df_train['sentiment']=='negative']['jaccard_similarity'], shade=True, color='g')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:07.054409Z","iopub.execute_input":"2022-07-12T00:23:07.054704Z","iopub.status.idle":"2022-07-12T00:23:07.308289Z","shell.execute_reply.started":"2022-07-12T00:23:07.054677Z","shell.execute_reply":"2022-07-12T00:23:07.307109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(16,6))\np1 = sns.kdeplot(df_train[df_train['sentiment']=='neutral']['jaccard_similarity'], shade=True, color='r').set_title(\"Kernel Distribution of Jaccard Similarity(Neutral)\")\n","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:07.309759Z","iopub.execute_input":"2022-07-12T00:23:07.310619Z","iopub.status.idle":"2022-07-12T00:23:07.577055Z","shell.execute_reply.started":"2022-07-12T00:23:07.310585Z","shell.execute_reply":"2022-07-12T00:23:07.575727Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def clean_text(text):\n    '''Make text lowercase, remove text in square brackets,remove links,remove punctuation\n    and remove words containing numbers.'''\n    text = str(text).lower()\n    text = re.sub('\\[.*?\\]', '', text)\n    text = re.sub('https?://\\S+|www\\.\\S+', '', text)\n    text = re.sub('http?://\\S+|www\\.\\S+', '', text)\n    text = re.sub('<.*?>+', '', text)\n    text = re.sub('[%s]' % re.escape(string.punctuation), '', text)\n    text = re.sub('\\n', '', text)\n    text = re.sub('\\w*\\d\\w*', '', text)\n    return text","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:07.578842Z","iopub.execute_input":"2022-07-12T00:23:07.579281Z","iopub.status.idle":"2022-07-12T00:23:07.587397Z","shell.execute_reply.started":"2022-07-12T00:23:07.579236Z","shell.execute_reply":"2022-07-12T00:23:07.586286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['text'] = df_train['text'].apply(lambda x : clean_text(x))\ndf_train['selected_text'] = df_train['selected_text'].apply(lambda x : clean_text(x))\ndf_train.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:07.588906Z","iopub.execute_input":"2022-07-12T00:23:07.589371Z","iopub.status.idle":"2022-07-12T00:23:09.148388Z","shell.execute_reply.started":"2022-07-12T00:23:07.589337Z","shell.execute_reply":"2022-07-12T00:23:09.147233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train['st_list'] = df_train['selected_text'].apply(lambda x : str(x).split())\ndf_train['text_list'] = df_train['text'].apply(lambda x : str(x).split())\n\ndef remove_stopwords(x):\n    return [y for y in x if y not in stopwords.words('english')]\n\ndf_train['st_list'] = df_train['st_list'].apply(lambda x : remove_stopwords(x))\ndf_train['text_list'] = df_train['text_list'].apply(lambda x : remove_stopwords(x))\ndf_train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:23:09.149730Z","iopub.execute_input":"2022-07-12T00:23:09.150038Z","iopub.status.idle":"2022-07-12T00:24:15.524461Z","shell.execute_reply.started":"2022-07-12T00:23:09.150011Z","shell.execute_reply":"2022-07-12T00:24:15.523141Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''most common words in positive sentiment selected text'''\ntop = Counter([item for sublist in df_train[df_train['sentiment']=='positive']['st_list'] for item in sublist])\ntop_pos = pd.DataFrame(top.most_common(20), columns=['Common Words', 'Count'])\ntop_pos.style.background_gradient(cmap='Greens')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:15.526434Z","iopub.execute_input":"2022-07-12T00:24:15.527262Z","iopub.status.idle":"2022-07-12T00:24:15.567887Z","shell.execute_reply.started":"2022-07-12T00:24:15.527214Z","shell.execute_reply":"2022-07-12T00:24:15.566945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''most common words in negative sentiment selected text'''\ntop = Counter([item for sublist in df_train[df_train['sentiment']=='negative']['st_list'] for item in sublist])\ntop_neg = pd.DataFrame(top.most_common(20), columns=['Common Words', 'Count'])\ntop_neg.style.background_gradient(cmap='Oranges')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:15.569357Z","iopub.execute_input":"2022-07-12T00:24:15.570460Z","iopub.status.idle":"2022-07-12T00:24:15.604378Z","shell.execute_reply.started":"2022-07-12T00:24:15.570419Z","shell.execute_reply":"2022-07-12T00:24:15.603551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''most common words in neutral sentiment selected text'''\ntop = Counter([item for sublist in df_train[df_train['sentiment']=='neutral']['st_list'] for item in sublist])\ntop_neu = pd.DataFrame(top.most_common(20), columns=['Common Words', 'Count'])\ntop_neu.style.background_gradient(cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:15.605691Z","iopub.execute_input":"2022-07-12T00:24:15.606267Z","iopub.status.idle":"2022-07-12T00:24:15.652611Z","shell.execute_reply.started":"2022-07-12T00:24:15.606234Z","shell.execute_reply":"2022-07-12T00:24:15.651794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def unique_words(sentiment, num):\n    all_other = []\n    for sublist in df_train[df_train['sentiment']!=sentiment]['st_list']:\n        for word in sublist:\n            all_other.append(word)\n    unique = Counter([word for sublist in df_train[df_train['sentiment']==sentiment]['st_list'] for word in sublist if word not in all_other])\n    return pd.DataFrame(unique.most_common(num), columns=['Words','Count'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:15.653891Z","iopub.execute_input":"2022-07-12T00:24:15.654436Z","iopub.status.idle":"2022-07-12T00:24:15.660868Z","shell.execute_reply.started":"2022-07-12T00:24:15.654403Z","shell.execute_reply":"2022-07-12T00:24:15.660068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_pos = unique_words('positive',20)\nprint(\"20 unique postive words:\")\nunique_pos.style.background_gradient(cmap='Greens')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:15.668123Z","iopub.execute_input":"2022-07-12T00:24:15.669474Z","iopub.status.idle":"2022-07-12T00:24:19.924395Z","shell.execute_reply.started":"2022-07-12T00:24:15.669428Z","shell.execute_reply":"2022-07-12T00:24:19.923255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_neg = unique_words('negative',20)\nprint(\"20 unique negative words:\")\nunique_neg.style.background_gradient(cmap='Oranges')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:19.926041Z","iopub.execute_input":"2022-07-12T00:24:19.926516Z","iopub.status.idle":"2022-07-12T00:24:25.384990Z","shell.execute_reply.started":"2022-07-12T00:24:19.926472Z","shell.execute_reply":"2022-07-12T00:24:25.383793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_neu = unique_words('neutral',20)\nprint(\"20 unique neutral words:\")\nunique_neu.style.background_gradient(cmap='Blues')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:25.386437Z","iopub.execute_input":"2022-07-12T00:24:25.386795Z","iopub.status.idle":"2022-07-12T00:24:40.724532Z","shell.execute_reply.started":"2022-07-12T00:24:25.386762Z","shell.execute_reply":"2022-07-12T00:24:40.723266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_wordcloud(text):\n    stopwords = set(STOPWORDS)\n    more_stopwords = {'u','im'}\n    stopwords = stopwords.union(more_stopwords)\n    wordcloud = WordCloud(background_color = 'white',\n                          stopwords = stopwords,\n                          max_words = 50,\n                          max_font_size = 40)\n    wordcloud.generate(str(text))\n    plt.figure(figsize=(12,8))\n    plt.imshow(wordcloud, interpolation='bilinear')\n    plt.axis('off')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:40.726358Z","iopub.execute_input":"2022-07-12T00:24:40.727212Z","iopub.status.idle":"2022-07-12T00:24:40.734815Z","shell.execute_reply.started":"2022-07-12T00:24:40.727149Z","shell.execute_reply":"2022-07-12T00:24:40.734014Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_wordcloud(df_train[df_train['sentiment']=='positive']['text'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:40.736290Z","iopub.execute_input":"2022-07-12T00:24:40.737333Z","iopub.status.idle":"2022-07-12T00:24:40.965903Z","shell.execute_reply.started":"2022-07-12T00:24:40.737291Z","shell.execute_reply":"2022-07-12T00:24:40.964796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_wordcloud(df_train[df_train['sentiment']=='negative']['text'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:40.967405Z","iopub.execute_input":"2022-07-12T00:24:40.968091Z","iopub.status.idle":"2022-07-12T00:24:41.214667Z","shell.execute_reply.started":"2022-07-12T00:24:40.968047Z","shell.execute_reply":"2022-07-12T00:24:41.213470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"create_wordcloud(df_train[df_train['sentiment']=='neutral']['text'])","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:41.216299Z","iopub.execute_input":"2022-07-12T00:24:41.217688Z","iopub.status.idle":"2022-07-12T00:24:41.442930Z","shell.execute_reply.started":"2022-07-12T00:24:41.217641Z","shell.execute_reply":"2022-07-12T00:24:41.441834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**NER Model**\n\nhttps://towardsdatascience.com/named-entity-recognition-with-nltk-and-spacy-8c4a7d88e7da\n\nhttps://spacy.io/usage/training#ner\n\nhttps://towardsdatascience.com/train-ner-with-custom-training-data-using-spacy-525ce748fab7","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('/kaggle/input/tweet-sentiment-extraction/train.csv')\ndf_test = pd.read_csv('/kaggle/input/tweet-sentiment-extraction/test.csv')\ndf_train = df_train.dropna()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:41.444648Z","iopub.execute_input":"2022-07-12T00:24:41.445405Z","iopub.status.idle":"2022-07-12T00:24:41.559848Z","shell.execute_reply.started":"2022-07-12T00:24:41.445358Z","shell.execute_reply":"2022-07-12T00:24:41.558669Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''return train data in a format needed for spacy NER'''\n\ndef get_training_data(sentiment):\n    train_data = []\n    for index, row in df_train.iterrows():\n        text = row.text\n        selected_text = row.selected_text\n        start = text.find(selected_text)\n        end = start + len(selected_text)\n        train_data.append((text, {\"entities\":[[start, end, 'selected_text']]}))\n    return train_data","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:41.561290Z","iopub.execute_input":"2022-07-12T00:24:41.561704Z","iopub.status.idle":"2022-07-12T00:24:41.569410Z","shell.execute_reply.started":"2022-07-12T00:24:41.561662Z","shell.execute_reply":"2022-07-12T00:24:41.567998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''return model output path'''\n\ndef get_model_out_path(sentiment):\n    model_out_path = None\n    if sentiment == 'positive':\n        model_out_path = 'models/model_pos'\n    elif sentiment == 'negative':\n        model_out_path = 'models/model_neg'\n    return model_out_path","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:41.571135Z","iopub.execute_input":"2022-07-12T00:24:41.571741Z","iopub.status.idle":"2022-07-12T00:24:41.578944Z","shell.execute_reply.started":"2022-07-12T00:24:41.571679Z","shell.execute_reply":"2022-07-12T00:24:41.577712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def trim_entity_spans(data: list) -> list:\n    \"\"\"Removes leading and trailing white spaces from entity spans.\n\n    Args:\n    data (list): The data to be cleaned in spaCy JSON format.\n\n    Returns:\n    list: The cleaned data.\n    \"\"\"\n    invalid_span_tokens = re.compile(r'\\s')\n\n    cleaned_data = []\n    for text, annotations in data:\n        entities = annotations['entities']\n        valid_entities = []\n        for start, end, label in entities:\n            valid_start = start\n            valid_end = end\n            # if there's preceding spaces, move the start position to nearest character\n            while valid_start < len(text) and invalid_span_tokens.match(\n                    text[valid_start]):\n                valid_start += 1\n            while valid_end > 1 and invalid_span_tokens.match(\n                    text[valid_end - 1]):\n                valid_end -= 1\n            valid_entities.append([valid_start, valid_end, label])\n        cleaned_data.append([text, {'entities': valid_entities}])\n    return cleaned_data","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:41.581088Z","iopub.execute_input":"2022-07-12T00:24:41.581993Z","iopub.status.idle":"2022-07-12T00:24:41.592825Z","shell.execute_reply.started":"2022-07-12T00:24:41.581944Z","shell.execute_reply":"2022-07-12T00:24:41.591766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(train_data, output_dir, n_iter=20, model=None):\n    train_data = trim_entity_spans(train_data)\n    if model is not None:\n        nlp = spacy.load(model)  \n        print(\"Loaded model '%s'\" % model)\n    else:\n        nlp = spacy.blank('en')  \n        print(\"Created blank 'en' model\")\n\n    # create the built-in pipeline components and add them to the pipeline\n    # nlp.create_pipe works for built-ins that are registered with spaCy\n    if \"ner\" not in nlp.pipe_names:\n        ner = nlp.create_pipe(\"ner\")\n        nlp.add_pipe(ner, last=True)\n    # otherwise, get it so we can add labels\n    else:\n        ner = nlp.get_pipe(\"ner\")\n    \n    # add labels\n    for _, annotations in train_data:\n        for ent in annotations.get(\"entities\"):\n            ner.add_label(ent[2])\n\n    # get names of other pipes to disable them during training\n    other_pipes = [pipe for pipe in nlp.pipe_names if pipe != \"ner\"]\n    with nlp.disable_pipes(*other_pipes):  # only train NER\n        optimizer = nlp.begin_training()\n        \n        for itn in tqdm(range(n_iter)):\n            random.shuffle(train_data)\n            losses = {}\n            for text, annotations in train_data:\n                try:\n                    nlp.update(\n                        [text],  \n                        [annotations],  \n                        drop=0.2,  \n                        sgd=optimizer,  \n                        losses=losses)\n                except Exception as error:\n                    continue\n            print(losses)\n    save_model(output_dir, nlp, 'st_ner')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:41.594541Z","iopub.execute_input":"2022-07-12T00:24:41.595362Z","iopub.status.idle":"2022-07-12T00:24:41.608218Z","shell.execute_reply.started":"2022-07-12T00:24:41.595323Z","shell.execute_reply":"2022-07-12T00:24:41.607220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def save_model(output_dir, nlp, new_model_name):\n    ''' This Function Saves model to \n    given output directory'''\n    \n    output_dir = f'../working/{output_dir}'\n    if output_dir is not None:        \n        if not os.path.exists(output_dir):\n            os.makedirs(output_dir)\n        nlp.meta[\"name\"] = new_model_name\n        nlp.to_disk(output_dir)\n        print(\"Saved model to\", output_dir)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:41.609823Z","iopub.execute_input":"2022-07-12T00:24:41.610116Z","iopub.status.idle":"2022-07-12T00:24:41.621834Z","shell.execute_reply.started":"2022-07-12T00:24:41.610088Z","shell.execute_reply":"2022-07-12T00:24:41.620938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sentiment = 'positive'\n\ntrain_data = get_training_data(sentiment)\nmodel_path = get_model_out_path(sentiment)\ntrain(train_data, model_path, n_iter=2, model=None)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:24:41.625500Z","iopub.execute_input":"2022-07-12T00:24:41.625852Z","iopub.status.idle":"2022-07-12T00:54:20.357916Z","shell.execute_reply.started":"2022-07-12T00:24:41.625820Z","shell.execute_reply":"2022-07-12T00:54:20.356747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sentiment = 'negative'\n\ntrain_data = get_training_data(sentiment)\nmodel_path = get_model_out_path(sentiment)\ntrain(train_data, model_path, n_iter=2, model=None)","metadata":{"execution":{"iopub.status.busy":"2022-07-12T00:54:20.359799Z","iopub.execute_input":"2022-07-12T00:54:20.360476Z","iopub.status.idle":"2022-07-12T01:23:44.111968Z","shell.execute_reply.started":"2022-07-12T00:54:20.360430Z","shell.execute_reply":"2022-07-12T01:23:44.111152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def predict_entities(text, model):\n    doc = model(text)\n    start=0\n    end=0\n    for ent in doc.ents:\n        start = text.find(ent.text)\n        end = start + len(ent.text)\n    selected_text = text[start: end]\n    return selected_text","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:23:44.113534Z","iopub.execute_input":"2022-07-12T01:23:44.114143Z","iopub.status.idle":"2022-07-12T01:23:44.119891Z","shell.execute_reply.started":"2022-07-12T01:23:44.114111Z","shell.execute_reply":"2022-07-12T01:23:44.119188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"selected_texts = []\nMODELS_BASE_PATH = 'models/'\n\nif MODELS_BASE_PATH is not None:\n    print(\"Loading Models  from \", MODELS_BASE_PATH)\n    model_pos = spacy.load(MODELS_BASE_PATH + 'model_pos')\n    model_neg = spacy.load(MODELS_BASE_PATH + 'model_neg')\n        \n    for index, row in df_test.iterrows():\n        text = row.text\n        output_str = \"\"\n        if row.sentiment == 'neutral' or len(text.split()) <= 2:\n            selected_texts.append(text)\n        elif row.sentiment == 'positive':\n            selected_texts.append(predict_entities(text, model_pos))\n        else:\n            selected_texts.append(predict_entities(text, model_neg))\n        \ndf_test['selected_text'] = selected_texts","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:23:44.121158Z","iopub.execute_input":"2022-07-12T01:23:44.122320Z","iopub.status.idle":"2022-07-12T01:23:52.806761Z","shell.execute_reply.started":"2022-07-12T01:23:44.122277Z","shell.execute_reply":"2022-07-12T01:23:52.805852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission = pd.read_csv('/kaggle/input/tweet-sentiment-extraction/sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:23:52.807967Z","iopub.execute_input":"2022-07-12T01:23:52.808803Z","iopub.status.idle":"2022-07-12T01:23:52.822651Z","shell.execute_reply.started":"2022-07-12T01:23:52.808759Z","shell.execute_reply":"2022-07-12T01:23:52.821454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:23:52.823851Z","iopub.execute_input":"2022-07-12T01:23:52.824131Z","iopub.status.idle":"2022-07-12T01:23:52.830312Z","shell.execute_reply.started":"2022-07-12T01:23:52.824105Z","shell.execute_reply":"2022-07-12T01:23:52.829294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:23:52.831690Z","iopub.execute_input":"2022-07-12T01:23:52.832476Z","iopub.status.idle":"2022-07-12T01:23:52.839461Z","shell.execute_reply.started":"2022-07-12T01:23:52.832445Z","shell.execute_reply":"2022-07-12T01:23:52.838508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:23:52.840631Z","iopub.execute_input":"2022-07-12T01:23:52.841509Z","iopub.status.idle":"2022-07-12T01:23:52.854431Z","shell.execute_reply.started":"2022-07-12T01:23:52.841469Z","shell.execute_reply":"2022-07-12T01:23:52.853528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_submission['selected_text'] = df_test['selected_text']\ndf_submission.to_csv(\"submission.csv\", index=False)\ndisplay(df_submission.head(10))","metadata":{"execution":{"iopub.status.busy":"2022-07-12T01:23:52.855690Z","iopub.execute_input":"2022-07-12T01:23:52.856126Z","iopub.status.idle":"2022-07-12T01:23:52.884117Z","shell.execute_reply.started":"2022-07-12T01:23:52.856092Z","shell.execute_reply":"2022-07-12T01:23:52.883224Z"},"trusted":true},"execution_count":null,"outputs":[]}]}