{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# This Notebook is my first attempt at understanding NLP and expand my knowledge. \n## Please feel free to comment for any doubts or improvement advice.","metadata":{}},{"cell_type":"code","source":"!pip install num2words","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:57:15.088469Z","iopub.execute_input":"2022-08-03T14:57:15.089009Z","iopub.status.idle":"2022-08-03T14:57:28.973514Z","shell.execute_reply.started":"2022-08-03T14:57:15.088924Z","shell.execute_reply":"2022-08-03T14:57:28.971911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\n%matplotlib inline\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\nimport string\nimport re\nimport nltk\nfrom nltk.corpus import stopwords\nfrom nltk.tokenize import word_tokenize\nfrom nltk.tokenize import WhitespaceTokenizer\nfrom nltk.stem import PorterStemmer, LancasterStemmer\nfrom nltk.stem import SnowballStemmer\nfrom collections import Counter\nfrom num2words import num2words\nfrom nltk.stem import WordNetLemmatizer\n\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.text import Tokenizer\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-03T14:57:37.059509Z","iopub.execute_input":"2022-08-03T14:57:37.059950Z","iopub.status.idle":"2022-08-03T14:57:45.862936Z","shell.execute_reply.started":"2022-08-03T14:57:37.059912Z","shell.execute_reply":"2022-08-03T14:57:45.861703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv(\"../input/nlp-getting-started/train.csv\")\ntest = pd.read_csv(\"../input/nlp-getting-started/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:58:05.845277Z","iopub.execute_input":"2022-08-03T14:58:05.846399Z","iopub.status.idle":"2022-08-03T14:58:05.888295Z","shell.execute_reply.started":"2022-08-03T14:58:05.846342Z","shell.execute_reply":"2022-08-03T14:58:05.886771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.info())","metadata":{"execution":{"iopub.status.busy":"2022-07-29T22:04:00.733135Z","iopub.execute_input":"2022-07-29T22:04:00.734207Z","iopub.status.idle":"2022-07-29T22:04:00.782960Z","shell.execute_reply.started":"2022-07-29T22:04:00.734151Z","shell.execute_reply":"2022-07-29T22:04:00.781779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(train.shape)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T22:04:02.658993Z","iopub.execute_input":"2022-07-29T22:04:02.659376Z","iopub.status.idle":"2022-07-29T22:04:02.665292Z","shell.execute_reply.started":"2022-07-29T22:04:02.659347Z","shell.execute_reply":"2022-07-29T22:04:02.664243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-29T22:04:10.810148Z","iopub.execute_input":"2022-07-29T22:04:10.810529Z","iopub.status.idle":"2022-07-29T22:04:10.831594Z","shell.execute_reply.started":"2022-07-29T22:04:10.810500Z","shell.execute_reply":"2022-07-29T22:04:10.830835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# In order to proceed further I will perform some EDA to understand the data.","metadata":{}},{"cell_type":"code","source":"sns.set_style('darkgrid')\nsns.countplot(x=train['target'],palette='GnBu')","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:58:18.009139Z","iopub.execute_input":"2022-08-03T14:58:18.009562Z","iopub.status.idle":"2022-08-03T14:58:18.247231Z","shell.execute_reply.started":"2022-08-03T14:58:18.009527Z","shell.execute_reply":"2022-08-03T14:58:18.245674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The following is the meaning of the target labels:\n* 0: Non Disaster\n* 1: Disaster","metadata":{}},{"cell_type":"code","source":"fig,ax=plt.subplots(1,2,figsize=(10,5))\ndisaster_tweet=train[train['target']==1]['text'].str.len()\nax[0].hist(disaster_tweet,color='#3b6df4')\nax[0].set_title('disaster tweets')\nnotdisaster_tweet=train[train['target']==0]['text'].str.len()\nax[1].hist(notdisaster_tweet,color='#8de9b8')\nax[1].set_title('Not disaster tweets')\nfig.suptitle('Characters in tweets')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:58:28.536633Z","iopub.execute_input":"2022-08-03T14:58:28.537047Z","iopub.status.idle":"2022-08-03T14:58:28.994405Z","shell.execute_reply.started":"2022-08-03T14:58:28.537012Z","shell.execute_reply":"2022-08-03T14:58:28.993042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-07-29T22:05:15.389809Z","iopub.execute_input":"2022-07-29T22:05:15.390207Z","iopub.status.idle":"2022-07-29T22:05:15.409761Z","shell.execute_reply.started":"2022-07-29T22:05:15.390176Z","shell.execute_reply":"2022-07-29T22:05:15.408987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['location'].value_counts().head(n=20)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T22:05:24.523829Z","iopub.execute_input":"2022-07-29T22:05:24.524207Z","iopub.status.idle":"2022-07-29T22:05:24.535677Z","shell.execute_reply.started":"2022-07-29T22:05:24.524176Z","shell.execute_reply":"2022-07-29T22:05:24.534568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize = (10, 7))\nax = plt.axes()\nax.set_facecolor('#d9ebf7')\nax = ((train.location.value_counts())[:10]).plot(kind = 'bar', color = '#8de9b8', linewidth = 2)\nplt.title('Location Count', fontsize = 14)\nplt.xlabel('Location', fontsize = 12)\nplt.ylabel('Count', fontsize = 12)\nax.xaxis.set_tick_params(labelsize = 12, rotation = 90)\nax.yaxis.set_tick_params(labelsize = 12)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:58:36.379182Z","iopub.execute_input":"2022-08-03T14:58:36.379757Z","iopub.status.idle":"2022-08-03T14:58:36.696369Z","shell.execute_reply.started":"2022-08-03T14:58:36.379720Z","shell.execute_reply":"2022-08-03T14:58:36.694724Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['keyword'].value_counts().head(10).plot(kind='bar',figsize= (20,10))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:58:56.062958Z","iopub.execute_input":"2022-08-03T14:58:56.063581Z","iopub.status.idle":"2022-08-03T14:58:56.389843Z","shell.execute_reply.started":"2022-08-03T14:58:56.063533Z","shell.execute_reply":"2022-08-03T14:58:56.388377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Text Cleaning","metadata":{}},{"cell_type":"markdown","source":"It is a good practice to remove noise from text data for NLP. Some examples of noise can be:\n* URL \n* Emoticons\n* Tags\n* Punctuation Marks\n* Stop Words\n* Stemming","metadata":{}},{"cell_type":"code","source":"# Some basic helper functions to clean text by removing urls, emojis, html tags, punctuations and Stop Words.\n\ndef remove_URL(data):\n    url = re.compile(r'https?://\\S+|www\\.\\S+')\n    return url.sub(r'', data)\n\n\ndef remove_emoji(data):\n    emoji_pattern = re.compile(\n        '['\n        u'\\U0001F600-\\U0001F64F'  # emoticons\n        u'\\U0001F300-\\U0001F5FF'  # symbols & pictographs\n        u'\\U0001F680-\\U0001F6FF'  # transport & map symbols\n        u'\\U0001F1E0-\\U0001F1FF'  # flags (iOS)\n        u'\\U00002702-\\U000027B0'\n        u'\\U000024C2-\\U0001F251'\n        ']+',\n        flags=re.UNICODE)\n    return emoji_pattern.sub(r'', data)\n\n\ndef remove_html(data):\n    html = re.compile(r'<.*?>|&([a-z0-9]+|#[0-9]{1,6}|#x[0-9a-f]{1,6});')\n    return re.sub(html, '', data)\n\n\ndef remove_punct(data):\n    table = str.maketrans('', '', string.punctuation)\n    return data.translate(table)\n\nstop = stopwords.words('english')\nporter = PorterStemmer()\nlancaster = LancasterStemmer()\nsnowball = SnowballStemmer(\"english\")\nlemmatizer = WordNetLemmatizer()\nw_tokenizer = WhitespaceTokenizer()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:59:25.772189Z","iopub.execute_input":"2022-08-03T14:59:25.772656Z","iopub.status.idle":"2022-08-03T14:59:25.783053Z","shell.execute_reply.started":"2022-08-03T14:59:25.772616Z","shell.execute_reply":"2022-08-03T14:59:25.782147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Applying helper functions on Train Dataset\n\ntrain['text_clean'] = train['text'].apply(lambda x: remove_URL(x))\ntrain['text_clean'] = train['text_clean'].apply(lambda x: remove_emoji(x))\ntrain['text_clean'] = train['text_clean'].apply(lambda x: remove_html(x))\ntrain['text_clean'] = train['text_clean'].apply(lambda x: remove_punct(x))\ntrain['text_clean'] = train['text_clean'].apply(lambda x: ' '.join([word for word in x.split() if word not in (stop)]))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:59:40.470075Z","iopub.execute_input":"2022-08-03T14:59:40.470539Z","iopub.status.idle":"2022-08-03T14:59:40.905894Z","shell.execute_reply.started":"2022-08-03T14:59:40.470501Z","shell.execute_reply":"2022-08-03T14:59:40.901584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['text_clean'] = train['text_clean'].apply(lambda x: ' '.join([porter.stem(word) for word in x.split()]))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:59:49.608063Z","iopub.execute_input":"2022-08-03T14:59:49.608524Z","iopub.status.idle":"2022-08-03T14:59:51.775850Z","shell.execute_reply.started":"2022-08-03T14:59:49.608485Z","shell.execute_reply":"2022-08-03T14:59:51.774122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:12:56.818509Z","iopub.execute_input":"2022-08-03T15:12:56.818972Z","iopub.status.idle":"2022-08-03T15:12:56.844179Z","shell.execute_reply.started":"2022-08-03T15:12:56.818935Z","shell.execute_reply":"2022-08-03T15:12:56.842504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test['text_clean'] = test['text'].apply(lambda x: remove_URL(x))\ntest['text_clean'] = test['text_clean'].apply(lambda x: remove_emoji(x))\ntest['text_clean'] = test['text_clean'].apply(lambda x: remove_html(x))\ntest['text_clean'] = test['text_clean'].apply(lambda x: remove_punct(x))\ntest['text_clean'] = test['text_clean'].apply(lambda x: ' '.join([word for word in x.split() if word not in (stop)]))\ntest['text_clean'] = test['text_clean'].apply(lambda x: ' '.join([porter.stem(word) for word in x.split()]))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T14:59:57.207558Z","iopub.execute_input":"2022-08-03T14:59:57.207994Z","iopub.status.idle":"2022-08-03T14:59:58.316814Z","shell.execute_reply.started":"2022-08-03T14:59:57.207951Z","shell.execute_reply":"2022-08-03T14:59:58.315540Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:27:11.022076Z","iopub.execute_input":"2022-08-03T15:27:11.022518Z","iopub.status.idle":"2022-08-03T15:27:11.047036Z","shell.execute_reply.started":"2022-08-03T15:27:11.022473Z","shell.execute_reply":"2022-08-03T15:27:11.045717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training LSTM Model","metadata":{}},{"cell_type":"code","source":"df_train = train.drop(['id','keyword','location','text'],axis=1)\ndf_test = test.drop(['keyword','location','text'],axis=1)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:00:03.107983Z","iopub.execute_input":"2022-08-03T15:00:03.108431Z","iopub.status.idle":"2022-08-03T15:00:03.122284Z","shell.execute_reply.started":"2022-08-03T15:00:03.108381Z","shell.execute_reply":"2022-08-03T15:00:03.121231Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Move texts and targets from train & test dataset to the list:\ntraining_message = []\ntesting_message = []\ntraining_labels = []\nfor i in range(len(df_train)):\n    training_message.append(df_train.loc[i,'text_clean'])\n    training_labels.append(df_train.loc[i,'target'])\n    \nfor i in range(len(df_test)):\n    testing_message.append(df_test.loc[i,'text_clean'])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:00:07.101152Z","iopub.execute_input":"2022-08-03T15:00:07.101597Z","iopub.status.idle":"2022-08-03T15:00:07.318160Z","shell.execute_reply.started":"2022-08-03T15:00:07.101561Z","shell.execute_reply":"2022-08-03T15:00:07.316750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vocab_size = 23000\nembedding_dim = 32\nmax_length = 100\ntrunc_type='post'\npadding_type='post'\noov_tok = \"<OOV>\"","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:00:12.022695Z","iopub.execute_input":"2022-08-03T15:00:12.023139Z","iopub.status.idle":"2022-08-03T15:00:12.029049Z","shell.execute_reply.started":"2022-08-03T15:00:12.023104Z","shell.execute_reply":"2022-08-03T15:00:12.027917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tokenization Process:\ntokenizer = Tokenizer(num_words = vocab_size, oov_token=oov_tok)\ntokenizer.fit_on_texts(training_message)\nword_index = tokenizer.word_index","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:00:15.606683Z","iopub.execute_input":"2022-08-03T15:00:15.607179Z","iopub.status.idle":"2022-08-03T15:00:15.781598Z","shell.execute_reply.started":"2022-08-03T15:00:15.607138Z","shell.execute_reply":"2022-08-03T15:00:15.780483Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# texts_to_sequences Process:\ntraining_message = tokenizer.texts_to_sequences(training_message)\ntesting_message = tokenizer.texts_to_sequences(testing_message)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:00:18.849178Z","iopub.execute_input":"2022-08-03T15:00:18.849666Z","iopub.status.idle":"2022-08-03T15:00:19.028522Z","shell.execute_reply.started":"2022-08-03T15:00:18.849628Z","shell.execute_reply":"2022-08-03T15:00:19.026314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#pad_sequences Process:\ntraining_padded = pad_sequences(training_message,maxlen=max_length, truncating=trunc_type, padding=padding_type)\ntesting_padded = pad_sequences(testing_message,maxlen=max_length, truncating=trunc_type, padding=padding_type)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:00:24.308213Z","iopub.execute_input":"2022-08-03T15:00:24.308945Z","iopub.status.idle":"2022-08-03T15:00:24.352962Z","shell.execute_reply.started":"2022-08-03T15:00:24.308890Z","shell.execute_reply":"2022-08-03T15:00:24.352061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_padded = np.array(training_padded)\ntesting_padded = np.array(testing_padded)\ntraining_labels = np.array(training_labels)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:00:27.491260Z","iopub.execute_input":"2022-08-03T15:00:27.491679Z","iopub.status.idle":"2022-08-03T15:00:27.500779Z","shell.execute_reply.started":"2022-08-03T15:00:27.491644Z","shell.execute_reply":"2022-08-03T15:00:27.499351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = tf.keras.Sequential([\n    tf.keras.layers.Embedding(vocab_size, embedding_dim, input_length=max_length),\n    tf.keras.layers.Bidirectional(tf.keras.layers.LSTM(12)),\n    tf.keras.layers.Dense(1, activation = 'sigmoid')\n])\n\nmodel.compile(\n    loss='binary_crossentropy',\n    optimizer='Adamax',\n    metrics=['accuracy', tf.keras.metrics.Precision(), tf.keras.metrics.Recall()]\n)\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:00:33.027298Z","iopub.execute_input":"2022-08-03T15:00:33.028538Z","iopub.status.idle":"2022-08-03T15:00:33.768154Z","shell.execute_reply.started":"2022-08-03T15:00:33.028483Z","shell.execute_reply":"2022-08-03T15:00:33.766817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    training_padded, \n    training_labels, \n    epochs = 50, \n    batch_size = 64,  \n    validation_split=0.2\n)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:01:22.885474Z","iopub.execute_input":"2022-08-03T15:01:22.885929Z","iopub.status.idle":"2022-08-03T15:09:49.421829Z","shell.execute_reply.started":"2022-08-03T15:01:22.885896Z","shell.execute_reply":"2022-08-03T15:09:49.420228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history[\"accuracy\"], color=\"r\")\nplt.plot(history.history[\"val_accuracy\"], color=\"g\")\nplt.legend([\"Training\", \"Validation\"])\nplt.xlabel(\"epochs\")\nplt.ylabel(\"accuracy\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:09:49.424524Z","iopub.execute_input":"2022-08-03T15:09:49.424941Z","iopub.status.idle":"2022-08-03T15:09:49.735034Z","shell.execute_reply.started":"2022-08-03T15:09:49.424905Z","shell.execute_reply":"2022-08-03T15:09:49.733694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.plot(history.history[\"loss\"], color=\"r\")\nplt.plot(history.history[\"val_loss\"], color=\"g\")\nplt.title('Model Loss')\nplt.ylabel('Loss')\nplt.xlabel('Epoch')\nplt.legend(['train', 'validation'])\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:10:22.818968Z","iopub.execute_input":"2022-08-03T15:10:22.819411Z","iopub.status.idle":"2022-08-03T15:10:23.100306Z","shell.execute_reply.started":"2022-08-03T15:10:22.819373Z","shell.execute_reply":"2022-08-03T15:10:23.099073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_predict = model.predict(testing_padded).round().astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:10:41.355807Z","iopub.execute_input":"2022-08-03T15:10:41.356810Z","iopub.status.idle":"2022-08-03T15:10:43.197514Z","shell.execute_reply.started":"2022-08-03T15:10:41.356753Z","shell.execute_reply":"2022-08-03T15:10:43.196021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission","metadata":{}},{"cell_type":"code","source":"submission = pd.DataFrame({'id':df_test['id'],'target':test_predict.ravel()})\nsubmission.to_csv('submission.csv',index=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-29T22:19:34.391181Z","iopub.execute_input":"2022-07-29T22:19:34.391611Z","iopub.status.idle":"2022-07-29T22:19:34.404975Z","shell.execute_reply.started":"2022-07-29T22:19:34.391577Z","shell.execute_reply":"2022-07-29T22:19:34.403894Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# I willl further use BERT and GloVe and compare them to the output of the LSTM. I do realize that the LSTM requires more tuning and am open to suggestions.\n\n## I also have to perform the following:\n* Remove Stopwords\n* Stemming\n* Lemmatisation\n* TF-IDF (Feature Engineering)","metadata":{}},{"cell_type":"markdown","source":"TF - IDF","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer\n\ncvec = CountVectorizer(min_df=.005, max_df=.9, ngram_range=(1,2), tokenizer=lambda doc: doc, lowercase=False)\ncvec.fit(train['text_clean'])","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:13:28.277065Z","iopub.execute_input":"2022-08-03T15:13:28.277512Z","iopub.status.idle":"2022-08-03T15:13:28.920479Z","shell.execute_reply.started":"2022-08-03T15:13:28.277475Z","shell.execute_reply":"2022-08-03T15:13:28.918928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cvec_counts = cvec.transform(train['text_clean'])\nprint('sparse matrix shape:', cvec_counts.shape)\nprint('nonzero count:', cvec_counts.nnz)\nprint('sparsity: %.2f%%' % (100.0 * cvec_counts.nnz / (cvec_counts.shape[0] * cvec_counts.shape[1])))","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:13:50.713070Z","iopub.execute_input":"2022-08-03T15:13:50.713880Z","iopub.status.idle":"2022-08-03T15:13:51.336303Z","shell.execute_reply.started":"2022-08-03T15:13:50.713819Z","shell.execute_reply":"2022-08-03T15:13:51.335179Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.feature_extraction.text import TfidfTransformer\ntransformer = TfidfTransformer()\ntransformed_weights = transformer.fit_transform(cvec_counts)\ntransformed_weights","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:14:22.282922Z","iopub.execute_input":"2022-08-03T15:14:22.284114Z","iopub.status.idle":"2022-08-03T15:14:22.320326Z","shell.execute_reply.started":"2022-08-03T15:14:22.284069Z","shell.execute_reply":"2022-08-03T15:14:22.318974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"transformed_weights = transformed_weights.toarray()\nvocab = cvec.get_feature_names()\ntfidf = pd.DataFrame(transformed_weights, columns=vocab)\ntfidf['Keyword'] = tfidf.idxmax(axis=1)\ntfidf['Max'] = tfidf.max(axis=1)\ntfidf['Sum'] = tfidf.drop('Max', axis=1).sum(axis=1)\ntfidf.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:14:45.003528Z","iopub.execute_input":"2022-08-03T15:14:45.004334Z","iopub.status.idle":"2022-08-03T15:14:47.392804Z","shell.execute_reply.started":"2022-08-03T15:14:45.004287Z","shell.execute_reply":"2022-08-03T15:14:47.391398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_1 = train.drop(['id', 'keyword','location'], axis=1)\ntfidf_new = pd.merge(train_1, tfidf, left_index=True, right_index=True)\ntfidf_new","metadata":{"execution":{"iopub.status.busy":"2022-08-03T15:25:36.512293Z","iopub.execute_input":"2022-08-03T15:25:36.512719Z","iopub.status.idle":"2022-08-03T15:25:36.582843Z","shell.execute_reply.started":"2022-08-03T15:25:36.512684Z","shell.execute_reply":"2022-08-03T15:25:36.580970Z"},"trusted":true},"execution_count":null,"outputs":[]}]}