{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Reading & Exploring Data:","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:07:36.713035Z","iopub.execute_input":"2023-08-26T19:07:36.713836Z","iopub.status.idle":"2023-08-26T19:07:36.719490Z","shell.execute_reply.started":"2023-08-26T19:07:36.713789Z","shell.execute_reply":"2023-08-26T19:07:36.718087Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = '/kaggle/input/quora-insincere-questions-classification'\nos.listdir(file_path)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:08:50.249971Z","iopub.execute_input":"2023-08-26T19:08:50.250385Z","iopub.status.idle":"2023-08-26T19:08:50.260225Z","shell.execute_reply.started":"2023-08-26T19:08:50.250350Z","shell.execute_reply":"2023-08-26T19:08:50.258941Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_fname = os.path.join(file_path, 'train.csv')\ntest_fname = os.path.join(file_path, 'test.csv')\nsubmission_fname = os.path.join(file_path, 'sample_submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:31:06.757158Z","iopub.execute_input":"2023-08-26T20:31:06.758410Z","iopub.status.idle":"2023-08-26T20:31:06.765982Z","shell.execute_reply.started":"2023-08-26T20:31:06.758357Z","shell.execute_reply":"2023-08-26T20:31:06.764474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_train_df = pd.read_csv(train_fname)\nraw_test_df = pd.read_csv(test_fname)\nsubmission_df = pd.read_csv(submission_fname)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:31:13.496360Z","iopub.execute_input":"2023-08-26T20:31:13.496792Z","iopub.status.idle":"2023-08-26T20:31:18.299302Z","shell.execute_reply.started":"2023-08-26T20:31:13.496759Z","shell.execute_reply":"2023-08-26T20:31:18.298138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:11:37.710668Z","iopub.execute_input":"2023-08-26T19:11:37.711078Z","iopub.status.idle":"2023-08-26T19:11:37.734901Z","shell.execute_reply.started":"2023-08-26T19:11:37.711048Z","shell.execute_reply":"2023-08-26T19:11:37.733616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_train_df['target'].value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:11:40.247596Z","iopub.execute_input":"2023-08-26T19:11:40.248046Z","iopub.status.idle":"2023-08-26T19:11:40.278830Z","shell.execute_reply.started":"2023-08-26T19:11:40.248014Z","shell.execute_reply":"2023-08-26T19:11:40.277148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_train_df['target'].value_counts(normalize=True).plot(kind='bar')","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:11:41.484976Z","iopub.execute_input":"2023-08-26T19:11:41.486018Z","iopub.status.idle":"2023-08-26T19:11:41.756363Z","shell.execute_reply.started":"2023-08-26T19:11:41.485977Z","shell.execute_reply":"2023-08-26T19:11:41.754966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sincere_df = raw_train_df[raw_train_df['target']==0].copy()\ninsincere_df = raw_train_df[raw_train_df['target']==1].copy()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:11:43.289890Z","iopub.execute_input":"2023-08-26T19:11:43.290326Z","iopub.status.idle":"2023-08-26T19:11:43.502428Z","shell.execute_reply.started":"2023-08-26T19:11:43.290291Z","shell.execute_reply":"2023-08-26T19:11:43.500900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sincere_df['question_text'].values[:10]","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:11:44.332057Z","iopub.execute_input":"2023-08-26T19:11:44.332521Z","iopub.status.idle":"2023-08-26T19:11:44.341488Z","shell.execute_reply.started":"2023-08-26T19:11:44.332480Z","shell.execute_reply":"2023-08-26T19:11:44.339818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"insincere_df['question_text'].values[:10]","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:11:44.901254Z","iopub.execute_input":"2023-08-26T19:11:44.901714Z","iopub.status.idle":"2023-08-26T19:11:44.910300Z","shell.execute_reply.started":"2023-08-26T19:11:44.901678Z","shell.execute_reply":"2023-08-26T19:11:44.909166Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_test_df","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:11:46.788583Z","iopub.execute_input":"2023-08-26T19:11:46.789506Z","iopub.status.idle":"2023-08-26T19:11:46.803001Z","shell.execute_reply.started":"2023-08-26T19:11:46.789467Z","shell.execute_reply":"2023-08-26T19:11:46.801692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:31:32.044596Z","iopub.execute_input":"2023-08-26T20:31:32.045079Z","iopub.status.idle":"2023-08-26T20:31:32.060167Z","shell.execute_reply.started":"2023-08-26T20:31:32.045039Z","shell.execute_reply":"2023-08-26T20:31:32.058846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Text Preprocessing Techniques:","metadata":{}},{"cell_type":"markdown","source":"### Tokenization:\n\n<p>\n\nTokenization is a basic text processing technique which breaks down text into smaller units called as tokens. Tokens can be words, sub-words, or even characters dependes on the level of granularity.\n    \nTokenization is an important step as it forms basis for various NLP tasks and provide structured representation of text to be used by machine learning algorithms. It makes it easier to analyze data based on tokens to find out the patterns across the text which can't be done on text directly.\n    \nThere are many NLP libraries that provide builtin functions to tokenize text for further analysis. \n\n</p>","metadata":{}},{"cell_type":"code","source":"import nltk","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:11:50.948308Z","iopub.execute_input":"2023-08-26T19:11:50.948724Z","iopub.status.idle":"2023-08-26T19:11:52.183026Z","shell.execute_reply.started":"2023-08-26T19:11:50.948693Z","shell.execute_reply":"2023-08-26T19:11:52.181976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from nltk import word_tokenize\nnltk.download('punkt')","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:11:52.184848Z","iopub.execute_input":"2023-08-26T19:11:52.192922Z","iopub.status.idle":"2023-08-26T19:11:52.388245Z","shell.execute_reply.started":"2023-08-26T19:11:52.192857Z","shell.execute_reply":"2023-08-26T19:11:52.386984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q0 = sincere_df['question_text'].values[0]\nq0","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:11:53.923564Z","iopub.execute_input":"2023-08-26T19:11:53.923984Z","iopub.status.idle":"2023-08-26T19:11:53.931809Z","shell.execute_reply.started":"2023-08-26T19:11:53.923951Z","shell.execute_reply":"2023-08-26T19:11:53.930452Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q0_token = word_tokenize(q0)\nq0_token","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:11:54.348589Z","iopub.execute_input":"2023-08-26T19:11:54.348978Z","iopub.status.idle":"2023-08-26T19:11:54.368879Z","shell.execute_reply.started":"2023-08-26T19:11:54.348950Z","shell.execute_reply":"2023-08-26T19:11:54.367621Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q1 = insincere_df['question_text'].values[0]\nq1","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:00.369911Z","iopub.execute_input":"2023-08-26T19:12:00.370359Z","iopub.status.idle":"2023-08-26T19:12:00.378768Z","shell.execute_reply.started":"2023-08-26T19:12:00.370324Z","shell.execute_reply":"2023-08-26T19:12:00.377210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q1_token = word_tokenize(q1)\nq1_token","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:00.773838Z","iopub.execute_input":"2023-08-26T19:12:00.774255Z","iopub.status.idle":"2023-08-26T19:12:00.783287Z","shell.execute_reply.started":"2023-08-26T19:12:00.774223Z","shell.execute_reply":"2023-08-26T19:12:00.782015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Removing Stop Words:\n<p>\nStopwords are very common in a language which occur frequently but usually don't contribute significantly to the meaning of a sentence or text. They may also not exhibit any pattern in text or documents due to their occurence across the text. They include words like \"the\", \"and\", \"is\" etc and often removed during text preprocessing to reduce noise and focus on meaningful words.\n    \nNLTK provides built-in stopwords for various languages but have many words with negation such as haven't, couldn't etc. and removing all of them may alter the intended meaning of the text. It's really important to consider the context and analyze whether removing certain stopwords migth change the sentiment or meaning of a sentence.\n    \nCreating a custom stopwords collection is a good solution to address the issue of removing negation words and other contextually important stopwords. Creating a custom list allows to carefully select which stopwords to remove while preserving the necessary ones.\n</p>","metadata":{}},{"cell_type":"code","source":"from nltk.corpus import stopwords","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:04.315815Z","iopub.execute_input":"2023-08-26T19:12:04.316640Z","iopub.status.idle":"2023-08-26T19:12:04.321633Z","shell.execute_reply.started":"2023-08-26T19:12:04.316601Z","shell.execute_reply":"2023-08-26T19:12:04.320548Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nltk.download('stopwords')","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:05.287695Z","iopub.execute_input":"2023-08-26T19:12:05.288350Z","iopub.status.idle":"2023-08-26T19:12:05.302697Z","shell.execute_reply.started":"2023-08-26T19:12:05.288317Z","shell.execute_reply":"2023-08-26T19:12:05.301327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"en_stopwords = stopwords.words('english')","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:05.722331Z","iopub.execute_input":"2023-08-26T19:12:05.723107Z","iopub.status.idle":"2023-08-26T19:12:05.730738Z","shell.execute_reply.started":"2023-08-26T19:12:05.723067Z","shell.execute_reply":"2023-08-26T19:12:05.729741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# look at the available stopwords\n\", \".join(en_stopwords)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:28.025500Z","iopub.execute_input":"2023-08-26T19:12:28.025911Z","iopub.status.idle":"2023-08-26T19:12:28.034869Z","shell.execute_reply.started":"2023-08-26T19:12:28.025882Z","shell.execute_reply":"2023-08-26T19:12:28.033439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_stopwords(tokens):\n    return [token for token in tokens if token.lower() not in en_stopwords]","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:31.044840Z","iopub.execute_input":"2023-08-26T19:12:31.045260Z","iopub.status.idle":"2023-08-26T19:12:31.050810Z","shell.execute_reply.started":"2023-08-26T19:12:31.045229Z","shell.execute_reply":"2023-08-26T19:12:31.049748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q0_stopwords = remove_stopwords(q0_token)\nprint(f\"Tokens with stopwords: {q0_token}\")\nprint(f\"Tokens without stopwords: {q0_stopwords}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:33.537725Z","iopub.execute_input":"2023-08-26T19:12:33.538207Z","iopub.status.idle":"2023-08-26T19:12:33.544686Z","shell.execute_reply.started":"2023-08-26T19:12:33.538173Z","shell.execute_reply":"2023-08-26T19:12:33.543639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q1_stopwords = remove_stopwords(q1_token)\nprint(f\"Tokens with stopwords: {q1_token}\")\nprint(f\"Tokens without stopwords: {q1_stopwords}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:33.569273Z","iopub.execute_input":"2023-08-26T19:12:33.569723Z","iopub.status.idle":"2023-08-26T19:12:33.576084Z","shell.execute_reply.started":"2023-08-26T19:12:33.569688Z","shell.execute_reply":"2023-08-26T19:12:33.575216Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Stemming:\n\nTo reduce the vocabulary length & increase word frequency, we use stemming which reduce words to their stem words, base or root form. Stemming is a heuristic process which reduce inflated words (or sometimes drive) based on set of rules to a common form so that they can be treated as the same word during information retrieval and analysis.\n\nIt is important to acknowledge that stemming does not provide valid and meangiful stem word always. In many cases, stemming produces results that are not actual words or might not capture the intended meaning of the original word.\n\nHowever, stemming can still be useful in scenarios where you're more concerned with pattern recognition and text classification, and where the potential loss of meaning is acceptable. It can help group together variations of words to simplify analysis, even if the resulting stems are not always linguistically meaningful.\n\n**For instance:**\n\nThe stem of \"history\" is \"histori,\" which isn't a recognized English word.</br>\nThe stem of \"people\" is \"peopl,\" which again is not a valid word.","metadata":{}},{"cell_type":"code","source":"from nltk.stem import SnowballStemmer","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:36.334655Z","iopub.execute_input":"2023-08-26T19:12:36.335070Z","iopub.status.idle":"2023-08-26T19:12:36.341027Z","shell.execute_reply.started":"2023-08-26T19:12:36.335039Z","shell.execute_reply":"2023-08-26T19:12:36.339610Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stemmer = SnowballStemmer('english')","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:36.654639Z","iopub.execute_input":"2023-08-26T19:12:36.655039Z","iopub.status.idle":"2023-08-26T19:12:36.660207Z","shell.execute_reply.started":"2023-08-26T19:12:36.655009Z","shell.execute_reply":"2023-08-26T19:12:36.658726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q0_stem = [stemmer.stem(word) for word in q0_stopwords]\nq1_stem = [stemmer.stem(word) for word in q1_stopwords]","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:37.960234Z","iopub.execute_input":"2023-08-26T19:12:37.961000Z","iopub.status.idle":"2023-08-26T19:12:37.966675Z","shell.execute_reply.started":"2023-08-26T19:12:37.960961Z","shell.execute_reply":"2023-08-26T19:12:37.965434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Tokens before Stemming: {q0_stopwords}\")\nprint(f\"Tokens after Stemming: {q0_stem}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:38.235524Z","iopub.execute_input":"2023-08-26T19:12:38.236025Z","iopub.status.idle":"2023-08-26T19:12:38.242089Z","shell.execute_reply.started":"2023-08-26T19:12:38.235984Z","shell.execute_reply":"2023-08-26T19:12:38.240891Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Tokens before Stemming: {q1_stopwords}\")\nprint(f\"Tokens after Stemming: {q1_stem}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:38.718321Z","iopub.execute_input":"2023-08-26T19:12:38.718763Z","iopub.status.idle":"2023-08-26T19:12:38.725638Z","shell.execute_reply.started":"2023-08-26T19:12:38.718729Z","shell.execute_reply":"2023-08-26T19:12:38.724316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Lemmatization:\n\n<p>\nLemmatization also reduces words to their base or dictionary form, but unlike stemming, ensures that the resulting base word is a valid word, found in a dictionary. Therefore, it typically provides more meaningful results for humans. It considers the context and part of speech of the word to provide more accurate results, making it suitable for applications where human readability is crucial, like chatbots or Q&A apps.\n</p>","metadata":{}},{"cell_type":"code","source":"from nltk import WordNetLemmatizer","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:12:40.367308Z","iopub.execute_input":"2023-08-26T19:12:40.367761Z","iopub.status.idle":"2023-08-26T19:12:40.373485Z","shell.execute_reply.started":"2023-08-26T19:12:40.367726Z","shell.execute_reply":"2023-08-26T19:12:40.372115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nltk.download('wordnet')","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:13:19.579668Z","iopub.execute_input":"2023-08-26T19:13:19.580119Z","iopub.status.idle":"2023-08-26T19:13:19.588862Z","shell.execute_reply.started":"2023-08-26T19:13:19.580086Z","shell.execute_reply":"2023-08-26T19:13:19.587588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!unzip /usr/share/nltk_data/corpora/wordnet.zip -d /usr/share/nltk_data/corpora/","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:21:39.457128Z","iopub.execute_input":"2023-08-26T19:21:39.457617Z","iopub.status.idle":"2023-08-26T19:21:40.946195Z","shell.execute_reply.started":"2023-08-26T19:21:39.457583Z","shell.execute_reply":"2023-08-26T19:21:40.944789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lemmatizer = WordNetLemmatizer()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:21:45.731944Z","iopub.execute_input":"2023-08-26T19:21:45.732354Z","iopub.status.idle":"2023-08-26T19:21:45.738161Z","shell.execute_reply.started":"2023-08-26T19:21:45.732323Z","shell.execute_reply":"2023-08-26T19:21:45.736748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"q0_lemma = [lemmatizer.lemmatize(word) for word in q0_stopwords]\nq1_lemma = [lemmatizer.lemmatize(word) for word in q1_stopwords]","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:21:46.074292Z","iopub.execute_input":"2023-08-26T19:21:46.074988Z","iopub.status.idle":"2023-08-26T19:21:48.616394Z","shell.execute_reply.started":"2023-08-26T19:21:46.074954Z","shell.execute_reply":"2023-08-26T19:21:48.615174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Tokens before Lemmatization: {q0_stopwords}\")\nprint(f\"Tokens after Lemmatization: {q0_lemma}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:21:50.588284Z","iopub.execute_input":"2023-08-26T19:21:50.589139Z","iopub.status.idle":"2023-08-26T19:21:50.596030Z","shell.execute_reply.started":"2023-08-26T19:21:50.589096Z","shell.execute_reply":"2023-08-26T19:21:50.594594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Tokens before Lemmatization: {q1_stopwords}\")\nprint(f\"Tokens after Lemmatization: {q1_lemma}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:21:52.202015Z","iopub.execute_input":"2023-08-26T19:21:52.202427Z","iopub.status.idle":"2023-08-26T19:21:52.209942Z","shell.execute_reply.started":"2023-08-26T19:21:52.202397Z","shell.execute_reply":"2023-08-26T19:21:52.208138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bag of Words (BoW):\n<p>\n    \nBag of Words (BoW) is a feature extraction technique for textual data. It converts documents into vectors of numeric values so that we can feed them into Machine Learning model. Machine Learning models cannot process text directly that's why we first need to convert to have fixed length input for modeling. As its name suggest, it holds all data in such a way that it disregards the sequence or order of the words. It is mainly concerned with whether words are present in the documents or not, and represent their presence or absence by binary values (0 or 1) or by frequency with respect to each document. It is mainly consists on couple of steps:\n\n<ol>\n        <li>It creates a vocabulary consisting of all unique words present in the provided corpus of data. It sorts each word in the vocabulary in alphabetical order. This vocabulary forms the basis for mapping words to their corresponding positions in the vector representation.</li><br>\n        <li>Each sentence is then represented as a vector where the dimensions correspond to the words in the vocabulary. Words that are present in the sentence receive their respective counts, while words not present in the sentence get a count of zero.\nThe resulting vectors can be sparse, especially if the vocabulary is large and documents are short.</li>\n</ol>\n\nTo minimize the sparsity, we apply some text preprocessing techniques prior to the BoW, such as:\n\n<ol>\n    <li><b>Removing Stopwords:</b> Removing stopwords can help to reduce the sparsity of the BoW representation. Stopwords are frequently occurring words like \"the,\" \"and,\" \"is,\" etc., which might not contribute much to the meaning of the text. By removing them, we can reduce the number of zero-count features in the vector, leading to denser representations.</li><br>\n    <li><b>Stemming or Lemmatization:</b> Applying stemming or lemmatization can also contribute to reduce sparsity. By converting words to their base or root forms, we basically collapse different variations of words into a common representation. This helps in aggregating the counts of similar words, which can reduce the dimensionality of the BoW vectors and further densify the representation.</li>\n</ol>\n\n\n<i>Note:</i>\nWhile the techniques mentioned above can help reduce sparsity, it's important to note that they may not completely eliminate sparsity, especially when dealing with larger vocabularies or more complex documents. Additionally, it's a good practice to consider the trade-off between reducing sparsity and potentially losing some information. Over-aggressive preprocessing might lead to the loss of important nuances in the text.\n\n</p>","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import CountVectorizer","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:22:00.621572Z","iopub.execute_input":"2023-08-26T19:22:00.621996Z","iopub.status.idle":"2023-08-26T19:22:00.628305Z","shell.execute_reply.started":"2023-08-26T19:22:00.621956Z","shell.execute_reply":"2023-08-26T19:22:00.626830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAMPLE_SIZE = 10\nsample_df = raw_train_df.iloc[:SAMPLE_SIZE]","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:22:00.875990Z","iopub.execute_input":"2023-08-26T19:22:00.876429Z","iopub.status.idle":"2023-08-26T19:22:00.882067Z","shell.execute_reply.started":"2023-08-26T19:22:00.876395Z","shell.execute_reply":"2023-08-26T19:22:00.880738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer = CountVectorizer()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:22:02.735165Z","iopub.execute_input":"2023-08-26T19:22:02.735644Z","iopub.status.idle":"2023-08-26T19:22:02.741700Z","shell.execute_reply.started":"2023-08-26T19:22:02.735608Z","shell.execute_reply":"2023-08-26T19:22:02.740089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''fitting count vectorizer will create a vocabulary based on the provided corpus (text) where vocabulary is collection of \\\nunique words present in the corpus'''\nvectorizer.fit(sample_df['question_text'])","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:22:03.108766Z","iopub.execute_input":"2023-08-26T19:22:03.109252Z","iopub.status.idle":"2023-08-26T19:22:03.134354Z","shell.execute_reply.started":"2023-08-26T19:22:03.109218Z","shell.execute_reply":"2023-08-26T19:22:03.133069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# look at the index of each word provided after fitting count vectorizer\nvectorizer.vocabulary_","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:22:05.194110Z","iopub.execute_input":"2023-08-26T19:22:05.194508Z","iopub.status.idle":"2023-08-26T19:22:05.207737Z","shell.execute_reply.started":"2023-08-26T19:22:05.194479Z","shell.execute_reply":"2023-08-26T19:22:05.206392Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# look at the list of unique words extracted from the corpus after fitting count vectorizer, sorted in alphabetical order\nvectorizer.get_feature_names_out()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:22:08.511085Z","iopub.execute_input":"2023-08-26T19:22:08.511722Z","iopub.status.idle":"2023-08-26T19:22:08.519638Z","shell.execute_reply.started":"2023-08-26T19:22:08.511689Z","shell.execute_reply":"2023-08-26T19:22:08.518324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transforming the corpus to have vector representation of each document\nvectors = vectorizer.transform(sample_df['question_text'])","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:22:11.402669Z","iopub.execute_input":"2023-08-26T19:22:11.403101Z","iopub.status.idle":"2023-08-26T19:22:11.408609Z","shell.execute_reply.started":"2023-08-26T19:22:11.403057Z","shell.execute_reply":"2023-08-26T19:22:11.407750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectors","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:22:11.718792Z","iopub.execute_input":"2023-08-26T19:22:11.719557Z","iopub.status.idle":"2023-08-26T19:22:11.726115Z","shell.execute_reply.started":"2023-08-26T19:22:11.719507Z","shell.execute_reply":"2023-08-26T19:22:11.725057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectors.toarray()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:22:11.950493Z","iopub.execute_input":"2023-08-26T19:22:11.951361Z","iopub.status.idle":"2023-08-26T19:22:11.959463Z","shell.execute_reply.started":"2023-08-26T19:22:11.951324Z","shell.execute_reply":"2023-08-26T19:22:11.957782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Actual sample sentence before implementation of Bow:\\n{sample_df['question_text'].values[0]}\")\nprint(f\"\\nSample sentence vector representation after BoW implementation: \\n{vectors[0].toarray()}\")","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:22:13.642451Z","iopub.execute_input":"2023-08-26T19:22:13.642929Z","iopub.status.idle":"2023-08-26T19:22:13.651241Z","shell.execute_reply.started":"2023-08-26T19:22:13.642868Z","shell.execute_reply":"2023-08-26T19:22:13.649761Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Configure Count Vectorizer Parameters:","metadata":{}},{"cell_type":"code","source":"stemmer = SnowballStemmer(language=\"english\")","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:22:15.279803Z","iopub.execute_input":"2023-08-26T19:22:15.280737Z","iopub.status.idle":"2023-08-26T19:22:15.289230Z","shell.execute_reply.started":"2023-08-26T19:22:15.280698Z","shell.execute_reply":"2023-08-26T19:22:15.288163Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tokenize_and_stem(text):\n    return [stemmer.stem(token) for token in word_tokenize(text)]","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:22:16.387700Z","iopub.execute_input":"2023-08-26T19:22:16.389090Z","iopub.status.idle":"2023-08-26T19:22:16.395499Z","shell.execute_reply.started":"2023-08-26T19:22:16.389035Z","shell.execute_reply":"2023-08-26T19:22:16.394631Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# testing tokenization and stemming function\ntokenize_and_stem(\"What is the really dealing here?\")","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:22:17.893206Z","iopub.execute_input":"2023-08-26T19:22:17.893613Z","iopub.status.idle":"2023-08-26T19:22:17.901545Z","shell.execute_reply.started":"2023-08-26T19:22:17.893583Z","shell.execute_reply":"2023-08-26T19:22:17.900076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating count vectorizer object with configuration to incorporate preprocessing techniques together\nvectorizer = CountVectorizer(lowercase=True,\n                             stop_words=en_stopwords, #removing stopwords\n                             tokenizer=tokenize_and_stem, #tokenization and stemming at the same time\n                             max_features=1000 # to limit the vector length (top 1000 vocabs will be considered)\n                             )","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:51:02.527848Z","iopub.execute_input":"2023-08-26T19:51:02.528266Z","iopub.status.idle":"2023-08-26T19:51:02.535243Z","shell.execute_reply.started":"2023-08-26T19:51:02.528235Z","shell.execute_reply":"2023-08-26T19:51:02.533779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# SAMPLE_SIZE = 100_000\n# sample_df = raw_train_df[:SAMPLE_SIZE]\n# sample_df","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:51:04.476127Z","iopub.execute_input":"2023-08-26T19:51:04.476558Z","iopub.status.idle":"2023-08-26T19:51:04.482251Z","shell.execute_reply.started":"2023-08-26T19:51:04.476508Z","shell.execute_reply":"2023-08-26T19:51:04.480438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nvectorizer.fit(raw_train_df['question_text'])","metadata":{"execution":{"iopub.status.busy":"2023-08-26T19:51:04.729219Z","iopub.execute_input":"2023-08-26T19:51:04.729672Z","iopub.status.idle":"2023-08-26T20:01:51.874972Z","shell.execute_reply.started":"2023-08-26T19:51:04.729637Z","shell.execute_reply":"2023-08-26T20:01:51.873618Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(vectorizer.vocabulary_)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:01:51.877563Z","iopub.execute_input":"2023-08-26T20:01:51.878652Z","iopub.status.idle":"2023-08-26T20:01:51.888240Z","shell.execute_reply.started":"2023-08-26T20:01:51.878604Z","shell.execute_reply":"2023-08-26T20:01:51.886614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"vectorizer.get_feature_names_out()[:100]","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:01:51.889985Z","iopub.execute_input":"2023-08-26T20:01:51.890511Z","iopub.status.idle":"2023-08-26T20:01:51.905579Z","shell.execute_reply.started":"2023-08-26T20:01:51.890467Z","shell.execute_reply":"2023-08-26T20:01:51.904106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nmodel_inputs = vectorizer.transform(raw_train_df['question_text'])","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:01:51.908312Z","iopub.execute_input":"2023-08-26T20:01:51.908740Z","iopub.status.idle":"2023-08-26T20:12:32.926172Z","shell.execute_reply.started":"2023-08-26T20:01:51.908706Z","shell.execute_reply":"2023-08-26T20:12:32.924994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_inputs.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:12:32.927376Z","iopub.execute_input":"2023-08-26T20:12:32.927743Z","iopub.status.idle":"2023-08-26T20:12:32.936915Z","shell.execute_reply.started":"2023-08-26T20:12:32.927713Z","shell.execute_reply":"2023-08-26T20:12:32.935695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_inputs","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:12:32.938295Z","iopub.execute_input":"2023-08-26T20:12:32.938708Z","iopub.status.idle":"2023-08-26T20:12:32.952442Z","shell.execute_reply.started":"2023-08-26T20:12:32.938674Z","shell.execute_reply":"2023-08-26T20:12:32.951043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_train_df['question_text'].values[0]","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:12:32.954318Z","iopub.execute_input":"2023-08-26T20:12:32.955030Z","iopub.status.idle":"2023-08-26T20:12:32.967168Z","shell.execute_reply.started":"2023-08-26T20:12:32.954976Z","shell.execute_reply":"2023-08-26T20:12:32.965827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_inputs[0].toarray()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:12:32.969266Z","iopub.execute_input":"2023-08-26T20:12:32.969748Z","iopub.status.idle":"2023-08-26T20:12:32.986048Z","shell.execute_reply.started":"2023-08-26T20:12:32.969706Z","shell.execute_reply":"2023-08-26T20:12:32.984851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\ntest_inputs = vectorizer.transform(raw_test_df['question_text'])","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:12:32.987875Z","iopub.execute_input":"2023-08-26T20:12:32.988232Z","iopub.status.idle":"2023-08-26T20:15:36.637063Z","shell.execute_reply.started":"2023-08-26T20:12:32.988203Z","shell.execute_reply":"2023-08-26T20:15:36.635620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model Training:","metadata":{}},{"cell_type":"code","source":"# Splitting data into train_set & validation_set\nfrom sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:36.643350Z","iopub.execute_input":"2023-08-26T20:15:36.644028Z","iopub.status.idle":"2023-08-26T20:15:36.649445Z","shell.execute_reply.started":"2023-08-26T20:15:36.643979Z","shell.execute_reply":"2023-08-26T20:15:36.648114Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_inputs, valid_inputs, train_target, valid_target = train_test_split(model_inputs, raw_train_df['target'], test_size=0.3, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:36.651244Z","iopub.execute_input":"2023-08-26T20:15:36.651685Z","iopub.status.idle":"2023-08-26T20:15:37.020197Z","shell.execute_reply.started":"2023-08-26T20:15:36.651650Z","shell.execute_reply":"2023-08-26T20:15:37.019070Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_inputs.shape, valid_inputs.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:37.021971Z","iopub.execute_input":"2023-08-26T20:15:37.022375Z","iopub.status.idle":"2023-08-26T20:15:37.031259Z","shell.execute_reply.started":"2023-08-26T20:15:37.022300Z","shell.execute_reply":"2023-08-26T20:15:37.029871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_target.shape, valid_target.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:37.032893Z","iopub.execute_input":"2023-08-26T20:15:37.033361Z","iopub.status.idle":"2023-08-26T20:15:37.045315Z","shell.execute_reply.started":"2023-08-26T20:15:37.033314Z","shell.execute_reply":"2023-08-26T20:15:37.044063Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_target.value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:37.047338Z","iopub.execute_input":"2023-08-26T20:15:37.047769Z","iopub.status.idle":"2023-08-26T20:15:37.073738Z","shell.execute_reply.started":"2023-08-26T20:15:37.047735Z","shell.execute_reply":"2023-08-26T20:15:37.071390Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_target.value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:37.075138Z","iopub.execute_input":"2023-08-26T20:15:37.075703Z","iopub.status.idle":"2023-08-26T20:15:37.095098Z","shell.execute_reply.started":"2023-08-26T20:15:37.075657Z","shell.execute_reply":"2023-08-26T20:15:37.093521Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Logistic Regression Model:","metadata":{}},{"cell_type":"code","source":"from sklearn.linear_model import LogisticRegression","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:37.096576Z","iopub.execute_input":"2023-08-26T20:15:37.097007Z","iopub.status.idle":"2023-08-26T20:15:37.106475Z","shell.execute_reply.started":"2023-08-26T20:15:37.096947Z","shell.execute_reply":"2023-08-26T20:15:37.105206Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_ITER = 1000","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:37.108496Z","iopub.execute_input":"2023-08-26T20:15:37.109257Z","iopub.status.idle":"2023-08-26T20:15:37.120338Z","shell.execute_reply.started":"2023-08-26T20:15:37.109214Z","shell.execute_reply":"2023-08-26T20:15:37.118934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"log_reg_model = LogisticRegression(max_iter=MAX_ITER)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:37.122634Z","iopub.execute_input":"2023-08-26T20:15:37.123609Z","iopub.status.idle":"2023-08-26T20:15:37.134553Z","shell.execute_reply.started":"2023-08-26T20:15:37.123559Z","shell.execute_reply":"2023-08-26T20:15:37.133182Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nlog_reg_model.fit(train_inputs, train_target)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:37.136450Z","iopub.execute_input":"2023-08-26T20:15:37.137575Z","iopub.status.idle":"2023-08-26T20:15:54.238756Z","shell.execute_reply.started":"2023-08-26T20:15:37.137491Z","shell.execute_reply":"2023-08-26T20:15:54.237589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_preds = log_reg_model.predict(valid_inputs)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:54.240287Z","iopub.execute_input":"2023-08-26T20:15:54.240874Z","iopub.status.idle":"2023-08-26T20:15:54.262962Z","shell.execute_reply.started":"2023-08-26T20:15:54.240843Z","shell.execute_reply":"2023-08-26T20:15:54.260084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:54.264766Z","iopub.execute_input":"2023-08-26T20:15:54.265514Z","iopub.status.idle":"2023-08-26T20:15:54.271709Z","shell.execute_reply.started":"2023-08-26T20:15:54.265477Z","shell.execute_reply":"2023-08-26T20:15:54.270293Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_accuracy = accuracy_score(valid_target, valid_preds)\nvalid_precision = precision_score(valid_target, valid_preds)\nvalid_recall = recall_score(valid_target, valid_preds)\nvalid_f1 = f1_score(valid_target, valid_preds)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:54.273346Z","iopub.execute_input":"2023-08-26T20:15:54.273843Z","iopub.status.idle":"2023-08-26T20:15:54.743321Z","shell.execute_reply.started":"2023-08-26T20:15:54.273811Z","shell.execute_reply":"2023-08-26T20:15:54.741958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_accuracy, valid_precision, valid_recall, valid_f1","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:54.744857Z","iopub.execute_input":"2023-08-26T20:15:54.745225Z","iopub.status.idle":"2023-08-26T20:15:54.754306Z","shell.execute_reply.started":"2023-08-26T20:15:54.745195Z","shell.execute_reply":"2023-08-26T20:15:54.752893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:54.755874Z","iopub.execute_input":"2023-08-26T20:15:54.756234Z","iopub.status.idle":"2023-08-26T20:15:54.768834Z","shell.execute_reply.started":"2023-08-26T20:15:54.756204Z","shell.execute_reply":"2023-08-26T20:15:54.767142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_preds = np.random.choice((0, 1), len(valid_target))","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:54.770828Z","iopub.execute_input":"2023-08-26T20:15:54.771284Z","iopub.status.idle":"2023-08-26T20:15:54.784244Z","shell.execute_reply.started":"2023-08-26T20:15:54.771244Z","shell.execute_reply":"2023-08-26T20:15:54.783023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_accuracy = accuracy_score(valid_target, random_preds)\nrandom_f1 = f1_score(valid_target, random_preds)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:54.785996Z","iopub.execute_input":"2023-08-26T20:15:54.786382Z","iopub.status.idle":"2023-08-26T20:15:55.001336Z","shell.execute_reply.started":"2023-08-26T20:15:54.786349Z","shell.execute_reply":"2023-08-26T20:15:55.000197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"random_accuracy, random_f1","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:15:55.002988Z","iopub.execute_input":"2023-08-26T20:15:55.003384Z","iopub.status.idle":"2023-08-26T20:15:55.011490Z","shell.execute_reply.started":"2023-08-26T20:15:55.003351Z","shell.execute_reply":"2023-08-26T20:15:55.010132Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Submission to Kaggle:","metadata":{}},{"cell_type":"code","source":"test_preds = log_reg_model.predict(test_inputs)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:29:44.014851Z","iopub.execute_input":"2023-08-26T20:29:44.015338Z","iopub.status.idle":"2023-08-26T20:29:44.035815Z","shell.execute_reply.started":"2023-08-26T20:29:44.015304Z","shell.execute_reply":"2023-08-26T20:29:44.034512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df['prediction'] = test_preds","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:32:08.807674Z","iopub.execute_input":"2023-08-26T20:32:08.808084Z","iopub.status.idle":"2023-08-26T20:32:08.814054Z","shell.execute_reply.started":"2023-08-26T20:32:08.808045Z","shell.execute_reply":"2023-08-26T20:32:08.812791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df['prediction'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:32:21.259815Z","iopub.execute_input":"2023-08-26T20:32:21.260203Z","iopub.status.idle":"2023-08-26T20:32:21.273879Z","shell.execute_reply.started":"2023-08-26T20:32:21.260175Z","shell.execute_reply":"2023-08-26T20:32:21.272560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2023-08-26T20:32:25.443862Z","iopub.execute_input":"2023-08-26T20:32:25.444263Z","iopub.status.idle":"2023-08-26T20:32:26.649139Z","shell.execute_reply.started":"2023-08-26T20:32:25.444232Z","shell.execute_reply":"2023-08-26T20:32:26.647766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}