{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Exploratory Data Analysis for Jigsaw Miltilingual Toxic Comment Classification","metadata":{}},{"cell_type":"code","source":"# Import the Libraries\nimport os\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport random\nimport re\nimport string\nimport nltk\nfrom nltk import FreqDist\nfrom nltk.corpus import stopwords\nfrom nltk.stem import WordNetLemmatizer\nfrom collections import Counter\nfrom textblob import TextBlob","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-02T06:12:04.288368Z","iopub.execute_input":"2023-05-02T06:12:04.289111Z","iopub.status.idle":"2023-05-02T06:12:05.463289Z","shell.execute_reply.started":"2023-05-02T06:12:04.289068Z","shell.execute_reply":"2023-05-02T06:12:05.462228Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"nltk.download('vader_lexicon')\nfrom nltk.sentiment import SentimentIntensityAnalyzer\n\nnltk.download('stopwords')\nnltk.download('wordnet')","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:12:05.466436Z","iopub.execute_input":"2023-05-02T06:12:05.467472Z","iopub.status.idle":"2023-05-02T06:12:05.687825Z","shell.execute_reply.started":"2023-05-02T06:12:05.467435Z","shell.execute_reply":"2023-05-02T06:12:05.686550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the Dataset","metadata":{}},{"cell_type":"code","source":"# Read all train .csv files\ntoxic_comment_processed_seqlen = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train-processed-seqlen128.csv')\ntoxic_comment = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv')\nunintended_bias_preprocessed = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train-processed-seqlen128.csv')\nunintended_bias = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv')\n\ntoxic_comment_processed_seqlen.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:12:05.691806Z","iopub.execute_input":"2023-05-02T06:12:05.692089Z","iopub.status.idle":"2023-05-02T06:14:00.314569Z","shell.execute_reply.started":"2023-05-02T06:12:05.692063Z","shell.execute_reply":"2023-05-02T06:14:00.313507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"toxic_comment.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:00.315983Z","iopub.execute_input":"2023-05-02T06:14:00.317869Z","iopub.status.idle":"2023-05-02T06:14:00.331361Z","shell.execute_reply.started":"2023-05-02T06:14:00.317828Z","shell.execute_reply":"2023-05-02T06:14:00.330416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unintended_bias_preprocessed.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:00.333617Z","iopub.execute_input":"2023-05-02T06:14:00.334245Z","iopub.status.idle":"2023-05-02T06:14:00.366521Z","shell.execute_reply.started":"2023-05-02T06:14:00.334208Z","shell.execute_reply":"2023-05-02T06:14:00.365646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unintended_bias.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:00.367887Z","iopub.execute_input":"2023-05-02T06:14:00.368325Z","iopub.status.idle":"2023-05-02T06:14:00.391164Z","shell.execute_reply.started":"2023-05-02T06:14:00.368290Z","shell.execute_reply":"2023-05-02T06:14:00.390082Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"To understand these datasets better, we should start by exploring the distribution of toxicity levels and their subcategories. Analyze the comment text to identify patterns or specific words that might be indicative of toxic behavior. We can also investigate the relationships between different subcategories to identify possible correlations. Ultimately, we will use this data to train a machine learning model that can accurately predict the toxicity level of unseen comments.","metadata":{}},{"cell_type":"code","source":"# Calculate the percentage of toxic comments in both datasets\ntoxic_percentage = toxic_comment['toxic'].value_counts(normalize=True) * 100\nunintended_bias_toxic_percentage = (unintended_bias['toxic'] > 0.5).value_counts(normalize=True) * 100","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:00.392961Z","iopub.execute_input":"2023-05-02T06:14:00.394009Z","iopub.status.idle":"2023-05-02T06:14:00.424383Z","shell.execute_reply.started":"2023-05-02T06:14:00.393941Z","shell.execute_reply":"2023-05-02T06:14:00.423499Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Let us print the result to see the disrtribution\nprint(toxic_percentage)","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:00.425736Z","iopub.execute_input":"2023-05-02T06:14:00.426070Z","iopub.status.idle":"2023-05-02T06:14:00.432377Z","shell.execute_reply.started":"2023-05-02T06:14:00.426035Z","shell.execute_reply":"2023-05-02T06:14:00.430736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(unintended_bias_toxic_percentage)","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:00.439232Z","iopub.execute_input":"2023-05-02T06:14:00.439591Z","iopub.status.idle":"2023-05-02T06:14:00.445316Z","shell.execute_reply.started":"2023-05-02T06:14:00.439555Z","shell.execute_reply":"2023-05-02T06:14:00.444254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, we want to plot the distribution of toxic comments in both datasets.","metadata":{}},{"cell_type":"code","source":"# Plot the distribution of toxic comments in both datasets\nplt.figure(figsize=(16, 5))\n\nplt.subplot(1, 2, 1)\ntoxic_percentage.plot(kind='bar', color=['blue', 'orange'])\nplt.title('Toxic Comment Distribution (jigsaw-toxic-comment-train.csv)')\nplt.xlabel('Toxic')\nplt.ylabel('Percentage')\nplt.xticks([0, 1], ['Non-toxic', 'Toxic'], rotation=0)\n\nplt.subplot(1, 2, 2)\nunintended_bias_toxic_percentage.plot(kind='bar', color=['blue', 'orange'])\nplt.title('Toxic Comment Distribution (jigsaw-unintended-bias-train.csv)')\nplt.xlabel('Toxic')\nplt.ylabel('Percentage')\nplt.xticks([0, 1], ['Non-toxic', 'Toxic'], rotation=0)\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:00.446890Z","iopub.execute_input":"2023-05-02T06:14:00.447627Z","iopub.status.idle":"2023-05-02T06:14:00.857591Z","shell.execute_reply.started":"2023-05-02T06:14:00.447575Z","shell.execute_reply":"2023-05-02T06:14:00.856643Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Text Length Analysis","metadata":{}},{"cell_type":"markdown","source":"Now, this analysis might sounds like strange, but we do might get an insight for analysing the length of the text. For example, by comparing the distribution of text lengths for toxic and non-toxic comments, we may find that toxic comments tend to be shorter or longer than non-toxic comments. Such patterns can be informative when building your model.","metadata":{}},{"cell_type":"code","source":"# Calculate comment length in characters\ntoxic_comment['char_length'] = toxic_comment['comment_text'].apply(len)\n\n# Calculate comment length in words\ntoxic_comment['word_length'] = toxic_comment['comment_text'].apply(lambda x: len(x.split()))","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:00.859256Z","iopub.execute_input":"2023-05-02T06:14:00.859619Z","iopub.status.idle":"2023-05-02T06:14:01.852390Z","shell.execute_reply.started":"2023-05-02T06:14:00.859567Z","shell.execute_reply":"2023-05-02T06:14:01.851300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot character length distribution\nplt.figure(figsize=(12, 6))\nsns.histplot(data=toxic_comment, x='char_length', hue='toxic', bins=100, common_norm=False)\nplt.title('Character Length Distribution by Toxicity')\nplt.xlabel('Character Length')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:01.854330Z","iopub.execute_input":"2023-05-02T06:14:01.854770Z","iopub.status.idle":"2023-05-02T06:14:02.676485Z","shell.execute_reply.started":"2023-05-02T06:14:01.854729Z","shell.execute_reply":"2023-05-02T06:14:02.675520Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot word length distribution\nplt.figure(figsize=(12, 6))\nsns.histplot(data=toxic_comment, x='word_length', hue='toxic', bins=100, common_norm=False)\nplt.title('Word Length Distribution by Toxicity')\nplt.xlabel('Word Length')\nplt.ylabel('Frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:02.678157Z","iopub.execute_input":"2023-05-02T06:14:02.678505Z","iopub.status.idle":"2023-05-02T06:14:03.337235Z","shell.execute_reply.started":"2023-05-02T06:14:02.678469Z","shell.execute_reply":"2023-05-02T06:14:03.335857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The distribution itself seems like it's more right-skewed for both toxic and non-toxic comment. But, there are some outliers as well, which kinda looks like the outliers is for the non-toxic comment. Does this mean anything? Well, maybe, since we can make an assumption if the toxic comment tends to be shorter than the non toxic comments, but we can't really depends on that since the non toxic comment also has right skewed distribution.\n\nThe question that I want to ask is, are we interested with words with more length? To answer it, let's visualize the distribution once more, but we will only see from **> 2000 word length**.","metadata":{}},{"cell_type":"code","source":"# Create a new column 'is_toxic' to identify toxic and non-toxic comments\ntoxic_comment['is_toxic'] = toxic_comment['toxic'] == 1","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:03.338907Z","iopub.execute_input":"2023-05-02T06:14:03.339613Z","iopub.status.idle":"2023-05-02T06:14:03.345589Z","shell.execute_reply.started":"2023-05-02T06:14:03.339560Z","shell.execute_reply":"2023-05-02T06:14:03.344472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Plot the distribution of character length for toxic and non-toxic comments (from 2000 characters)\nplt.figure(figsize=(12, 6))\nsns.histplot(data=toxic_comment[toxic_comment['char_length'] > 2000], x='char_length', hue='is_toxic', bins=50, kde=True)\nplt.title('Distribution of Character Length for Toxic and Non-Toxic Comments (from 2000 characters)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:03.347416Z","iopub.execute_input":"2023-05-02T06:14:03.347779Z","iopub.status.idle":"2023-05-02T06:14:03.845658Z","shell.execute_reply.started":"2023-05-02T06:14:03.347743Z","shell.execute_reply":"2023-05-02T06:14:03.844628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, this is interesting. See how in every bins except the last one the majority of comments are non-toxic? But then, all of a sudden, the toxic comments become majority in the last bins (around 5000, more or less). Before we continue our discussion, let's also plot the distribution for the word length.","metadata":{}},{"cell_type":"code","source":"# Plot the distribution of word length for toxic and non-toxic comments (From 900 words)\nplt.figure(figsize=(12, 6))\nsns.histplot(data=toxic_comment[toxic_comment['word_length'] > 900], x='word_length', hue='is_toxic', bins=50, kde=True)\nplt.title('Distribution of Word Length for Toxic and Non-Toxic Comments (from 900 words)')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:03.847373Z","iopub.execute_input":"2023-05-02T06:14:03.848115Z","iopub.status.idle":"2023-05-02T06:14:04.302694Z","shell.execute_reply.started":"2023-05-02T06:14:03.848075Z","shell.execute_reply":"2023-05-02T06:14:04.301684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Based on our observation, it appears that although non-toxic comments generally have a higher frequency across most bins, there is a higher proportion of toxic comments in the last bin (around 5000 words or more). This could suggest that extremely lengthy comments have a higher likelihood of containing toxic content.\n\nHere are some possible explanations for this observation:\n1. Trolls and spammers: Users who deliberately post offensive or provocative content may also be more likely to create excessively long comments to gain attention or disrupt the conversation.\n2. Venting and ranting: Users who are expressing strong negative emotions might write more at length to describe their feelings, experiences, or thoughts. These longer comments could potentially contain toxic language as a result.\n3. Heated debates or arguments: Long comments might emerge from extended back-and-forth discussions where users passionately argue their points. In these situations, emotions can run high, and the language used may become toxic.\n\nHowever, it's essential to note that this observation alone does not necessarily imply causality, and further analysis would be needed to determine any significant relationship between comment length and toxicity. For instance, you could investigate correlations between comment length and toxicity or analyze the content of these lengthy comments to gain a deeper understanding of the context in which they appear.\n\nNow, to see our answer, let's just print the toxic comment that has many words.","metadata":{}},{"cell_type":"markdown","source":"Below is the code that I made to print out the toxic comment that has > 4500 words. If you want to, you can copy this notebook and try different threshold and number of random outliers that will be printed.","metadata":{}},{"cell_type":"code","source":"# Set the character length threshold\nchar_length_threshold = 4500\n\n# Find the outliers\noutliers = toxic_comment[toxic_comment['char_length'] > char_length_threshold]\n\n# Print the number of outliers\nprint(f\"Number of outliers: {len(outliers)}\")\n\n# Randomly select a subset of outliers\nnum_random_outliers = 1  # Change this value to print more or fewer outliers\nrandom_indices = random.sample(range(len(outliers)), num_random_outliers)\nrandom_outliers = outliers.iloc[random_indices]\n\n# Print the randomly selected outliers and their text\nfor idx, outlier in random_outliers.iterrows():\n    print(f\"\\nIndex: {idx}\\nComment: {outlier['comment_text']}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:04.304450Z","iopub.execute_input":"2023-05-02T06:14:04.304868Z","iopub.status.idle":"2023-05-02T06:14:04.314484Z","shell.execute_reply.started":"2023-05-02T06:14:04.304827Z","shell.execute_reply":"2023-05-02T06:14:04.313342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"If you want to, you can go deeper and check if you gain any more insight for the length of text analysis.","metadata":{}},{"cell_type":"markdown","source":"# Word Frequency Analysis\n\nThe second thing we want to analyze is the frequency of the word. This will be very helpful, since we can detect spam and trolls, and look at the pattern on what words frequently used for toxic comment. To do this, we want tokenize the words. When we tokenize the words in a sentence or a paragraph, we essentially separate each word, taking into account punctuation, whitespace, and other delimiters. This process allows us to easily count word occurrences, perform frequency analysis, or apply other NLP techniques to the tokens.\n\nSo, the plan here is, first, we want to see what words are most frequent. Then, we might want to seperately analyzing most common words for toxic and non-toxic comments.","metadata":{}},{"cell_type":"code","source":"# Tokenize words in the comments\ntoxic_comment['tokenized_words'] = toxic_comment['comment_text'].apply(lambda x: nltk.word_tokenize(x.lower()))","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:14:04.316225Z","iopub.execute_input":"2023-05-02T06:14:04.316646Z","iopub.status.idle":"2023-05-02T06:16:15.710824Z","shell.execute_reply.started":"2023-05-02T06:14:04.316593Z","shell.execute_reply":"2023-05-02T06:16:15.709788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a list of all words in the dataset\nall_words = []\nfor words in toxic_comment['tokenized_words']:\n    all_words.extend(words)","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:16:15.712533Z","iopub.execute_input":"2023-05-02T06:16:15.712921Z","iopub.status.idle":"2023-05-02T06:16:16.009370Z","shell.execute_reply.started":"2023-05-02T06:16:15.712883Z","shell.execute_reply":"2023-05-02T06:16:16.008308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the frequency distribution of words\nfreq_dist = FreqDist(all_words)\n\n# Plot the top 20 most common words\nplt.figure(figsize=(12, 6))\nfreq_dist.plot(20, cumulative=False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:16:16.010828Z","iopub.execute_input":"2023-05-02T06:16:16.011281Z","iopub.status.idle":"2023-05-02T06:16:29.175903Z","shell.execute_reply.started":"2023-05-02T06:16:16.011245Z","shell.execute_reply":"2023-05-02T06:16:29.174685Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We can see the plot above, but that plot is as useful as a paper umbrella in a hurricane. What I mean here is, of course we know that the words \"the\", \"you\", \"and\", and any general things will be the most common words. The second problem is the special character like \",\", \".\", \";\", and so on.\n\nTo solve this issue, we can use a common English stopwords and isalpha function when we tokenize the words.","metadata":{}},{"cell_type":"code","source":"english_stopwords = set(stopwords.words('english'))","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:16:29.177476Z","iopub.execute_input":"2023-05-02T06:16:29.178645Z","iopub.status.idle":"2023-05-02T06:16:29.186353Z","shell.execute_reply.started":"2023-05-02T06:16:29.178581Z","shell.execute_reply":"2023-05-02T06:16:29.185200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tokenize words in the comments and exclude stopwords and non-alphabetic tokens\ntoxic_comment['tokenized_words'] = toxic_comment['comment_text'].apply(\n    lambda x: [word for word in nltk.word_tokenize(x.lower())\n               if word not in english_stopwords and word.isalpha()]\n    )","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:16:29.187795Z","iopub.execute_input":"2023-05-02T06:16:29.188769Z","iopub.status.idle":"2023-05-02T06:18:44.598399Z","shell.execute_reply.started":"2023-05-02T06:16:29.188719Z","shell.execute_reply":"2023-05-02T06:18:44.597288Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create a list of all words in the dataset\nall_words = []\nfor words in toxic_comment['tokenized_words']:\n    all_words.extend(words)","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:18:44.600866Z","iopub.execute_input":"2023-05-02T06:18:44.601638Z","iopub.status.idle":"2023-05-02T06:18:45.116354Z","shell.execute_reply.started":"2023-05-02T06:18:44.601583Z","shell.execute_reply":"2023-05-02T06:18:45.115292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Calculate the frequency distribution of words\nfreq_dist = FreqDist(all_words)\n\n# Plot the top 20 most common words\nplt.figure(figsize=(12, 6))\nfreq_dist.plot(20, cumulative=False)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:18:45.117767Z","iopub.execute_input":"2023-05-02T06:18:45.119622Z","iopub.status.idle":"2023-05-02T06:18:50.589120Z","shell.execute_reply.started":"2023-05-02T06:18:45.119563Z","shell.execute_reply":"2023-05-02T06:18:50.588052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now this look much better. But still, the insight we gain from the plot above is just that we know the dataset we're dealing with is a comment from article.\n\nSo, the next step we want to do is to seperate the toxic and non-toxic frequent words.","metadata":{}},{"cell_type":"code","source":"def plot_word_frequency(words, title):\n    word_freq = Counter(words)\n    top_words = word_freq.most_common(20)\n    word, frequency = zip(*top_words)\n\n    plt.figure(figsize=(10, 6))\n    plt.bar(word, frequency)\n    plt.title(title)\n    plt.xlabel('Words')\n    plt.ylabel('Frequency')\n    plt.xticks(rotation=45)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:18:50.590732Z","iopub.execute_input":"2023-05-02T06:18:50.591095Z","iopub.status.idle":"2023-05-02T06:18:50.598953Z","shell.execute_reply.started":"2023-05-02T06:18:50.591059Z","shell.execute_reply":"2023-05-02T06:18:50.596977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"toxic_comments = toxic_comment[toxic_comment['toxic'] == 1]['comment_text']\nnon_toxic_comments = toxic_comment[toxic_comment['toxic'] == 0]['comment_text']\n\ntoxic_words = []\nnon_toxic_words = []\n\nfor comment in toxic_comments:\n    toxic_words.extend([word.lower() for word in nltk.word_tokenize(comment) if word.lower() not in english_stopwords and word.isalpha()])\n\nfor comment in non_toxic_comments:\n    non_toxic_words.extend([word.lower() for word in nltk.word_tokenize(comment) if word.lower() not in english_stopwords and word.isalpha()])\n\nplot_word_frequency(toxic_words, \"Top 20 Most Common Words in Toxic Comments (excluding stopwords and special characters)\")\nplot_word_frequency(non_toxic_words, \"Top 20 Most Common Words in Non-Toxic Comments (excluding stopwords and special characters)\")","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:18:50.600795Z","iopub.execute_input":"2023-05-02T06:18:50.601667Z","iopub.status.idle":"2023-05-02T06:21:09.068583Z","shell.execute_reply.started":"2023-05-02T06:18:50.601628Z","shell.execute_reply":"2023-05-02T06:21:09.067575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now we see the differences. Even though there are some non-related toxic words like \"wikipedia\" and \"people\", toxic words contains way more offensive words.","metadata":{}},{"cell_type":"markdown","source":"# N-grams Analysis\n\nAnalyzing word frequency is one thing, but analyzing it \"sequentially\" is another thing. We might want to see the continuation of the words. Yes, there are certain words that we don't need to analyze the continuation, especially the offensive words, but we're still curious how the word \"people\" and \"wikipedia\" are still in most toxic category. So, what we're going to do now is to analyze the N-grams analysis. Here, I'm going to use Bi-grams.\n\nBi-grams are sequences of two consecutive words, and they can provide useful insights into the relationships between words in a given text. Analyzing bi-grams can help us understand how words are commonly paired together, which can be especially informative when examining text data.\n\nBy analyzing bi-grams in toxic and non-toxic comments, we aim to identify common word pairs that appear in each category. This can help us understand the differences in language patterns between toxic and non-toxic comments. The insights gained can be valuable when building models to identify and classify toxic comments, as these patterns can be used as features for machine learning algorithms.","metadata":{}},{"cell_type":"code","source":"def plot_ngrams_frequency(ngrams, title):\n    ngram_freq = Counter(ngrams)\n    top_ngrams = ngram_freq.most_common(20)\n    ngram, frequency = zip(*top_ngrams)\n    ngram = [' '.join(gram) for gram in ngram]\n\n    plt.figure(figsize=(12, 6))\n    plt.bar(ngram, frequency)\n    plt.title(title)\n    plt.xlabel('N-grams')\n    plt.ylabel('Frequency')\n    plt.xticks(rotation=45)\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:21:09.074226Z","iopub.execute_input":"2023-05-02T06:21:09.074530Z","iopub.status.idle":"2023-05-02T06:21:09.081012Z","shell.execute_reply.started":"2023-05-02T06:21:09.074501Z","shell.execute_reply":"2023-05-02T06:21:09.079795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"toxic_comments = toxic_comment[toxic_comment['toxic'] == 1]['comment_text']\nnon_toxic_comments = toxic_comment[toxic_comment['toxic'] == 0]['comment_text']","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:21:09.082532Z","iopub.execute_input":"2023-05-02T06:21:09.082989Z","iopub.status.idle":"2023-05-02T06:21:09.124967Z","shell.execute_reply.started":"2023-05-02T06:21:09.082952Z","shell.execute_reply":"2023-05-02T06:21:09.123996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def generate_ngrams(text, n):\n    tokens = [token.lower() for token in nltk.word_tokenize(text) if token.lower() not in english_stopwords and token.isalpha()]\n    ngrams = list(nltk.ngrams(tokens, n)) if len(tokens) >= n else []\n    return ngrams","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:21:09.126512Z","iopub.execute_input":"2023-05-02T06:21:09.126905Z","iopub.status.idle":"2023-05-02T06:21:09.133282Z","shell.execute_reply.started":"2023-05-02T06:21:09.126869Z","shell.execute_reply":"2023-05-02T06:21:09.132264Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"toxic_ngrams = []\nnon_toxic_ngrams = []\n\nfor comment in toxic_comments:\n    toxic_ngrams.extend(generate_ngrams(comment, 2))\n\nfor comment in non_toxic_comments:\n    non_toxic_ngrams.extend(generate_ngrams(comment, 2))\n\nplot_ngrams_frequency(toxic_ngrams, \"Top 20 Most Common Bi-grams in Toxic Comments (excluding stopwords)\")\nplot_ngrams_frequency(non_toxic_ngrams, \"Top 20 Most Common Bi-grams in Non-Toxic Comments (excluding stopwords)\")","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:21:09.134565Z","iopub.execute_input":"2023-05-02T06:21:09.135707Z","iopub.status.idle":"2023-05-02T06:23:31.836594Z","shell.execute_reply.started":"2023-05-02T06:21:09.135671Z","shell.execute_reply":"2023-05-02T06:23:31.835463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now, it's more obvious what the toxic words are. Most of them are repeatable words. We can speculate that most repeatable words are in toxic category, since I can't image someone comments like \"happy happy\". We could use this information to improve our model's ability to recognize toxic language.","metadata":{}},{"cell_type":"markdown","source":"# Sentiment Analysis\n\nThe next analysis we want to do is Sentiment Analysis. Sentiment Analysis is being used so we can understand if a piece of text is positive, negative, or neutral.\n\nIn this context, which contains comments that are labeled as toxic and non-toxic, performing sentiment analysis can provide additional insights into the emotional tone of the comments. By comparing the sentiment scores of toxic and non-toxic comments, you can explore whether there is a relationship between the sentiment and toxicity of a comment.\n\nI used texblob to do sentiment analysis, since it's faster and more simple. I tried to use sentiment analyzer from nltk, but it's just so slow. If wanted to, or if you have accelerator in kaggle, you might want to try it. The sentiment scores you obtained using TextBlob range from -1 (most negative sentiment) to 1 (most positive sentiment), with 0 representing neutral sentiment.","metadata":{}},{"cell_type":"code","source":"def get_sentiment_scores(text):\n    tb = TextBlob(text)\n    return {'compound': tb.sentiment.polarity}","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:23:31.837983Z","iopub.execute_input":"2023-05-02T06:23:31.838892Z","iopub.status.idle":"2023-05-02T06:23:31.844393Z","shell.execute_reply.started":"2023-05-02T06:23:31.838852Z","shell.execute_reply":"2023-05-02T06:23:31.843308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The code below is for sentiment analysis from nltk. Uncomment it if you want to use it instead.","metadata":{}},{"cell_type":"code","source":"# def get_sentiment_scores(text):\n#     sia = SentimentIntensityAnalyzer()\n#     return sia.polarity_scores(text)","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:23:31.845724Z","iopub.execute_input":"2023-05-02T06:23:31.846691Z","iopub.status.idle":"2023-05-02T06:23:31.858274Z","shell.execute_reply.started":"2023-05-02T06:23:31.846650Z","shell.execute_reply":"2023-05-02T06:23:31.857297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"toxic_comment['sentiment'] = toxic_comment['comment_text'].apply(get_sentiment_scores)","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:23:31.859781Z","iopub.execute_input":"2023-05-02T06:23:31.860210Z","iopub.status.idle":"2023-05-02T06:25:30.460036Z","shell.execute_reply.started":"2023-05-02T06:23:31.860173Z","shell.execute_reply":"2023-05-02T06:25:30.458964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"toxic_comments_sentiment = toxic_comment[toxic_comment['toxic'] == 1]['sentiment']\nnon_toxic_comments_sentiment = toxic_comment[toxic_comment['toxic'] == 0]['sentiment']\n\navg_sentiment_toxic = toxic_comments_sentiment.apply(lambda x: x['compound']).mean()\navg_sentiment_non_toxic = non_toxic_comments_sentiment.apply(lambda x: x['compound']).mean()\n\nprint(\"Average sentiment score for toxic comments:\", avg_sentiment_toxic)\nprint(\"Average sentiment score for non-toxic comments:\", avg_sentiment_non_toxic)","metadata":{"execution":{"iopub.status.busy":"2023-05-02T06:25:30.461652Z","iopub.execute_input":"2023-05-02T06:25:30.462025Z","iopub.status.idle":"2023-05-02T06:25:30.635138Z","shell.execute_reply.started":"2023-05-02T06:25:30.461987Z","shell.execute_reply":"2023-05-02T06:25:30.633943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the result that we've got, now we know that:\n- The average sentiment score for toxic comments is approximately -0.123, which indicates that toxic comments, on average, have a somewhat negative sentiment. This is expected because toxic comments typically contain negative language, insults, or offensive content.\n- The average sentiment score for non-toxic comments is approximately 0.088, which indicates that non-toxic comments, on average, have a slightly positive sentiment. This is expected because non-toxic comments generally include more neutral or positive language, as they don't contain the negative elements present in toxic comments.\n\nIn our case, the average sentiment scores for toxic and non-toxic comments don't cover the entire range, but they still provide insights into the difference in sentiment between the two types of comments.\n\nThe average sentiment scores for toxic and non-toxic comments in our dataset indicate that there is a noticeable difference in sentiment between the two groups. Toxic comments have a somewhat negative average sentiment, while non-toxic comments have a slightly positive average sentiment. This distinction can be useful in understanding the overall sentiment differences between toxic and non-toxic comments.","metadata":{}}]}