{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":19018,"databundleVersionId":2703900,"sourceType":"competition"}],"dockerImageVersionId":30804,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:32.366555Z","iopub.execute_input":"2024-12-14T06:26:32.366998Z","iopub.status.idle":"2024-12-14T06:26:32.377539Z","shell.execute_reply.started":"2024-12-14T06:26:32.366964Z","shell.execute_reply":"2024-12-14T06:26:32.376190Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:32.380256Z","iopub.execute_input":"2024-12-14T06:26:32.380784Z","iopub.status.idle":"2024-12-14T06:26:32.393654Z","shell.execute_reply.started":"2024-12-14T06:26:32.380729Z","shell.execute_reply":"2024-12-14T06:26:32.392390Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:32.395435Z","iopub.execute_input":"2024-12-14T06:26:32.395751Z","iopub.status.idle":"2024-12-14T06:26:32.407364Z","shell.execute_reply.started":"2024-12-14T06:26:32.395722Z","shell.execute_reply":"2024-12-14T06:26:32.405933Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:32.408988Z","iopub.execute_input":"2024-12-14T06:26:32.409512Z","iopub.status.idle":"2024-12-14T06:26:34.001726Z","shell.execute_reply.started":"2024-12-14T06:26:32.409428Z","shell.execute_reply":"2024-12-14T06:26:34.000307Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:34.005369Z","iopub.execute_input":"2024-12-14T06:26:34.005761Z","iopub.status.idle":"2024-12-14T06:26:34.019735Z","shell.execute_reply.started":"2024-12-14T06:26:34.005726Z","shell.execute_reply":"2024-12-14T06:26:34.018570Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check for missing data\nmissing_data = df.isnull().sum()\nprint(\"Missing Data:\\n\", missing_data)\n\nif missing_data.sum() == 0:\n    print(\"No missing data in the dataset.\")\nelse:\n    print(\"There is missing data in the dataset.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:34.021337Z","iopub.execute_input":"2024-12-14T06:26:34.021789Z","iopub.status.idle":"2024-12-14T06:26:34.076948Z","shell.execute_reply.started":"2024-12-14T06:26:34.021732Z","shell.execute_reply":"2024-12-14T06:26:34.075379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Calculate message lengths based on the comment_text column\ndf['message_length'] = df['comment_text'].str.len()\n\n# Plot message length vs. frequency\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(10, 6))\nplt.hist(df['message_length'], bins=60, color='green', alpha=0.7, label='All Comments')\nplt.xlabel('Message Length')\nplt.ylabel('Frequency')\nplt.title('Message Length for Training Data')\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:34.078293Z","iopub.execute_input":"2024-12-14T06:26:34.078617Z","iopub.status.idle":"2024-12-14T06:26:34.578164Z","shell.execute_reply.started":"2024-12-14T06:26:34.078587Z","shell.execute_reply":"2024-12-14T06:26:34.576952Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# Calculate message lengths\ndf['message_length'] = df['comment_text'].str.len()\n\n# Determine if a comment is dirty or clean\ndf['is_dirty'] = df[['toxic', 'severe_toxic', 'obscene', 'threat', 'insult', 'identity_hate']].sum(axis=1) > 0\n\n# Separate clean and dirty comments\nclean_comments = df[df['is_dirty'] == False]['message_length']\ndirty_comments = df[df['is_dirty'] == True]['message_length']\n\n# Plot message length vs. frequency for clean and dirty comments\nplt.figure(figsize=(12, 6))\nplt.hist(clean_comments, bins=100, color='blue', alpha=0.5, label='Clean Comments')\nplt.hist(dirty_comments, bins=100, color='red', alpha=0.5, label='Dirty Comments')\nplt.xlabel('Message Length')\nplt.ylabel('Frequency')\nplt.title('Message Length Distribution: Clean vs Dirty Comments')\nplt.legend()\nplt.xticks(ticks=range(0, 2000, 500))  # Adjusting x-axis ticks for better scaling\nplt.grid(axis='y', linestyle='--', alpha=0.7)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:34.579508Z","iopub.execute_input":"2024-12-14T06:26:34.579839Z","iopub.status.idle":"2024-12-14T06:26:35.271144Z","shell.execute_reply.started":"2024-12-14T06:26:34.579808Z","shell.execute_reply":"2024-12-14T06:26:35.270016Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\n\n# Calculate the number of occurrences for each tag\ntag_counts = {\n    'toxic': df['toxic'].sum(),\n    'severe_toxic': df['severe_toxic'].sum(),\n    'obscene': df['obscene'].sum(),\n    'threat': df['threat'].sum(),\n    'insult': df['insult'].sum(),\n    'identity_hate': df['identity_hate'].sum(),\n    'clean': (df[['toxic', 'severe_toxic', 'obscene', 'threat', 'insult', 'identity_hate']].sum(axis=1) == 0).sum()\n}\n\n# Convert the dictionary into a DataFrame for easy plotting\ntag_counts_df = pd.DataFrame(list(tag_counts.items()), columns=['Type', 'Occurrences'])\n\n# Plot the bar chart\nplt.figure(figsize=(10, 6))\nplt.bar(tag_counts_df['Type'], tag_counts_df['Occurrences'], color=['blue', 'orange', 'green', 'red', 'purple', 'brown', 'pink'])\nplt.xlabel('Type')\nplt.ylabel('Occurrences')\nplt.title('Number of Tags')\n# Annotating the bar values\nfor i, val in enumerate(tag_counts_df['Occurrences']):\n    plt.text(i, val + 1000, f'{val:.1f}', ha='center', fontsize=10)\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:35.272519Z","iopub.execute_input":"2024-12-14T06:26:35.272843Z","iopub.status.idle":"2024-12-14T06:26:35.534783Z","shell.execute_reply.started":"2024-12-14T06:26:35.272810Z","shell.execute_reply":"2024-12-14T06:26:35.533650Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tag_columns = ['toxic', 'severe_toxic', 'obscene', 'threat', 'insult', 'identity_hate']\ndf['num_tags'] = df[tag_columns].sum(axis=1)\n\n# Count occurrences of each number of tags\ntag_counts = df['num_tags'].value_counts().sort_index()\n\n# Plot the bar chart\nplt.figure(figsize=(10, 6))\nbar_colors = plt.cm.tab20(range(len(tag_counts)))  # Optional: Colorful bars\ntag_counts.plot(kind='bar', color=bar_colors)\nplt.title('Number of Multiple Tags per Comment')\nplt.xlabel('Number of Tags')\nplt.ylabel('Occurrences')\n\n# Annotate bar plot with numbers\nfor index, value in enumerate(tag_counts):\n    plt.text(index, value + 500, str(value), ha='center', va='bottom', fontsize=10)\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:35.536518Z","iopub.execute_input":"2024-12-14T06:26:35.536995Z","iopub.status.idle":"2024-12-14T06:26:35.804261Z","shell.execute_reply.started":"2024-12-14T06:26:35.536948Z","shell.execute_reply":"2024-12-14T06:26:35.802991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to calculate the percentage of unique words in a comment\ndef percent_unique_words(text):\n    words = text.split()  # Split comment into words\n    if len(words) == 0:\n        return 0\n    unique_words = set(words)\n    return len(unique_words) / len(words) * 100\n\n# Add a new column for the percentage of unique words\ndf[\"percent_unique_words\"] = df[\"comment_text\"].apply(percent_unique_words)\n\n# Split the data into dirty and clean based on the label\ndirty_comments = df[df['is_dirty'] == 1][\"percent_unique_words\"]\nclean_comments = df[df['is_dirty'] == 0][\"percent_unique_words\"]\n\n# Plot the distributions using seaborn\nplt.figure(figsize=(10, 6))\nsns.kdeplot(dirty_comments, fill=True, color=\"red\", label=\"Dirty\")\nsns.kdeplot(clean_comments, fill=True, color=\"blue\", label=\"Clean\")\n\n# Add titles and labels\nplt.title(\"Percentage of Unique Words of Total Words in Comments\", fontsize=14)\nplt.xlabel(\"Percent Unique Words\", fontsize=12)\nplt.ylabel(\"Number of Occurrences\", fontsize=12)\nplt.legend()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:35.806265Z","iopub.execute_input":"2024-12-14T06:26:35.806681Z","iopub.status.idle":"2024-12-14T06:26:39.824498Z","shell.execute_reply.started":"2024-12-14T06:26:35.806643Z","shell.execute_reply":"2024-12-14T06:26:39.823186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"percent_unique_words\"]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:39.826141Z","iopub.execute_input":"2024-12-14T06:26:39.826630Z","iopub.status.idle":"2024-12-14T06:26:39.836635Z","shell.execute_reply.started":"2024-12-14T06:26:39.826581Z","shell.execute_reply":"2024-12-14T06:26:39.835350Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:39.838296Z","iopub.execute_input":"2024-12-14T06:26:39.838753Z","iopub.status.idle":"2024-12-14T06:26:39.855722Z","shell.execute_reply.started":"2024-12-14T06:26:39.838703Z","shell.execute_reply":"2024-12-14T06:26:39.854482Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\ndef remove_ip_addresses(text):\n    if isinstance(text, str):\n        return re.sub(r'\\b(?:\\d{1,3}\\.){3}\\d{1,3}\\b', '', text)\n    return text\n\ndf['comment_text'] = df['comment_text'].apply(remove_ip_addresses)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:39.862129Z","iopub.execute_input":"2024-12-14T06:26:39.863164Z","iopub.status.idle":"2024-12-14T06:26:43.592571Z","shell.execute_reply.started":"2024-12-14T06:26:39.863109Z","shell.execute_reply":"2024-12-14T06:26:43.591382Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import langid\n\n\n# # Function to detect language\n# def detect_language(text):\n#     try:\n#         return langid.classify(text)[0]\n#     except Exception:\n#         return 'unknown'\n\n# # Apply language detection to the comment_text column\n# df['detected_lang'] = df['comment_text'].apply(detect_language)\n\n# # Display the dataset with detected languages\n# print(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:43.593950Z","iopub.execute_input":"2024-12-14T06:26:43.594311Z","iopub.status.idle":"2024-12-14T06:26:43.599443Z","shell.execute_reply.started":"2024-12-14T06:26:43.594280Z","shell.execute_reply":"2024-12-14T06:26:43.598170Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_validation = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:43.600970Z","iopub.execute_input":"2024-12-14T06:26:43.601431Z","iopub.status.idle":"2024-12-14T06:26:43.678869Z","shell.execute_reply.started":"2024-12-14T06:26:43.601386Z","shell.execute_reply":"2024-12-14T06:26:43.677768Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_validation.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:43.680244Z","iopub.execute_input":"2024-12-14T06:26:43.680708Z","iopub.status.idle":"2024-12-14T06:26:43.691521Z","shell.execute_reply.started":"2024-12-14T06:26:43.680661Z","shell.execute_reply":"2024-12-14T06:26:43.690313Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_unintended_bias_train_processed.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:43.692941Z","iopub.execute_input":"2024-12-14T06:26:43.693393Z","iopub.status.idle":"2024-12-14T06:26:43.736381Z","shell.execute_reply.started":"2024-12-14T06:26:43.693360Z","shell.execute_reply":"2024-12-14T06:26:43.734366Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_unintended_bias_train_processed=pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train-processed-seqlen128.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:43.738033Z","iopub.status.idle":"2024-12-14T06:26:43.738499Z","shell.execute_reply.started":"2024-12-14T06:26:43.738288Z","shell.execute_reply":"2024-12-14T06:26:43.738309Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_unintended_bias_train=pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:43.739975Z","iopub.status.idle":"2024-12-14T06:26:43.740481Z","shell.execute_reply.started":"2024-12-14T06:26:43.740265Z","shell.execute_reply":"2024-12-14T06:26:43.740294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_unintended_bias_train.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:43.741799Z","iopub.status.idle":"2024-12-14T06:26:43.742222Z","shell.execute_reply.started":"2024-12-14T06:26:43.742023Z","shell.execute_reply":"2024-12-14T06:26:43.742042Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count the occurrences of each language\nlanguage_counts = df_validation['lang'].value_counts()\n\n# Create the bar chart for language breakdown\nplt.figure(figsize=(8, 6))\nplt.bar(language_counts.index, language_counts.values, color=['blue', 'orange', 'green'],alpha=0.7)\nplt.title('Validation Language Breakdown')\nplt.xlabel('Language')\nplt.ylabel('Count')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:27:58.241662Z","iopub.execute_input":"2024-12-14T06:27:58.242090Z","iopub.status.idle":"2024-12-14T06:27:58.483925Z","shell.execute_reply.started":"2024-12-14T06:27:58.242060Z","shell.execute_reply":"2024-12-14T06:27:58.482733Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count toxic and non-toxic comments for each language\ntoxic_counts = df_validation[df_validation['toxic'] == 1]['lang'].value_counts()\nnon_toxic_counts = df_validation[df_validation['toxic'] == 0]['lang'].value_counts()\n\n# Align indexes to ensure both counts match across all languages\nall_languages = non_toxic_counts.index.union(toxic_counts.index)\ntoxic_counts = toxic_counts.reindex(all_languages, fill_value=0)\nnon_toxic_counts = non_toxic_counts.reindex(all_languages, fill_value=0)\n\n# Create positions for grouped bar chart\nx = np.arange(len(all_languages))  # Position of bars\nwidth = 0.35  # Bar width\n\n# Plot bars side by side\nplt.figure(figsize=(10, 6))\nplt.bar(x - width/2, non_toxic_counts.values, width, label='Non-toxic', color='blue',alpha=0.7)\nplt.bar(x + width/2, toxic_counts.values, width, label='Toxic', color='red', alpha=0.7)\n\n# Add labels and title\nplt.title('Language Distribution in the Validation Dataset')\nplt.xlabel('Language')\nplt.ylabel('Count')\nplt.xticks(x, all_languages, rotation=45)  # Add langua\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:31:13.672984Z","iopub.execute_input":"2024-12-14T06:31:13.673490Z","iopub.status.idle":"2024-12-14T06:31:13.885864Z","shell.execute_reply.started":"2024-12-14T06:31:13.673450Z","shell.execute_reply":"2024-12-14T06:31:13.884741Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test=pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:31:18.899112Z","iopub.execute_input":"2024-12-14T06:31:18.899596Z","iopub.status.idle":"2024-12-14T06:31:19.490408Z","shell.execute_reply.started":"2024-12-14T06:31:18.899478Z","shell.execute_reply":"2024-12-14T06:31:19.489093Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:31:21.792244Z","iopub.execute_input":"2024-12-14T06:31:21.792626Z","iopub.status.idle":"2024-12-14T06:31:21.803054Z","shell.execute_reply.started":"2024-12-14T06:31:21.792593Z","shell.execute_reply":"2024-12-14T06:31:21.801886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test_labels=pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test_labels.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:31:26.439281Z","iopub.execute_input":"2024-12-14T06:31:26.439685Z","iopub.status.idle":"2024-12-14T06:31:26.458660Z","shell.execute_reply.started":"2024-12-14T06:31:26.439651Z","shell.execute_reply":"2024-12-14T06:31:26.457571Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_test_labels.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:31:30.165618Z","iopub.execute_input":"2024-12-14T06:31:30.166027Z","iopub.status.idle":"2024-12-14T06:31:30.175723Z","shell.execute_reply.started":"2024-12-14T06:31:30.165992Z","shell.execute_reply":"2024-12-14T06:31:30.174491Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Count the occurrences of each language in the test dataset\nlanguage_counts = df_test['lang'].value_counts()\n\n# Plot the bar chart with new colors\nplt.figure(figsize=(10, 6))\nlanguage_counts.plot(\n    kind='bar', \n    color=['#1f77b4', '#ff7f0e', '#2ca02c', '#d62728', '#9467bd', '#8c564b']  # Custom color palette\n)\nplt.title('Test Language Breakdown')\nplt.xlabel('Language')\nplt.ylabel('Count')\nplt.xticks(rotation=45)\nplt.tight_layout()\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:31:33.963524Z","iopub.execute_input":"2024-12-14T06:31:33.963944Z","iopub.status.idle":"2024-12-14T06:31:34.212241Z","shell.execute_reply.started":"2024-12-14T06:31:33.963908Z","shell.execute_reply":"2024-12-14T06:31:34.211053Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def W_Cloud(token):\n    \"\"\"\n    Visualize the most common words contributing to the token.\n    \"\"\"\n    threat_context = df[df[token] == 1]\n    threat_text = threat_context.comment_text\n    neg_text = pd.Series(threat_text).str.cat(sep=' ')\n    wordcloud = WordCloud(width=1600, height=800,\n                          max_font_size=200).generate(neg_text)\n\n    plt.figure(figsize=(15, 10))\n    plt.imshow(wordcloud.recolor(colormap=\"Blues\"), interpolation='bilinear')\n    plt.axis(\"off\")\n    plt.title(f\"Most common words assosiated with {token} comment\", size=20)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:31:37.592707Z","iopub.execute_input":"2024-12-14T06:31:37.593378Z","iopub.status.idle":"2024-12-14T06:31:37.605370Z","shell.execute_reply.started":"2024-12-14T06:31:37.593308Z","shell.execute_reply":"2024-12-14T06:31:37.603393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from wordcloud import WordCloud\ntoken = input(\n    'Choose a class to visualize the most common words contributing to the class:')\nW_Cloud(token.lower())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:31:40.902226Z","iopub.execute_input":"2024-12-14T06:31:40.902650Z","iopub.status.idle":"2024-12-14T06:31:49.782649Z","shell.execute_reply.started":"2024-12-14T06:31:40.902614Z","shell.execute_reply":"2024-12-14T06:31:49.781336Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# !pip install transformers ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:43.764184Z","iopub.status.idle":"2024-12-14T06:26:43.764618Z","shell.execute_reply.started":"2024-12-14T06:26:43.764431Z","shell.execute_reply":"2024-12-14T06:26:43.764451Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# from transformers import BertTokenizer\n# tokenizer = BertTokenizer.from_pretrained('bert-base-uncased')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:43.766605Z","iopub.status.idle":"2024-12-14T06:26:43.766966Z","shell.execute_reply.started":"2024-12-14T06:26:43.766782Z","shell.execute_reply":"2024-12-14T06:26:43.766799Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def tokenize_and_split(text, tokenizer, max_length=512):\n#     tokens = tokenizer.encode_plus(\n#         str(text),\n#         add_special_tokens=True,  # Add special tokens like [CLS] and [SEP]\n#         max_length=max_length,\n#         truncation=True,  # Truncate if exceeds max_length\n#         padding='max_length',  # Pad to max_length\n#         return_tensors=\"np\"  # Return as NumPy arrays\n#     )\n#     return tokens['input_ids'], tokens['token_type_ids']\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:43.768669Z","iopub.status.idle":"2024-12-14T06:26:43.769055Z","shell.execute_reply.started":"2024-12-14T06:26:43.768879Z","shell.execute_reply":"2024-12-14T06:26:43.768898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Assuming `tokenizer` is already initialized\n# df_train_unprocessed['tokenized_data'] = df_train_unprocessed['comment_text'].apply(lambda x: tokenize_and_split(x, tokenizer))\n\n# # Separate into individual columns for clarity\n# df_train_unprocessed['input_word_ids'] = df_train_unprocessed['tokenized_data'].apply(lambda x: x[0])\n# df_train_unprocessed['all_segment_id'] = df_train_unprocessed['tokenized_data'].apply(lambda x: x[1])\n\n#  # Drop intermediate column if no longer needed\n# # df.drop(columns=['tokenized_data'], inplace=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:26:43.769984Z","iopub.status.idle":"2024-12-14T06:26:43.770380Z","shell.execute_reply.started":"2024-12-14T06:26:43.770168Z","shell.execute_reply":"2024-12-14T06:26:43.770186Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\n\nfrom tokenizers import BertWordPieceTokenizer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:31:58.018275Z","iopub.execute_input":"2024-12-14T06:31:58.018635Z","iopub.status.idle":"2024-12-14T06:31:58.024710Z","shell.execute_reply.started":"2024-12-14T06:31:58.018605Z","shell.execute_reply":"2024-12-14T06:31:58.023502Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm.auto import tqdm  # This works well in Jupyter notebooks and scripts\n# OR\nfrom tqdm import tqdm  # Standard import for most Python environments","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T06:32:01.432903Z","iopub.execute_input":"2024-12-14T06:32:01.433361Z","iopub.status.idle":"2024-12-14T06:32:01.439151Z","shell.execute_reply.started":"2024-12-14T06:32:01.433316Z","shell.execute_reply":"2024-12-14T06:32:01.437881Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=256, maxlen=512, pad_token_id=0):\n    \"\"\"\n    Encoder for encoding the text into a sequence of integers for BERT Input.\n    Assumes pad_token_id is 0 by default (if not set in the tokenizer).\n    \"\"\"\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size]\n        \n        # Encode each text individually\n        encs = tokenizer.encode_batch(text_chunk)\n        \n        # Manually truncate and pad to maxlen\n        padded_encs = [\n            enc.ids[:maxlen] + [pad_token_id] * (maxlen - len(enc.ids)) if len(enc.ids) < maxlen else enc.ids[:maxlen]\n            for enc in encs\n        ]\n        \n        all_ids.extend(padded_encs)\n    \n    return np.array(all_ids)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T07:41:09.402972Z","iopub.execute_input":"2024-12-14T07:41:09.403497Z","iopub.status.idle":"2024-12-14T07:41:09.413378Z","shell.execute_reply.started":"2024-12-14T07:41:09.403460Z","shell.execute_reply":"2024-12-14T07:41:09.411753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\nstrategy = tf.distribute.MirroredStrategy()\n\n# Configuration\nEPOCHS = 3\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 192\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T07:41:15.405793Z","iopub.execute_input":"2024-12-14T07:41:15.406284Z","iopub.status.idle":"2024-12-14T07:41:15.415737Z","shell.execute_reply.started":"2024-12-14T07:41:15.406243Z","shell.execute_reply":"2024-12-14T07:41:15.413993Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tokenizer = transformers.DistilBertTokenizer.from_pretrained('distilbert-base-multilingual-cased')\n# Save the loaded tokenizer locally\ntokenizer.save_pretrained('.')\n# Reload it with the huggingface tokenizers library\nfast_tokenizer = BertWordPieceTokenizer('vocab.txt', lowercase=False)\nfast_tokenizer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T07:41:18.566715Z","iopub.execute_input":"2024-12-14T07:41:18.567132Z","iopub.status.idle":"2024-12-14T07:41:19.371425Z","shell.execute_reply.started":"2024-12-14T07:41:18.567096Z","shell.execute_reply":"2024-12-14T07:41:19.370190Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"MAX_LEN=192\nMAX_LEN = 192\nx_train = fast_encode(df.comment_text.astype(str).tolist(), fast_tokenizer, maxlen=MAX_LEN)\nx_valid = fast_encode(df_validation.comment_text.astype(str).tolist(), fast_tokenizer, maxlen=MAX_LEN)\nx_test = fast_encode(df_test.content.astype(str).tolist(), fast_tokenizer, maxlen=MAX_LEN)\n\ny_train = df.toxic.values\ny_valid = df_validation.toxic.values\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T07:41:26.106423Z","iopub.execute_input":"2024-12-14T07:41:26.106823Z","iopub.status.idle":"2024-12-14T07:42:19.635907Z","shell.execute_reply.started":"2024-12-14T07:41:26.106791Z","shell.execute_reply":"2024-12-14T07:42:19.634154Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T07:32:02.006142Z","iopub.execute_input":"2024-12-14T07:32:02.006569Z","iopub.status.idle":"2024-12-14T07:32:02.747303Z","shell.execute_reply.started":"2024-12-14T07:32:02.006528Z","shell.execute_reply":"2024-12-14T07:32:02.746287Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Check tokenizer output\n# sample_text = [\"This is a sample comment.\"]\n# sample_encoded = fast_encode(sample_text, fast_tokenizer, maxlen=MAX_LEN)\n# print(\"Encoded Shape:\", sample_encoded.shape)  # Should be (1, MAX_LEN)\n\n# # Test input with the transformer layer\n# sample_input = tf.convert_to_tensor(sample_encoded, dtype=tf.int32)\n# print(\"Input Shape for Model:\", sample_input.shape)  # Should be (1, MAX_LEN)\n\n# # Try passing the input through the transformer\n# try:\n#     output = transformer_layer(sample_input)\n#     print(\"Output:\", output)\n# except Exception as e:\n#     print(f\"Error: {e}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T07:43:22.974304Z","iopub.execute_input":"2024-12-14T07:43:22.974750Z","iopub.status.idle":"2024-12-14T07:43:23.547376Z","shell.execute_reply.started":"2024-12-14T07:43:22.974718Z","shell.execute_reply":"2024-12-14T07:43:23.545898Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# sample_text = [\"This is a sample comment.\"]\n#sample_encoded = fast_encode(sample_text, fast_tokenizer, maxlen=MAX_LEN, pad_token_id=0)\n# print(\"Encoded Shape:\", sample_encoded.shape)  # Should be (1, MAX_LEN)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T07:44:14.872895Z","iopub.execute_input":"2024-12-14T07:44:14.873318Z","iopub.status.idle":"2024-12-14T07:44:14.885080Z","shell.execute_reply.started":"2024-12-14T07:44:14.873285Z","shell.execute_reply":"2024-12-14T07:44:14.883629Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_model(transformer, max_len=512):\n    \"\"\"\n    function for training the BERT model\n    \"\"\"\n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n  \n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    out = Dense(1, activation='sigmoid')(cls_token)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=['accuracy'])\n    \n    return model","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T07:53:14.658396Z","iopub.execute_input":"2024-12-14T07:53:14.658908Z","iopub.status.idle":"2024-12-14T07:53:14.667378Z","shell.execute_reply.started":"2024-12-14T07:53:14.658871Z","shell.execute_reply":"2024-12-14T07:53:14.665742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with strategy.scope():\n    # Load the DistilBert model\n    transformer_layer = transformers.TFDistilBertModel.from_pretrained('distilbert-base-multilingual-cased')\n    \n    # Build the model\n    model = build_model(transformer_layer, max_len=MAX_LEN)\n\n# Model summary to verify architecture\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T07:53:18.598855Z","iopub.execute_input":"2024-12-14T07:53:18.599292Z","iopub.status.idle":"2024-12-14T07:53:22.553614Z","shell.execute_reply.started":"2024-12-14T07:53:18.599257Z","shell.execute_reply":"2024-12-14T07:53:22.551883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"n_steps = x_train.shape[0] // BATCH_SIZE\ntrain_history = model.fit(\n    train_dataset,\n    steps_per_epoch=n_steps,\n    validation_data=valid_dataset,\n    epochs=EPOCHS\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-14T07:13:28.403693Z","iopub.execute_input":"2024-12-14T07:13:28.404110Z","iopub.status.idle":"2024-12-14T07:13:28.445989Z","shell.execute_reply.started":"2024-12-14T07:13:28.404074Z","shell.execute_reply":"2024-12-14T07:13:28.443970Z"}},"outputs":[],"execution_count":null}]}