{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-30T15:47:26.855811Z","iopub.execute_input":"2023-05-30T15:47:26.856594Z","iopub.status.idle":"2023-05-30T15:47:26.866850Z","shell.execute_reply.started":"2023-05-30T15:47:26.856559Z","shell.execute_reply":"2023-05-30T15:47:26.866001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install seaborn","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:29:25.482228Z","iopub.execute_input":"2023-06-01T11:29:25.483137Z","iopub.status.idle":"2023-06-01T11:29:31.195720Z","shell.execute_reply.started":"2023-06-01T11:29:25.483091Z","shell.execute_reply":"2023-06-01T11:29:31.194646Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pip install wordcloud","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:29:34.165022Z","iopub.execute_input":"2023-06-01T11:29:34.165437Z","iopub.status.idle":"2023-06-01T11:29:38.906857Z","shell.execute_reply.started":"2023-06-01T11:29:34.165401Z","shell.execute_reply":"2023-06-01T11:29:38.905853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#imports\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport warnings\nwarnings.filterwarnings(\"ignore\")\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom keras.preprocessing import sequence, text\nfrom tensorflow.keras.preprocessing import sequence\nfrom tensorflow.keras.models import Sequential\nfrom keras.layers import Embedding\nfrom keras.layers import LSTM,GRU, SimpleRNN, GlobalMaxPooling1D, Conv1D, MaxPooling1D, Flatten, Bidirectional, SpatialDropout1D\nfrom keras.layers.core import Dense, Activation, Dropout\nfrom sklearn.metrics import roc_auc_score,roc_curve,auc\nfrom sklearn.preprocessing import LabelEncoder\n\n\nfrom wordcloud import WordCloud ,STOPWORDS\nfrom collections import Counter","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:29:38.908747Z","iopub.execute_input":"2023-06-01T11:29:38.909056Z","iopub.status.idle":"2023-06-01T11:30:21.235184Z","shell.execute_reply.started":"2023-06-01T11:29:38.909026Z","shell.execute_reply":"2023-06-01T11:30:21.234046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:30:21.237024Z","iopub.execute_input":"2023-06-01T11:30:21.237566Z","iopub.status.idle":"2023-06-01T11:30:30.578239Z","shell.execute_reply.started":"2023-06-01T11:30:21.237534Z","shell.execute_reply":"2023-06-01T11:30:30.577367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading datsets\ntrain_df = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\") \ntest_df = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv\")\ntest_labels = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test_labels.csv\")\nval_df= pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:30:34.130941Z","iopub.execute_input":"2023-06-01T11:30:34.131423Z","iopub.status.idle":"2023-06-01T11:30:37.334531Z","shell.execute_reply.started":"2023-06-01T11:30:34.131388Z","shell.execute_reply":"2023-06-01T11:30:37.333500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shape of train data\nprint('train shape:',train_df.shape)\ntrain_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:30:37.861299Z","iopub.execute_input":"2023-06-01T11:30:37.862198Z","iopub.status.idle":"2023-06-01T11:30:37.882548Z","shell.execute_reply.started":"2023-06-01T11:30:37.862158Z","shell.execute_reply":"2023-06-01T11:30:37.881514Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shape of test data\nprint('test shape:',test_df.shape)\ntest_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:30:41.732558Z","iopub.execute_input":"2023-06-01T11:30:41.733007Z","iopub.status.idle":"2023-06-01T11:30:41.744422Z","shell.execute_reply.started":"2023-06-01T11:30:41.732973Z","shell.execute_reply":"2023-06-01T11:30:41.743500Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shape of test labels\nprint('test labels shape:',test_labels.shape)\ntest_labels.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:30:44.512872Z","iopub.execute_input":"2023-06-01T11:30:44.513301Z","iopub.status.idle":"2023-06-01T11:30:44.524227Z","shell.execute_reply.started":"2023-06-01T11:30:44.513266Z","shell.execute_reply":"2023-06-01T11:30:44.523296Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# shape of test labels\nprint('val shape:',val_df.shape)\nval_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:30:47.264982Z","iopub.execute_input":"2023-06-01T11:30:47.265928Z","iopub.status.idle":"2023-06-01T11:30:47.276967Z","shell.execute_reply.started":"2023-06-01T11:30:47.265887Z","shell.execute_reply":"2023-06-01T11:30:47.275994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"light_green = (0.5, 0.8, 0.6)\ndark_green=(0.5, 0.7, 0.5)\ntrain_shape = train_df.shape[0]  # Number of samples in the training dataset\ntest_shape = test_df.shape[0]  # Number of samples in the test dataset\nvalidation_shape = val_df.shape[0]  # Number of samples in the validation dataset\nfig = plt.figure(facecolor=light_green)\ndataset_labels = ['Train', 'Test', 'Validation']\ndataset_shapes = [train_shape, test_shape, validation_shape]\n\nfig, ax = plt.subplots(facecolor=light_green)\n\n\nax.set_facecolor(light_green)\nplt.bar(dataset_labels, dataset_shapes, color=dark_green)\nplt.xlabel('Datasets')\nplt.ylabel('Number of Samples')\nplt.title('Number of Samples in Each Dataset')\nfor i, count in enumerate(dataset_shapes):\n    plt.text(i, count, str(count), ha='center', va='bottom')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:30:49.678231Z","iopub.execute_input":"2023-06-01T11:30:49.678640Z","iopub.status.idle":"2023-06-01T11:30:49.956598Z","shell.execute_reply.started":"2023-06-01T11:30:49.678599Z","shell.execute_reply":"2023-06-01T11:30:49.955449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Exploring our dataset:**","metadata":{}},{"cell_type":"code","source":"#checking data format and types for train data\nprint(train_df.info())","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:30:53.184242Z","iopub.execute_input":"2023-06-01T11:30:53.185371Z","iopub.status.idle":"2023-06-01T11:30:53.234601Z","shell.execute_reply.started":"2023-06-01T11:30:53.185325Z","shell.execute_reply":"2023-06-01T11:30:53.233478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checking data format and types for test data\nprint(test_df.info())","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:30:55.233433Z","iopub.execute_input":"2023-06-01T11:30:55.234534Z","iopub.status.idle":"2023-06-01T11:30:55.255237Z","shell.execute_reply.started":"2023-06-01T11:30:55.234490Z","shell.execute_reply":"2023-06-01T11:30:55.254364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Statistics for train data\ntrain_df.describe()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:30:58.797998Z","iopub.execute_input":"2023-06-01T11:30:58.799024Z","iopub.status.idle":"2023-06-01T11:30:58.854096Z","shell.execute_reply.started":"2023-06-01T11:30:58.798961Z","shell.execute_reply":"2023-06-01T11:30:58.853001Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# checking for null values in trainset\ntrain_df.isnull().sum()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:31:01.735704Z","iopub.execute_input":"2023-06-01T11:31:01.736119Z","iopub.status.idle":"2023-06-01T11:31:01.778146Z","shell.execute_reply.started":"2023-06-01T11:31:01.736088Z","shell.execute_reply":"2023-06-01T11:31:01.777149Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Data visualization and exploratory data analysis:**","metadata":{}},{"cell_type":"markdown","source":"**Plotting comment length:**","metadata":{}},{"cell_type":"code","source":"# plotting\ng1 = (0.5, 0.8, 0.6)  # RGB values for a light green shade\ng2 = (0.7, 0.8, 0.7)  # RGB values for a dark green shade\n\nfig, ax = plt.subplots(facecolor=g2)\nax.set_facecolor(g2)\n\ncomment_length = train_df['comment_text'].str.split().apply(len)\nsns.histplot(comment_length, bins=50, color=g1)\n\nplt.title(\"Distribution for Lengths of Words\")\nplt.xlabel(\"Number of Words\")\nplt.ylabel(\"Density\")\nplt.xlim(0, 600)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:31:04.363797Z","iopub.execute_input":"2023-06-01T11:31:04.364228Z","iopub.status.idle":"2023-06-01T11:31:07.570837Z","shell.execute_reply.started":"2023-06-01T11:31:04.364195Z","shell.execute_reply":"2023-06-01T11:31:07.569879Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The above graph shows the comment length plot. we can observe that the length of most of the words is in the range to 0-50 words. More than 1,60000 comments have less than 50 words. Around 45000 comments have 50-100 words. Very few comments have large number of words in it.","metadata":{}},{"cell_type":"markdown","source":"**Checking total number of clean, labelled comments:**","metadata":{}},{"cell_type":"code","source":"# Iterating from 3rd column to the end for train_df\nrowSums = train_df.iloc[:,2:].sum(axis=1)\n# counting clean comments if rowSums==0\nclean_comments_count = (rowSums==0).sum(axis=0)\n\n# printing clean, label comments\nprint(\"Total number of comments = \",len(train_df))\nprint(\"Number of clean comments = \",clean_comments_count)\nprint(\"Number of comments with labels =\",(len(train_df)-clean_comments_count))","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:31:07.572486Z","iopub.execute_input":"2023-06-01T11:31:07.573159Z","iopub.status.idle":"2023-06-01T11:31:07.607975Z","shell.execute_reply.started":"2023-06-01T11:31:07.573127Z","shell.execute_reply":"2023-06-01T11:31:07.607013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We observe that the total number of comments in our datset is \"223549\" out of which \"201081\" comments are clean and 22468 comments seem to be toxic. So we can see a clear imbalance in our dataset.","metadata":{}},{"cell_type":"markdown","source":"**Calculate percentage of rows with only zeros in training labels(clean comments percentage)**","metadata":{}},{"cell_type":"code","source":"print(f\"{round(clean_comments_count /len(train_df),3)} % percentage of rows contains only zeros in training data and come under clean comments\")","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:31:10.850510Z","iopub.execute_input":"2023-06-01T11:31:10.850951Z","iopub.status.idle":"2023-06-01T11:31:10.856330Z","shell.execute_reply.started":"2023-06-01T11:31:10.850915Z","shell.execute_reply":"2023-06-01T11:31:10.855459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Checking number of comments for each category:**","metadata":{}},{"cell_type":"code","source":"# Getting column names from train_df\ncategories = list(train_df.columns.values)\n# extracting from 3rd column to end\ncategories = categories[2:]\nprint(categories)\n\ncounts = []\n# Iterating over categories\nfor category in categories:\n    counts.append((category, train_df[category].sum()))\n# storing values in a dataframe\ndf_stats = pd.DataFrame(counts, columns=['category', 'number of comments'])\ndf_stats","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:31:12.895890Z","iopub.execute_input":"2023-06-01T11:31:12.896977Z","iopub.status.idle":"2023-06-01T11:31:12.913250Z","shell.execute_reply.started":"2023-06-01T11:31:12.896918Z","shell.execute_reply":"2023-06-01T11:31:12.912193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Plotting the number of comments for each category in a graph:**","metadata":{}},{"cell_type":"code","source":"# plotting\n\ng1 = (0.5, 0.8, 0.6)  # RGB values for a light green shade\ng2 = (0.7, 0.8, 0.7)\n\nsns.set(font_scale=2)\nfig, ax = plt.subplots(figsize=(15, 8), facecolor=g2)\nax.set_facecolor(g2)\n\nsns.barplot(x=categories, y=train_df.iloc[:, 2:].sum().values, color=g1)\nplt.title(\"Comments in each category\", fontsize=24)\nplt.ylabel('Number of comments', fontsize=18)\nplt.xlabel('Comment Type', fontsize=18)\n\n# Adding the text labels\nrects = ax.patches\nlabels = train_df.iloc[:, 2:].sum().values\nfor rect, label in zip(rects, labels):\n    height = rect.get_height()\n    ax.text(rect.get_x() + rect.get_width() / 2, height + 5, label, ha='center', va='bottom', fontsize=18)\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:31:15.364369Z","iopub.execute_input":"2023-06-01T11:31:15.365315Z","iopub.status.idle":"2023-06-01T11:31:15.740758Z","shell.execute_reply.started":"2023-06-01T11:31:15.365272Z","shell.execute_reply":"2023-06-01T11:31:15.739731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the above plot, we can observe that most of the comments come under toxic category followed by obscene and insult. The least number of comments come under threat category.","metadata":{}},{"cell_type":"code","source":"# plotting\nrowSums = train_df.iloc[:,2:].sum(axis=1)\nmultiLabel_counts = rowSums.value_counts()\nmultiLabel_counts = multiLabel_counts.iloc[1:]\n\ng1 = (0.5, 0.8, 0.6)  # RGB values for a light green shade\ng2 = (0.7, 0.8, 0.7)\n\nsns.set(font_scale=2)\nfig, ax = plt.subplots(figsize=(15, 8), facecolor=g2)\nax.set_facecolor(g2)\n\nsns.barplot(x=multiLabel_counts.index, y=multiLabel_counts.values, color=g1)\nplt.title(\"Comments in each category\", fontsize=24)\nplt.ylabel('Number of comments', fontsize=18)\nplt.xlabel('Comment Type', fontsize=18)\n\n# Adding the text labels\nrects = ax.patches\nlabels = train_df.iloc[:, 2:].sum().values\nfor rect, label in zip(rects, labels):\n    height = rect.get_height()\n    ax.text(rect.get_x() + rect.get_width() / 2, height + 5, label, ha='center', va='bottom', fontsize=18)\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:31:20.107860Z","iopub.execute_input":"2023-06-01T11:31:20.108961Z","iopub.status.idle":"2023-06-01T11:31:20.493267Z","shell.execute_reply.started":"2023-06-01T11:31:20.108915Z","shell.execute_reply":"2023-06-01T11:31:20.492326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The above plot shows the number of comments that come under multiple labels. Most of the comments have only one label. Very few comments come under all categories of toxicity which is around 45 comments.","metadata":{}},{"cell_type":"markdown","source":"****Unique words count distribution in percentage for all the comments in train set**","metadata":{}},{"cell_type":"code","source":"count_word = train_df[\"comment_text\"].apply(lambda x: len(str(x).split()))\n#Unique word count\ncount_unique_word = train_df[\"comment_text\"].apply(lambda x: len(set(str(x).split())))\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:31:24.168103Z","iopub.execute_input":"2023-06-01T11:31:24.169063Z","iopub.status.idle":"2023-06-01T11:31:28.095946Z","shell.execute_reply.started":"2023-06-01T11:31:24.169017Z","shell.execute_reply":"2023-06-01T11:31:28.094864Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# calculating unique percentage\nunique_percent = count_unique_word*100/count_word\nunique_percent","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:31:37.850844Z","iopub.execute_input":"2023-06-01T11:31:37.851939Z","iopub.status.idle":"2023-06-01T11:31:37.861719Z","shell.execute_reply.started":"2023-06-01T11:31:37.851899Z","shell.execute_reply":"2023-06-01T11:31:37.860639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting\ng1 = (0.5, 0.8, 0.6)  # RGB values for a light green shade\ng2 = (0.7, 0.8, 0.7)\n\nunique_percent.hist(bins=30, figsize=(10, 7), color=g1)\nplt.suptitle(\"Histogram for unique words distribution\")\nplt.xlabel(\"Percentage\")\nplt.ylabel(\"Number of comments\")\n\n# Changing the background and face color\nplt.gca().set_facecolor(g2)\n\n# Changing the histogram color\nplt.gca().patch.set_color(g2)\nplt.gca().patches[0].set_color(g2)\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:31:40.171853Z","iopub.execute_input":"2023-06-01T11:31:40.172582Z","iopub.status.idle":"2023-06-01T11:31:40.581467Z","shell.execute_reply.started":"2023-06-01T11:31:40.172537Z","shell.execute_reply":"2023-06-01T11:31:40.580432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The above plot shows the percentage of unique words for comments in train data. we can see that around 50,000 comments have 100% unique words in them. Very few comments have repeated words.","metadata":{}},{"cell_type":"markdown","source":"**word count distribution for each category in training set**","metadata":{}},{"cell_type":"code","source":"# plotting\n\ng1 = (0.5, 0.8, 0.6)  # RGB values for a light green shade\ng2 = (0.7, 0.8, 0.7)  # RGB values for a different shade\n\n# Creating labels list\nlabels = ['toxic', 'severe_toxic', 'obscene', 'threat', 'insult', 'identity_hate']\n\n# Plotting\nfig, ax = plt.subplots(nrows=3, ncols=2, figsize=(15, 10), sharex=True)\naxes = ax.ravel()\n\nfor i in range(6):\n    comments = train_df.loc[train_df[labels[i]] == 1, :]\n    comment_len = [len(comment.split()) for comment in comments[\"comment_text\"]]\n    sns.histplot(comment_len, ax=axes[i], bins=50, color=g1)\n    plt.xlim(0, 400)\n    axes[i].title.set_text(labels[i])\n    axes[i].set_facecolor(g2)\n    axes[i].patch.set_color(g2)\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:31:45.083281Z","iopub.execute_input":"2023-06-01T11:31:45.084084Z","iopub.status.idle":"2023-06-01T11:31:47.161224Z","shell.execute_reply.started":"2023-06-01T11:31:45.084042Z","shell.execute_reply":"2023-06-01T11:31:47.159921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The above plot shows the number of words for comments in train data. we can see that most of the comments have less than 50 words. Very few comments have more than 100 words in them.","metadata":{}},{"cell_type":"markdown","source":"**Fequent words in clean comments:**","metadata":{}},{"cell_type":"code","source":"# marking comments without any tags as \"clean\"\ntag_sums = train_df.iloc[:,2:].sum(axis=1)\ntrain_df['clean'] = (tag_sums==0)\nprint(train_df['clean'].value_counts())\nclean_data = train_df[train_df['clean'] == True]\nclean_data.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:31:50.739828Z","iopub.execute_input":"2023-06-01T11:31:50.740998Z","iopub.status.idle":"2023-06-01T11:31:50.801989Z","shell.execute_reply.started":"2023-06-01T11:31:50.740955Z","shell.execute_reply":"2023-06-01T11:31:50.800796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# clean words\nsubset=train_df[train_df['clean'] == True]\ntext = \" \".join(i for i in subset.comment_text)\nstopwords = set(STOPWORDS)\nwordcloud = WordCloud(stopwords=stopwords, colormap=\"Greens\").generate(text)\nplt.figure( figsize=(8,4))\nplt.imshow(wordcloud, interpolation='bilinear')\nplt.axis(\"off\")\nplt.title(\"frequent words in Clean Comments\", fontsize=20)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:31:56.767058Z","iopub.execute_input":"2023-06-01T11:31:56.767431Z","iopub.status.idle":"2023-06-01T11:33:02.835707Z","shell.execute_reply.started":"2023-06-01T11:31:56.767402Z","shell.execute_reply":"2023-06-01T11:33:02.834408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The above word cloud shows the most frequent words in clean comments. The words with large size in the above plot are more frequent.","metadata":{}},{"cell_type":"markdown","source":"**Heatmap for training data**","metadata":{}},{"cell_type":"code","source":"\n# Create a copy of the DataFrame\nencoded_df = train_df.copy()\n\n# Apply label encoding to string columns\nlabel_encoder = LabelEncoder()\nfor column in encoded_df.columns:\n    if encoded_df[column].dtype == object:\n        encoded_df[column] = label_encoder.fit_transform(encoded_df[column])\n# plotting heatmap\nfig = plt.figure(figsize = (10,8))\nsns.heatmap(encoded_df.corr(), annot=True,cmap=\"Greens\")\nplt.suptitle('Heatmap of Training label Class Correlation',size = 14)\nplt.xlabel(\"Classes\")\nplt.ylabel(\"Classes\")\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:33:42.452492Z","iopub.execute_input":"2023-06-01T11:33:42.452880Z","iopub.status.idle":"2023-06-01T11:33:44.956415Z","shell.execute_reply.started":"2023-06-01T11:33:42.452851Z","shell.execute_reply":"2023-06-01T11:33:44.955361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"****Finding correlation between columns in train data to check the dependency of each column**","metadata":{}},{"cell_type":"code","source":"# Finding correlation\ncorrelation_val =  encoded_df.corr()\ncorrelation_val","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:33:48.970111Z","iopub.execute_input":"2023-06-01T11:33:48.970522Z","iopub.status.idle":"2023-06-01T11:33:49.042207Z","shell.execute_reply.started":"2023-06-01T11:33:48.970488Z","shell.execute_reply":"2023-06-01T11:33:49.041069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Checking if correlation is greater than or equal to 0.5\nabs(correlation_val) >= 0.5","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:33:51.719421Z","iopub.execute_input":"2023-06-01T11:33:51.720345Z","iopub.status.idle":"2023-06-01T11:33:51.737006Z","shell.execute_reply.started":"2023-06-01T11:33:51.720305Z","shell.execute_reply":"2023-06-01T11:33:51.735895Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Some classes are highly positively corelated to others i.e  Correlation>50%\n\n for example: the correlation between toxic and insult is 0.66\n\n\nThis means that if a comment is toxic then there is a 66% chance it comes under insult category.","metadata":{}},{"cell_type":"markdown","source":"**Finding the distribution percentage of each class for every category:**","metadata":{}},{"cell_type":"code","source":"print(\"Distribution of Training Classes in Percentage:\")\nprint()\n\nprint((train_df['toxic'].value_counts()/len(train_df))*100)\nprint()\nprint(train_df['severe_toxic'].value_counts()/len(train_df) *100)\nprint()\nprint(train_df['obscene'].value_counts()/len(train_df) *100)\nprint()\nprint(train_df['threat'].value_counts()/len(train_df) *100)\nprint()\nprint(train_df['insult'].value_counts()/len(train_df) *100)\nprint()\nprint(train_df['identity_hate'].value_counts()/len(train_df) *100)\nprint()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:33:54.455006Z","iopub.execute_input":"2023-06-01T11:33:54.456234Z","iopub.status.idle":"2023-06-01T11:33:54.489758Z","shell.execute_reply.started":"2023-06-01T11:33:54.456183Z","shell.execute_reply":"2023-06-01T11:33:54.488453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the above percentages, we can see that most of the comments are clean, More than 90% of the comments for each category are clean, SO our data is hightly imbalanced data.","metadata":{}},{"cell_type":"markdown","source":"**Dealing with imbalance data**","metadata":{}},{"cell_type":"code","source":"# plotting\ng1 = (0.5, 0.8, 0.6)  # RGB values for a light green shade\ng2 = (0.7, 0.8, 0.7)  # RGB values for a different shade\n\ntrain_toxic_comments = train_df[train_df[categories].sum(axis=1) > 0]\ntrain_clean_comments = train_df[train_df[categories].sum(axis=1) == 0]\n\ndf = pd.DataFrame(dict(\n    toxic=[len(train_toxic_comments)],\n    clean=[len(train_clean_comments)]\n))\n\nax = df.plot(kind='barh', fontsize=12)\nax.set_facecolor(g2)\nax.patch.set_color(g2)\nax.patches[0].set_color(g1)\nax.legend()\nax.set_title('class imbalance')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:34:01.911583Z","iopub.execute_input":"2023-06-01T11:34:01.912587Z","iopub.status.idle":"2023-06-01T11:34:02.287832Z","shell.execute_reply.started":"2023-06-01T11:34:01.912550Z","shell.execute_reply":"2023-06-01T11:34:02.286778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From the above plot, we can conclude that more than 2,00,000 comments are clean and only around 25000 comments are toxic which indicates clear data imbalance.There is a high chance of overfitting the model  predicts almost all comments as not toxic.\nSo, we have to modify the data such that it also predicts the toxicity and the type of toxicity accurately.","metadata":{}},{"cell_type":"code","source":"mod_training_df = pd.concat([\n  train_toxic_comments,\n  train_clean_comments.sample(15_000)\n])\nmod_training_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:34:26.186741Z","iopub.execute_input":"2023-06-01T11:34:26.187118Z","iopub.status.idle":"2023-06-01T11:34:26.204074Z","shell.execute_reply.started":"2023-06-01T11:34:26.187090Z","shell.execute_reply":"2023-06-01T11:34:26.202950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Here we are Under Sampling the clean comments by dropping a few rows.\nWe are performing under sampling so as to not increase the number of comments as our datset is already huge enough and can provude enough training.","metadata":{}},{"cell_type":"code","source":"mod_training_df.toxic.value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:34:29.064968Z","iopub.execute_input":"2023-06-01T11:34:29.065857Z","iopub.status.idle":"2023-06-01T11:34:29.074987Z","shell.execute_reply.started":"2023-06-01T11:34:29.065815Z","shell.execute_reply":"2023-06-01T11:34:29.073940Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting\n\ng1 = (0.5, 0.8, 0.6)  # RGB values for a light green shade\ng2 = (0.7, 0.8, 0.7)  # RGB values for a different shade\n\ntrain_toxic_comments = mod_training_df[mod_training_df[categories].sum(axis=1) > 0]\ntrain_clean_comments = mod_training_df[mod_training_df[categories].sum(axis=1) == 0]\n\ndf = pd.DataFrame(dict(\n    toxic=[len(train_toxic_comments)],\n    clean=[len(train_clean_comments)]\n))\n\nax = df.plot(kind='barh', fontsize=12)\nax.set_facecolor(g2)\nax.patch.set_color(g2)\nax.patches[0].set_color(g1)\nax.legend()\nax.set_title('corrected class imbalance')\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:34:33.831265Z","iopub.execute_input":"2023-06-01T11:34:33.831690Z","iopub.status.idle":"2023-06-01T11:34:34.082189Z","shell.execute_reply.started":"2023-06-01T11:34:33.831657Z","shell.execute_reply":"2023-06-01T11:34:34.081343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now if we compare toxic column of our train data, we can see that data imbalance is reduced to agreat extent. so now we can use our data for further processing.","metadata":{}},{"cell_type":"markdown","source":"**Building a Base line model for our modified train dataset**","metadata":{}},{"cell_type":"markdown","source":"**Splitting data into train and validation sets**","metadata":{}},{"cell_type":"code","source":"# splitting data with test size = 20%\nxtrain, xvalid, ytrain, yvalid = train_test_split(mod_training_df.comment_text.values, mod_training_df.toxic.values, \n                                                  stratify=mod_training_df.toxic.values, \n                                                  random_state=42, \n                                                  test_size=0.2, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:34:39.365832Z","iopub.execute_input":"2023-06-01T11:34:39.366774Z","iopub.status.idle":"2023-06-01T11:34:39.383890Z","shell.execute_reply.started":"2023-06-01T11:34:39.366734Z","shell.execute_reply":"2023-06-01T11:34:39.382888Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# performing tokenization\ntoken = tf.keras.preprocessing.text.Tokenizer(num_words=None)\nmax_len = 1500\n\ntoken.fit_on_texts(list(xtrain) + list(xvalid))\nxtrain_seq = token.texts_to_sequences(xtrain)\nxvalid_seq = token.texts_to_sequences(xvalid)\n\n#zero pad the sequences\nxtrain_pad = sequence.pad_sequences(xtrain_seq, maxlen=max_len)\nxvalid_pad = sequence.pad_sequences(xvalid_seq, maxlen=max_len)\n\nword_index = token.word_index","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:34:41.796420Z","iopub.execute_input":"2023-06-01T11:34:41.797588Z","iopub.status.idle":"2023-06-01T11:34:46.546622Z","shell.execute_reply.started":"2023-06-01T11:34:41.797544Z","shell.execute_reply":"2023-06-01T11:34:46.545459Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining the model\nmodel = Sequential()\nmodel.add(Embedding(len(word_index) + 1,\n                     300,\n                     input_length=max_len))\nmodel.add(SimpleRNN(32))\nmodel.add(Dense(1, activation='sigmoid'))\nmodel.compile(loss='binary_crossentropy', optimizer='adam', metrics=['accuracy'])\n    \nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:34:46.548394Z","iopub.execute_input":"2023-06-01T11:34:46.549156Z","iopub.status.idle":"2023-06-01T11:34:46.758074Z","shell.execute_reply.started":"2023-06-01T11:34:46.549125Z","shell.execute_reply":"2023-06-01T11:34:46.756973Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We are uisng simple RNN as our baseline model, The activation used is sigmoid function which is suitable for our binary classification problem.","metadata":{}},{"cell_type":"code","source":"#Fitting your model on the train data and choose appropriate batch_size\nmodel.fit(xtrain_pad, ytrain, epochs=3, batch_size=64)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T11:34:50.752017Z","iopub.execute_input":"2023-06-01T11:34:50.752963Z","iopub.status.idle":"2023-06-01T11:59:50.613759Z","shell.execute_reply.started":"2023-06-01T11:34:50.752926Z","shell.execute_reply":"2023-06-01T11:59:50.612581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Function to calculate roc_suc score and return fpr, tpr for plotting\ndef calculate_roc_auc(predictions,target):\n    '''\n    This methods returns the AUC Score, fpr,tpr when given the Predictions\n    and Labels\n    '''\n    \n    fpr, tpr, thresholds = roc_curve(target, predictions)\n    roc_auc = auc(fpr, tpr)\n    return roc_auc,fpr,tpr","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:02:26.377976Z","iopub.execute_input":"2023-06-01T12:02:26.379382Z","iopub.status.idle":"2023-06-01T12:02:26.385153Z","shell.execute_reply.started":"2023-06-01T12:02:26.379337Z","shell.execute_reply":"2023-06-01T12:02:26.384159Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predicting on validation set\nscores = model.predict(xvalid_pad)\n# calling roc_auc function\nroc_auc,fpr,tpr = calculate_roc_auc(scores,yvalid)\nprint(\"Roc_Auc: %.2f%%\" % (roc_auc))","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:32:33.296596Z","iopub.execute_input":"2023-06-01T12:32:33.297404Z","iopub.status.idle":"2023-06-01T12:32:55.813110Z","shell.execute_reply.started":"2023-06-01T12:32:33.297364Z","shell.execute_reply":"2023-06-01T12:32:55.811929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting\ng2 = (0.7, 0.8, 0.7)  # RGB values for a different shade\n\n# Plotting ROC curve\nplt.plot(fpr, tpr, label='ROC curve (area = %0.2f)' % roc_auc, color='green')\nplt.plot([0, 1], [0, 1], 'k--')\n\n# Setting background and face color\nplt.gca().set_facecolor(g2)\nplt.gca().patch.set_color(g2)\n\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve')\nplt.legend(loc=\"lower right\", fontsize=10)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:32:58.531534Z","iopub.execute_input":"2023-06-01T12:32:58.532563Z","iopub.status.idle":"2023-06-01T12:32:58.861998Z","shell.execute_reply.started":"2023-06-01T12:32:58.532523Z","shell.execute_reply":"2023-06-01T12:32:58.860809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The above graph shows the roc_auc score. our roc_auc score is above 90%.\nA ROC AUC score of 90% indicates that the model's predictions have a high degree of separability between the positive and negative classes.\n\nThe ROC curve is a plot of the true positive rate (TPR) against the false positive rate (FPR) for different threshold values. As our ROC AUC score near to 1 it means that the model has a perfect ability to distinguish between the positive and negative classes.","metadata":{}},{"cell_type":"code","source":"scores_model = []\nscores_model.append({'Model': 'SimpleRNN','AUC_Score': roc_auc})\nscores_model","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:08:56.611158Z","iopub.execute_input":"2023-06-01T12:08:56.612172Z","iopub.status.idle":"2023-06-01T12:08:56.618497Z","shell.execute_reply.started":"2023-06-01T12:08:56.612132Z","shell.execute_reply":"2023-06-01T12:08:56.617636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**LSTM**","metadata":{}},{"cell_type":"code","source":"# loading the GloVe vectors in a dictionary:\n\nembeddings_index = {}\nf = open('/kaggle/input/gloveinput1/glove.840B.300d.txt','r',encoding='utf-8')\nfor line in f:\n    values = line.split(' ')\n    word = values[0]\n    coefs = np.asarray([float(val) for val in values[1:]])\n    embeddings_index[word] = coefs\nf.close()\n\nprint('Found %s word vectors.' % len(embeddings_index))","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:08:59.552156Z","iopub.execute_input":"2023-06-01T12:08:59.553199Z","iopub.status.idle":"2023-06-01T12:13:03.523982Z","shell.execute_reply.started":"2023-06-01T12:08:59.553158Z","shell.execute_reply":"2023-06-01T12:13:03.522571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating an embedding matrix for the words we have in the dataset\nembedding_matrix = np.zeros((len(word_index) + 1, 300))\nfor word, i in word_index.items():\n    embedding_vector = embeddings_index.get(word)\n    if embedding_vector is not None:\n        embedding_matrix[i] = embedding_vector","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:13:14.281009Z","iopub.execute_input":"2023-06-01T12:13:14.281508Z","iopub.status.idle":"2023-06-01T12:13:14.531703Z","shell.execute_reply.started":"2023-06-01T12:13:14.281467Z","shell.execute_reply":"2023-06-01T12:13:14.530409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Modelling\n%%time\nwith strategy.scope():\n    \n    # A simple LSTM with glove embeddings and one dense layer\n    model_lstm = Sequential()\n    model_lstm.add(Embedding(len(word_index) + 1,\n                     300,\n                     weights=[embedding_matrix],\n                     input_length=max_len,\n                     trainable=False))\n\n    model_lstm.add(LSTM(32, dropout=0.3, recurrent_dropout=0.3))\n    model_lstm.add(Dense(1, activation='sigmoid'))\n    model_lstm.compile(loss='binary_crossentropy', optimizer='adam',metrics=['accuracy'])\n    \nmodel_lstm.summary()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:13:19.029391Z","iopub.execute_input":"2023-06-01T12:13:19.029770Z","iopub.status.idle":"2023-06-01T12:13:23.708755Z","shell.execute_reply.started":"2023-06-01T12:13:19.029741Z","shell.execute_reply":"2023-06-01T12:13:23.707741Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fitting the model on the train data and choose appropriate batch_size\nmodel_lstm.fit(xtrain_pad, ytrain, epochs=3, batch_size=64)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:13:27.300696Z","iopub.execute_input":"2023-06-01T12:13:27.301067Z","iopub.status.idle":"2023-06-01T12:16:27.830287Z","shell.execute_reply.started":"2023-06-01T12:13:27.301040Z","shell.execute_reply":"2023-06-01T12:16:27.828838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predicting on validation set\nscores = model_lstm.predict(xvalid_pad)\n# calling roc_auc function\nroc_auc,fpr,tpr = calculate_roc_auc(scores,yvalid)\nprint(\"Roc_Auc: %.2f%%\" % (roc_auc))","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:34:46.764738Z","iopub.execute_input":"2023-06-01T12:34:46.765482Z","iopub.status.idle":"2023-06-01T12:34:56.747688Z","shell.execute_reply.started":"2023-06-01T12:34:46.765440Z","shell.execute_reply":"2023-06-01T12:34:56.746467Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"scores","metadata":{"execution":{"iopub.status.busy":"2023-05-30T16:24:21.650801Z","iopub.execute_input":"2023-05-30T16:24:21.651176Z","iopub.status.idle":"2023-05-30T16:24:21.657224Z","shell.execute_reply.started":"2023-05-30T16:24:21.651151Z","shell.execute_reply":"2023-05-30T16:24:21.656347Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting\ng2 = (0.7, 0.8, 0.7)  # RGB values for a green shade\n\n# Plot ROC curve\nplt.plot(fpr, tpr, label='ROC curve (area = %0.2f)' % roc_auc, color='green')\nplt.plot([0, 1], [0, 1], 'k--')\n\n# Setting background and face color\nplt.gca().set_facecolor(g2)\nplt.gca().patch.set_color(g2)\n\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve')\nplt.legend(loc=\"lower right\", fontsize=10)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:35:02.413401Z","iopub.execute_input":"2023-06-01T12:35:02.414236Z","iopub.status.idle":"2023-06-01T12:35:02.736790Z","shell.execute_reply.started":"2023-06-01T12:35:02.414198Z","shell.execute_reply":"2023-06-01T12:35:02.735604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The above graph shows the roc_auc score for LSTM. our roc_auc score is around 96%. A ROC AUC score of 96% indicates that the model's predictions have a high degree of separability between the positive and negative classes.\nThe ROC curve is a plot of the true positive rate (TPR) against the false positive rate (FPR) for different threshold values. As our ROC AUC score near to 1 it means that the model has a perfect ability to distinguish between the positive and negative classes.\nWe now see that the model is not overfitting and achieves an auc score of 0.96 which is quite commendable. We see that in this case we used dropout and prevented overfitting of the data","metadata":{}},{"cell_type":"code","source":"scores_model = []\nscores_model.append({'Model': 'LSTM','AUC_Score': roc_auc})\nscores_model","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:17:26.724061Z","iopub.execute_input":"2023-06-01T12:17:26.724526Z","iopub.status.idle":"2023-06-01T12:17:26.732275Z","shell.execute_reply.started":"2023-06-01T12:17:26.724481Z","shell.execute_reply":"2023-06-01T12:17:26.731194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**GRU**","metadata":{}},{"cell_type":"code","source":"# modelling\n%%time\nwith strategy.scope():\n    # GRU with glove embeddings and two dense layers\n     model_gru = Sequential()\n     model_gru.add(Embedding(len(word_index) + 1,\n                     300,\n                     weights=[embedding_matrix],\n                     input_length=max_len,\n                     trainable=False))\n     model_gru.add(SpatialDropout1D(0.3))\n     model_gru.add(GRU(32))\n     model_gru.add(Dense(1, activation='sigmoid'))\n\n     model_gru.compile(loss='binary_crossentropy', optimizer='adam',metrics=['accuracy'])   \n    \nmodel_gru.summary()","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:17:31.658160Z","iopub.execute_input":"2023-06-01T12:17:31.659092Z","iopub.status.idle":"2023-06-01T12:17:35.024421Z","shell.execute_reply.started":"2023-06-01T12:17:31.659053Z","shell.execute_reply":"2023-06-01T12:17:35.023408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Fitting the model on the train data and choose appropriate batch_size\nmodel_gru.fit(xtrain_pad, ytrain, epochs=3, batch_size=64)","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:17:44.898933Z","iopub.execute_input":"2023-06-01T12:17:44.899924Z","iopub.status.idle":"2023-06-01T12:19:33.487397Z","shell.execute_reply.started":"2023-06-01T12:17:44.899883Z","shell.execute_reply":"2023-06-01T12:19:33.486204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predicting on validation set\nscores = model_gru.predict(xvalid_pad)\n# calling roc_auc function\nroc_auc,fpr,tpr = calculate_roc_auc(scores,yvalid)\nprint(\"Roc_Auc: %.2f%%\" % (roc_auc))","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:45:40.771461Z","iopub.execute_input":"2023-06-01T12:45:40.772470Z","iopub.status.idle":"2023-06-01T12:45:48.135698Z","shell.execute_reply.started":"2023-06-01T12:45:40.772430Z","shell.execute_reply":"2023-06-01T12:45:48.134241Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Scores from GRU model\",scores)","metadata":{"execution":{"iopub.status.busy":"2023-05-30T16:26:22.520590Z","iopub.execute_input":"2023-05-30T16:26:22.521000Z","iopub.status.idle":"2023-05-30T16:26:22.526112Z","shell.execute_reply.started":"2023-05-30T16:26:22.520973Z","shell.execute_reply":"2023-05-30T16:26:22.525000Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# plotting\ng2 = (0.7, 0.8, 0.7)  # RGB values for a different shade\n\n# Plotting ROC curve\nplt.plot(fpr, tpr, label='ROC curve (area = %0.2f)' % roc_auc, color='green')\nplt.plot([0, 1], [0, 1], 'k--')\n\n# Set background and face color\nplt.gca().set_facecolor(g2)\nplt.gca().patch.set_color(g2)\n\nplt.xlim([0.0, 1.0])\nplt.ylim([0.0, 1.05])\nplt.xlabel('False Positive Rate')\nplt.ylabel('True Positive Rate')\nplt.title('ROC Curve')\nplt.legend(loc=\"lower right\", fontsize=10)\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:45:50.383933Z","iopub.execute_input":"2023-06-01T12:45:50.384959Z","iopub.status.idle":"2023-06-01T12:45:50.734991Z","shell.execute_reply.started":"2023-06-01T12:45:50.384919Z","shell.execute_reply":"2023-06-01T12:45:50.733601Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The above graph shows the roc_auc score for GRU. our roc_auc score is around 95%. A ROC AUC score of 95% indicates that the model's predictions have a high degree of separability between the positive and negative classes. we have used spatial Dropout layer in GRU","metadata":{}},{"cell_type":"code","source":"scores_model = []\nscores_model.append({'Model': 'GRU','AUC_Score': roc_auc})\nscores_model","metadata":{"execution":{"iopub.status.busy":"2023-06-01T12:21:49.425440Z","iopub.execute_input":"2023-06-01T12:21:49.426521Z","iopub.status.idle":"2023-06-01T12:21:49.433072Z","shell.execute_reply.started":"2023-06-01T12:21:49.426482Z","shell.execute_reply":"2023-06-01T12:21:49.431917Z"},"trusted":true},"execution_count":null,"outputs":[]}]}