{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"raw","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T08:52:44.132645Z","iopub.execute_input":"2022-07-09T08:52:44.133431Z","iopub.status.idle":"2022-07-09T08:52:44.160877Z","shell.execute_reply.started":"2022-07-09T08:52:44.133333Z","shell.execute_reply":"2022-07-09T08:52:44.159657Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/jigsaw-toxic-comment-classification-challenge/train.csv.zip', encoding=\"ISO-8859-1\")\ntrain_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T08:52:44.162077Z","iopub.execute_input":"2022-07-09T08:52:44.162514Z","iopub.status.idle":"2022-07-09T08:52:46.160143Z","shell.execute_reply.started":"2022-07-09T08:52:44.162486Z","shell.execute_reply":"2022-07-09T08:52:46.159362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test = pd.read_csv('/kaggle/input/jigsaw-toxic-comment-classification-challenge/test.csv.zip', encoding=\"ISO-8859-1\")\nX_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T08:52:46.161750Z","iopub.execute_input":"2022-07-09T08:52:46.162073Z","iopub.status.idle":"2022-07-09T08:52:47.986182Z","shell.execute_reply.started":"2022-07-09T08:52:46.162045Z","shell.execute_reply":"2022-07-09T08:52:47.985048Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = pd.read_csv('/kaggle/input/jigsaw-toxic-comment-classification-challenge/test_labels.csv.zip')\ny_test.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T08:52:47.987696Z","iopub.execute_input":"2022-07-09T08:52:47.988061Z","iopub.status.idle":"2022-07-09T08:52:48.201015Z","shell.execute_reply.started":"2022-07-09T08:52:47.988033Z","shell.execute_reply":"2022-07-09T08:52:48.199932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# sample text to visualize","metadata":{}},{"cell_type":"code","source":"train_df.sample(1)['comment_text'].values[0]","metadata":{"execution":{"iopub.status.busy":"2022-07-09T08:52:48.202945Z","iopub.execute_input":"2022-07-09T08:52:48.203239Z","iopub.status.idle":"2022-07-09T08:52:48.223210Z","shell.execute_reply.started":"2022-07-09T08:52:48.203211Z","shell.execute_reply":"2022-07-09T08:52:48.221953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#remove hyperlinks\n#remove contractions\n#remove punctuation\n#lemmatization","metadata":{"execution":{"iopub.status.busy":"2022-07-09T08:52:48.224443Z","iopub.execute_input":"2022-07-09T08:52:48.225034Z","iopub.status.idle":"2022-07-09T08:52:48.229919Z","shell.execute_reply.started":"2022-07-09T08:52:48.224991Z","shell.execute_reply":"2022-07-09T08:52:48.228699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import nltk","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:39.209032Z","iopub.execute_input":"2022-07-09T09:00:39.209392Z","iopub.status.idle":"2022-07-09T09:00:39.215868Z","shell.execute_reply.started":"2022-07-09T09:00:39.209363Z","shell.execute_reply":"2022-07-09T09:00:39.214530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import re\nfrom nltk.stem import WordNetLemmatizer\nfrom nltk.tokenize import word_tokenize\nfrom nltk.corpus import stopwords\nnltk.download('stopwords')\nstop_words = stopwords.words('english')","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:39.591754Z","iopub.execute_input":"2022-07-09T09:00:39.593024Z","iopub.status.idle":"2022-07-09T09:00:39.747441Z","shell.execute_reply.started":"2022-07-09T09:00:39.592976Z","shell.execute_reply":"2022-07-09T09:00:39.746360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install contractions","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:40.114620Z","iopub.execute_input":"2022-07-09T09:00:40.115200Z","iopub.status.idle":"2022-07-09T09:00:48.830502Z","shell.execute_reply.started":"2022-07-09T09:00:40.115163Z","shell.execute_reply":"2022-07-09T09:00:48.829343Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import contractions\nimport string","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:48.833207Z","iopub.execute_input":"2022-07-09T09:00:48.834082Z","iopub.status.idle":"2022-07-09T09:00:48.839714Z","shell.execute_reply.started":"2022-07-09T09:00:48.834035Z","shell.execute_reply":"2022-07-09T09:00:48.838627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def remove_contractions(sent):\n    # creating an empty list\n    expanded_words = []   \n    for word in sent.split(\" \"):\n      # using contractions.fix to expand the shortened words\n        expanded_words.append(contractions.fix(word)) \n\n    return ' '.join(expanded_words)\n\n\ndef to_lowercase(text):\n    return text.lower()\n\n# Remove website links\ndef remove_links(text):\n    template = re.compile(r'https?://\\S+|www\\.\\S+') \n    text = template.sub(r'', text)\n    return text\n\n# Remove HTML tags\ndef remove_html(text):\n    template = re.compile(r'<[^>]*>') \n    text = template.sub(r'', text)\n    return text\n\n\n# Remove stopwords\ndef remove_stopwords(words, stop_words):\n    return [word for word in words if word not in stop_words]\n\n# Remove none ascii characters\ndef remove_non_ascii(text):\n    template = re.compile(r'[^\\x00-\\x7E]+') \n    text = template.sub(r'', text)\n    return text\n\n# Replace none printable characters\ndef remove_non_printable(text):\n    template = re.compile(r'[\\x00-\\x0F]+') \n    text = template.sub(r' ', text)\n    return text\n\n# Remove special characters\ndef remove_special_chars(text):\n        text = re.sub(\"'s\", '', text)\n        template = re.compile('[\"#$%&\\'()\\*\\+-/:;<=>@\\[\\]\\\\\\\\^_`{|}~]') \n        text = template.sub(r' ', text)\n        return text\n\n# Replace multiple punctuation \ndef replace_multiplt_punc(text):\n        text = re.sub('[.!?]{2,}', '.', text)\n        text = re.sub(',+', ',', text) \n        return text\n\n    # Remove numbers\ndef remove_numbers(text):\n        text = re.sub('\\d+', ' ', text)\n        return text\n\ndef handle_spaces(text):\n    # Remove extra spaces\n    text = re.sub('\\s+', ' ', text)\n    \n    # Remove spaces at the beginning and at the end of string\n    text = text.strip() \n    \n    return text\n\ndef remove_punctuation(text):\n    \"\"\"Remove punctuation from list of tokenized words\"\"\"\n    translator = str.maketrans('', '', string.punctuation)\n    return text.translate(translator)\n\n# def stem_words(words):\n#     \"\"\"Stem words in text\"\"\"\n#     stemmer = PorterStemmer()\n#     return [stemmer.stem(word) for word in words]\n\ndef text2words(text):\n      return word_tokenize(text)\n    \ndef lemmatize_words(words):\n    \"\"\"Lemmatize words in text\"\"\"\n\n    lemmatizer = WordNetLemmatizer()\n    return [lemmatizer.lemmatize(word) for word in words]\n\ndef lemmatize_verbs(words):\n    \"\"\"Lemmatize verbs in text\"\"\"\n\n    lemmatizer = WordNetLemmatizer()\n    return ([lemmatizer.lemmatize(word, pos='v') for word in words])\n\ndef remove_pattern(text): \n    # remove hi moron \n    text= re.sub(r'(hi)(.*)\\1', r'\\1', text)\n    # remove duplicate words\n    text= re.sub(r\"\\b(\\w+)(?:\\W+\\1\\b)+\",r'\\1', text,flags=re.IGNORECASE)\n    # remove [User:Cirt]] \n    text= re.sub(r\"\\[.*?\\]\", ' ', text)\n    # remove \\n\\n\n    text= re.sub(r\"\\n\", ' ', text)\n    return text\n\ndef clean_text( text):\n    text = remove_contractions(text)\n    text = remove_pattern(text)\n    text = remove_links(text)\n    text = remove_html(text)\n    text = remove_special_chars(text)\n    text = remove_non_ascii(text)\n    text = remove_non_printable(text)\n    text = remove_numbers(text)\n    text = remove_punctuation(text)\n    text = to_lowercase(text)\n    text = handle_spaces(text)\n    words = text2words(text)\n    words = remove_stopwords(words, stop_words)\n    #words = stem_words(words) #either stem or lemmatize\n    words = lemmatize_words(words)\n    words = lemmatize_verbs(words)\n\n    return ' '.join(words)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:48.841062Z","iopub.execute_input":"2022-07-09T09:00:48.841358Z","iopub.status.idle":"2022-07-09T09:00:48.858276Z","shell.execute_reply.started":"2022-07-09T09:00:48.841330Z","shell.execute_reply":"2022-07-09T09:00:48.857303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['comment_text'] = train_df['comment_text'].apply(lambda x: clean_text(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:00:48.860423Z","iopub.execute_input":"2022-07-09T09:00:48.860745Z","iopub.status.idle":"2022-07-09T09:03:03.323532Z","shell.execute_reply.started":"2022-07-09T09:00:48.860706Z","shell.execute_reply":"2022-07-09T09:03:03.322303Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:03:04.336045Z","iopub.execute_input":"2022-07-09T09:03:04.336846Z","iopub.status.idle":"2022-07-09T09:03:04.350908Z","shell.execute_reply.started":"2022-07-09T09:03:04.336789Z","shell.execute_reply":"2022-07-09T09:03:04.349628Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_test['comment_text'] = X_test['comment_text'].apply(lambda x: clean_text(x))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:03:38.030940Z","iopub.execute_input":"2022-07-09T09:03:38.031312Z","iopub.status.idle":"2022-07-09T09:05:37.220234Z","shell.execute_reply.started":"2022-07-09T09:03:38.031279Z","shell.execute_reply":"2022-07-09T09:05:37.219155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import string\n# string.punctuation","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:02:42.110951Z","iopub.execute_input":"2022-07-08T22:02:42.111689Z","iopub.status.idle":"2022-07-08T22:02:42.118476Z","shell.execute_reply.started":"2022-07-08T22:02:42.111641Z","shell.execute_reply":"2022-07-08T22:02:42.117372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def clean_text(text):\n#     cleaned_text = text.translate(str.maketrans('', '', string.punctuation))\n#     return cleaned_text","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:02:42.120024Z","iopub.execute_input":"2022-07-08T22:02:42.120383Z","iopub.status.idle":"2022-07-08T22:02:42.128462Z","shell.execute_reply.started":"2022-07-08T22:02:42.120354Z","shell.execute_reply":"2022-07-08T22:02:42.127728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df['comment_text'] = train_df['comment_text'].apply(clean_text)\n# X_test['comment_text'] = X_test['comment_text'].apply(clean_text)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:02:42.129752Z","iopub.execute_input":"2022-07-08T22:02:42.130238Z","iopub.status.idle":"2022-07-08T22:02:46.031434Z","shell.execute_reply.started":"2022-07-08T22:02:42.130199Z","shell.execute_reply":"2022-07-08T22:02:46.030263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import re\n# train_df['comment_text'] = train_df['comment_text'].apply(lambda s: re.sub(r'[0-9]+', '', s) )\n# train_df['comment_text'] = train_df['comment_text'].apply(lambda s: re.sub(r'[0-9]+', '', s) )","metadata":{"execution":{"iopub.status.busy":"2022-07-08T22:02:46.03278Z","iopub.execute_input":"2022-07-08T22:02:46.03312Z","iopub.status.idle":"2022-07-08T22:02:48.857151Z","shell.execute_reply.started":"2022-07-08T22:02:46.033089Z","shell.execute_reply":"2022-07-08T22:02:48.855907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.preprocessing.text import Tokenizer\n\n# We create a tokenizer, configured to only take\n# into account the top-1000 most common words\ntokenizer = Tokenizer(num_words=1000, oov_token='UNK')\n# This builds the word index\ntokenizer.fit_on_texts(train_df['comment_text'])\n\n# This turns strings into lists of integer indices.\nsequences = tokenizer.texts_to_sequences(train_df['comment_text'])\n\n# You could also directly get the one-hot binary representations.\n# Note that other vectorization modes than one-hot encoding are supported!\none_hot_results = tokenizer.texts_to_matrix(train_df['comment_text'], mode='binary')\n#tfidf = tokenizer.texts_to_matrix(train_df['comment_text'], mode='tfidf')\n# This is how you can recover the word index that was computed\nword_index = tokenizer.word_index\nprint('Found %s unique tokens.' % len(word_index))\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:23:58.114653Z","iopub.execute_input":"2022-07-09T09:23:58.115086Z","iopub.status.idle":"2022-07-09T09:24:11.800489Z","shell.execute_reply.started":"2022-07-09T09:23:58.115047Z","shell.execute_reply":"2022-07-09T09:24:11.799478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"one_hot_results.shape","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:24:11.802480Z","iopub.execute_input":"2022-07-09T09:24:11.803303Z","iopub.status.idle":"2022-07-09T09:24:11.809940Z","shell.execute_reply.started":"2022-07-09T09:24:11.803255Z","shell.execute_reply":"2022-07-09T09:24:11.808932Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This turns strings into lists of integer indices.\nsequences_test = tokenizer.texts_to_sequences(X_test['comment_text'])\n\n# You could also directly get the one-hot binary representations.\n# Note that other vectorization modes than one-hot encoding are supported!\none_hot_results_test = tokenizer.texts_to_matrix(X_test['comment_text'], mode='binary')\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:24:11.811239Z","iopub.execute_input":"2022-07-09T09:24:11.811737Z","iopub.status.idle":"2022-07-09T09:24:21.135760Z","shell.execute_reply.started":"2022-07-09T09:24:11.811704Z","shell.execute_reply":"2022-07-09T09:24:21.134488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import models\nfrom tensorflow.keras import layers\n\nmodel = models.Sequential()\nmodel.add(layers.Dense(64, activation='relu', input_shape=(1000,)))\nmodel.add(layers.Dense(32, activation='relu'))\nmodel.add(layers.Dense(16, activation='relu'))\nmodel.add(layers.Dense(6, activation='sigmoid'))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:24:21.140387Z","iopub.execute_input":"2022-07-09T09:24:21.140692Z","iopub.status.idle":"2022-07-09T09:24:21.197381Z","shell.execute_reply.started":"2022-07-09T09:24:21.140664Z","shell.execute_reply":"2022-07-09T09:24:21.196385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import optimizers\n\nmodel.compile(optimizer=optimizers.RMSprop(learning_rate=0.001),\n              loss='binary_crossentropy',\n              metrics=['accuracy'])\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:24:21.198700Z","iopub.execute_input":"2022-07-09T09:24:21.199024Z","iopub.status.idle":"2022-07-09T09:24:21.210808Z","shell.execute_reply.started":"2022-07-09T09:24:21.198996Z","shell.execute_reply":"2022-07-09T09:24:21.209837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(one_hot_results,\n                    train_df[['toxic' ,'severe_toxic' ,'obscene' ,'threat' ,'insult' ,'identity_hate']],\n                    epochs=4,\n                    batch_size=256,\n                    validation_split=0.2)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:25:15.173459Z","iopub.execute_input":"2022-07-09T09:25:15.174032Z","iopub.status.idle":"2022-07-09T09:25:23.360647Z","shell.execute_reply.started":"2022-07-09T09:25:15.173983Z","shell.execute_reply":"2022-07-09T09:25:23.359847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history_dict = history.history\nhistory_dict.keys()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:25:23.362537Z","iopub.execute_input":"2022-07-09T09:25:23.363599Z","iopub.status.idle":"2022-07-09T09:25:23.370701Z","shell.execute_reply.started":"2022-07-09T09:25:23.363556Z","shell.execute_reply":"2022-07-09T09:25:23.369533Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nacc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(1, len(acc) + 1)\n\n# \"bo\" is for \"blue dot\"\nplt.plot(epochs, loss, 'bo', label='Training loss')\n# b is for \"solid blue line\"\nplt.plot(epochs, val_loss, 'b', label='Validation loss')\nplt.title('Training and validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:25:23.372089Z","iopub.execute_input":"2022-07-09T09:25:23.372818Z","iopub.status.idle":"2022-07-09T09:25:23.590735Z","shell.execute_reply.started":"2022-07-09T09:25:23.372773Z","shell.execute_reply":"2022-07-09T09:25:23.589890Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.clf()   # clear figure\nacc_values = history_dict['accuracy']\nval_acc_values = history_dict['accuracy']\n\nplt.plot(epochs, acc, 'bo', label='Training acc')\nplt.plot(epochs, val_acc, 'b', label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:25:23.593440Z","iopub.execute_input":"2022-07-09T09:25:23.593927Z","iopub.status.idle":"2022-07-09T09:25:23.785166Z","shell.execute_reply.started":"2022-07-09T09:25:23.593897Z","shell.execute_reply":"2022-07-09T09:25:23.784422Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.read_csv('/kaggle/input/jigsaw-toxic-comment-classification-challenge/sample_submission.csv.zip')\ny_pred = model.predict(one_hot_results_test)\ny_pred = pd.DataFrame(y_pred)#.applymap(lambda x: 1 if x>0.5 else 0)\ny_pred.columns = ['toxic' ,'severe_toxic' ,'obscene' ,'threat' ,'insult' ,'identity_hate']\nsubmit[['toxic' ,'severe_toxic' ,'obscene' ,'threat' ,'insult' ,'identity_hate']] = y_pred\nsubmit.to_csv('submit_file.csv', index=None)\n#Score: 0.91","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:25:23.786110Z","iopub.execute_input":"2022-07-09T09:25:23.786996Z","iopub.status.idle":"2022-07-09T09:25:30.707495Z","shell.execute_reply.started":"2022-07-09T09:25:23.786950Z","shell.execute_reply":"2022-07-09T09:25:30.705998Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# LSTM","metadata":{}},{"cell_type":"code","source":"vocab_size = len(tokenizer.word_index) + 1\nvocab_size","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:25:30.710913Z","iopub.execute_input":"2022-07-09T09:25:30.711339Z","iopub.status.idle":"2022-07-09T09:25:30.719320Z","shell.execute_reply.started":"2022-07-09T09:25:30.711305Z","shell.execute_reply":"2022-07-09T09:25:30.718209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.preprocessing.sequence import pad_sequences\nlengths = [len(sequence) for sequence in sequences]\nmax_length = max(lengths)\nsequences_pad = pad_sequences(sequences, maxlen=50)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:25:30.721597Z","iopub.execute_input":"2022-07-09T09:25:30.721956Z","iopub.status.idle":"2022-07-09T09:25:31.263487Z","shell.execute_reply.started":"2022-07-09T09:25:30.721925Z","shell.execute_reply":"2022-07-09T09:25:31.262268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mean_lengths = np.mean(lengths)\nmean_lengths","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:26:35.841638Z","iopub.execute_input":"2022-07-09T09:26:35.842061Z","iopub.status.idle":"2022-07-09T09:26:35.863391Z","shell.execute_reply.started":"2022-07-09T09:26:35.842022Z","shell.execute_reply":"2022-07-09T09:26:35.862205Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nplt.hist(lengths, bins=100);\nplt.xlim([min(lengths), max(lengths)-1000]);","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:27:11.574874Z","iopub.execute_input":"2022-07-09T09:27:11.575639Z","iopub.status.idle":"2022-07-09T09:27:12.377293Z","shell.execute_reply.started":"2022-07-09T09:27:11.575602Z","shell.execute_reply":"2022-07-09T09:27:12.376434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seq_length = sequences_pad.shape[1]\nseq_length","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:25:33.713627Z","iopub.execute_input":"2022-07-09T09:25:33.714700Z","iopub.status.idle":"2022-07-09T09:25:33.722089Z","shell.execute_reply.started":"2022-07-09T09:25:33.714643Z","shell.execute_reply":"2022-07-09T09:25:33.720926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max_length = 40","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:49:49.230161Z","iopub.execute_input":"2022-07-09T09:49:49.230689Z","iopub.status.idle":"2022-07-09T09:49:49.235053Z","shell.execute_reply.started":"2022-07-09T09:49:49.230656Z","shell.execute_reply":"2022-07-09T09:49:49.233861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = models.Sequential()\nmodel.add(layers.Embedding(vocab_size, 50, input_length=max_length))\nmodel.add(layers.LSTM(128, dropout=0.2, recurrent_dropout=0.2, return_sequences=True))\nmodel.add(layers.LSTM(64,dropout=0.2, recurrent_dropout=0.2,))\nmodel.add(layers.Dense(16, activation='relu'))\nmodel.add(layers.Dense(6, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:27:22.947327Z","iopub.execute_input":"2022-07-09T09:27:22.948056Z","iopub.status.idle":"2022-07-09T09:27:23.246862Z","shell.execute_reply.started":"2022-07-09T09:27:22.948018Z","shell.execute_reply":"2022-07-09T09:27:23.245914Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=optimizers.RMSprop(learning_rate=0.01),\n              loss='binary_crossentropy',\n              metrics=['accuracy'])\n\nhistory = model.fit(sequences_pad,\n                    train_df[['toxic' ,'severe_toxic' ,'obscene' ,'threat' ,'insult' ,'identity_hate']],\n                    epochs=4,\n                    batch_size=256,\n                    validation_split=0.2)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:27:34.235822Z","iopub.execute_input":"2022-07-09T09:27:34.236225Z","iopub.status.idle":"2022-07-09T09:49:48.698254Z","shell.execute_reply.started":"2022-07-09T09:27:34.236192Z","shell.execute_reply":"2022-07-09T09:49:48.697112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(1, len(acc) + 1)\n\n# \"bo\" is for \"blue dot\"\nplt.plot(epochs, loss, 'bo', label='Training loss')\n# b is for \"solid blue line\"\nplt.plot(epochs, val_loss, 'b', label='Validation loss')\nplt.title('Training and validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:49:48.700685Z","iopub.execute_input":"2022-07-09T09:49:48.701510Z","iopub.status.idle":"2022-07-09T09:49:48.845666Z","shell.execute_reply.started":"2022-07-09T09:49:48.701476Z","shell.execute_reply":"2022-07-09T09:49:48.844839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.clf()   # clear figure\nacc_values = history_dict['accuracy']\nval_acc_values = history_dict['accuracy']\n\nplt.plot(epochs, acc, 'bo', label='Training acc')\nplt.plot(epochs, val_acc, 'b', label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:49:48.847138Z","iopub.execute_input":"2022-07-09T09:49:48.848364Z","iopub.status.idle":"2022-07-09T09:49:48.993341Z","shell.execute_reply.started":"2022-07-09T09:49:48.848320Z","shell.execute_reply":"2022-07-09T09:49:48.992407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit = pd.read_csv('/kaggle/input/jigsaw-toxic-comment-classification-challenge/sample_submission.csv.zip')\nsubmit.head()","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:49:48.996204Z","iopub.execute_input":"2022-07-09T09:49:48.996833Z","iopub.status.idle":"2022-07-09T09:49:49.228809Z","shell.execute_reply.started":"2022-07-09T09:49:48.996798Z","shell.execute_reply":"2022-07-09T09:49:49.227576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequences_test_pad =  pad_sequences(sequences_test, maxlen=max_length)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:49:49.236276Z","iopub.execute_input":"2022-07-09T09:49:49.236542Z","iopub.status.idle":"2022-07-09T09:49:49.792622Z","shell.execute_reply.started":"2022-07-09T09:49:49.236518Z","shell.execute_reply":"2022-07-09T09:49:49.791531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = model.predict(sequences_test_pad)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:49:49.794223Z","iopub.execute_input":"2022-07-09T09:49:49.795399Z","iopub.status.idle":"2022-07-09T09:54:25.738831Z","shell.execute_reply.started":"2022-07-09T09:49:49.795344Z","shell.execute_reply":"2022-07-09T09:54:25.737625Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred = pd.DataFrame(y_pred)#.applymap(lambda x: 1 if x>0.5 else 0)\ny_pred.columns = ['toxic' ,'severe_toxic' ,'obscene' ,'threat' ,'insult' ,'identity_hate']","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:54:25.740738Z","iopub.execute_input":"2022-07-09T09:54:25.741069Z","iopub.status.idle":"2022-07-09T09:54:25.746220Z","shell.execute_reply.started":"2022-07-09T09:54:25.741040Z","shell.execute_reply":"2022-07-09T09:54:25.745340Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit[['toxic' ,'severe_toxic' ,'obscene' ,'threat' ,'insult' ,'identity_hate']] = y_pred","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:54:25.747454Z","iopub.execute_input":"2022-07-09T09:54:25.748026Z","iopub.status.idle":"2022-07-09T09:54:25.775529Z","shell.execute_reply.started":"2022-07-09T09:54:25.747995Z","shell.execute_reply":"2022-07-09T09:54:25.774750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submit.to_csv('submit_lstm.csv', index=None)#Score: 0.65039","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:54:25.779677Z","iopub.execute_input":"2022-07-09T09:54:25.780068Z","iopub.status.idle":"2022-07-09T09:54:26.877262Z","shell.execute_reply.started":"2022-07-09T09:54:25.780038Z","shell.execute_reply":"2022-07-09T09:54:26.876180Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Bidirectional GRU","metadata":{}},{"cell_type":"code","source":"model = models.Sequential()\nmodel.add(layers.Embedding(vocab_size, 100, input_length=seq_length))\nmodel.add(layers.Bidirectional(layers.GRU(256, return_sequences=True)))\nmodel.add(layers.Bidirectional(layers.GRU(128)))\nmodel.add(layers.Dropout(0.4))\nmodel.add(layers.Dense(64, activation='relu'))\nmodel.add(layers.Dropout(0.2))\nmodel.add(layers.Dense(16, activation='relu'))\nmodel.add(layers.Dense(6, activation='sigmoid'))","metadata":{"execution":{"iopub.status.busy":"2022-07-09T09:59:50.988133Z","iopub.execute_input":"2022-07-09T09:59:50.988609Z","iopub.status.idle":"2022-07-09T09:59:51.908721Z","shell.execute_reply.started":"2022-07-09T09:59:50.988571Z","shell.execute_reply":"2022-07-09T09:59:51.907803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=optimizers.RMSprop(learning_rate=0.01),\n              loss='binary_crossentropy',\n              metrics=['accuracy'])\n\nhistory = model.fit(sequences_pad,\n                    train_df[['toxic' ,'severe_toxic' ,'obscene' ,'threat' ,'insult' ,'identity_hate']],\n                    epochs=5,\n                    batch_size=256,\n                    validation_split=0.2)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T11:07:25.305904Z","iopub.execute_input":"2022-07-09T11:07:25.306328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"acc = history.history['accuracy']\nval_acc = history.history['val_accuracy']\nloss = history.history['loss']\nval_loss = history.history['val_loss']\n\nepochs = range(1, len(acc) + 1)\n\n# \"bo\" is for \"blue dot\"\nplt.plot(epochs, loss, 'bo', label='Training loss')\n# b is for \"solid blue line\"\nplt.plot(epochs, val_loss, 'b', label='Validation loss')\nplt.title('Training and validation loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:39:17.748269Z","iopub.execute_input":"2022-07-09T10:39:17.748744Z","iopub.status.idle":"2022-07-09T10:39:17.970301Z","shell.execute_reply.started":"2022-07-09T10:39:17.748684Z","shell.execute_reply":"2022-07-09T10:39:17.968961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.clf()   # clear figure\nacc_values = history_dict['accuracy']\nval_acc_values = history_dict['accuracy']\n\nplt.plot(epochs, acc, 'bo', label='Training acc')\nplt.plot(epochs, val_acc, 'b', label='Validation acc')\nplt.title('Training and validation accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\n\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:39:17.971921Z","iopub.execute_input":"2022-07-09T10:39:17.972263Z","iopub.status.idle":"2022-07-09T10:39:18.185781Z","shell.execute_reply.started":"2022-07-09T10:39:17.972228Z","shell.execute_reply":"2022-07-09T10:39:18.184987Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sequences_test_pad =  pad_sequences(sequences_test, maxlen=max_length)\ny_pred = model.predict(sequences_test_pad)\ny_pred = pd.DataFrame(y_pred)#.applymap(lambda x: 1 if x>0.5 else 0)\ny_pred.columns = ['toxic' ,'severe_toxic' ,'obscene' ,'threat' ,'insult' ,'identity_hate']\nsubmit[['toxic' ,'severe_toxic' ,'obscene' ,'threat' ,'insult' ,'identity_hate']] = y_pred\nsubmit.to_csv('submit_gru.csv', index=None)","metadata":{"execution":{"iopub.status.busy":"2022-07-09T10:55:00.701447Z","iopub.execute_input":"2022-07-09T10:55:00.702454Z","iopub.status.idle":"2022-07-09T11:01:10.622612Z","shell.execute_reply.started":"2022-07-09T10:55:00.702410Z","shell.execute_reply":"2022-07-09T11:01:10.621479Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}