{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from IPython.display import clear_output","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:11:00.486210Z","iopub.execute_input":"2022-01-12T11:11:00.486631Z","iopub.status.idle":"2022-01-12T11:11:00.515666Z","shell.execute_reply.started":"2022-01-12T11:11:00.486532Z","shell.execute_reply":"2022-01-12T11:11:00.514991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <font size=4>Set up notebook</font>","metadata":{}},{"cell_type":"markdown","source":"<font size=4>Check for packages update</font>","metadata":{}},{"cell_type":"code","source":"!pip install contractions\n!pip install scikit-learn  -U\n!pip install nltk  -U\n!pip install spacy  -U\n!pip install emoji -U\n!pip install transformers -U\nclear_output()","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:11:00.517473Z","iopub.execute_input":"2022-01-12T11:11:00.517978Z","iopub.status.idle":"2022-01-12T11:12:29.882543Z","shell.execute_reply.started":"2022-01-12T11:11:00.517934Z","shell.execute_reply":"2022-01-12T11:12:29.881416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python -m spacy download en_core_web_md\nclear_output()","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:12:29.884951Z","iopub.execute_input":"2022-01-12T11:12:29.885331Z","iopub.status.idle":"2022-01-12T11:12:51.379801Z","shell.execute_reply.started":"2022-01-12T11:12:29.885279Z","shell.execute_reply":"2022-01-12T11:12:51.378974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4>Declare dictionaries with emoticons and abbreviation</font>","metadata":{}},{"cell_type":"code","source":"EMOTICONS = {\n    u\":‑\\)\":\"Happy\",\n    u\":\\)\":\"Happy\",\n    u\":-\\]\":\"Happy\",\n    u\":\\]\":\"Happy\",\n    u\":-3\":\"Happy\",\n    u\":3\":\"Happy\",\n    u\":->\":\"Happy\",\n    u\":>\":\"Happy\",\n    u\"8-\\)\":\"Happy\",\n    u\":o\\)\":\"Happy\",\n    u\":-\\}\":\"Happy\",\n    u\":\\}\":\"Happy\",\n    u\":-\\)\":\"Happy\",\n    u\":c\\)\":\"Happy\",\n    u\":\\^\\)\":\"Happy\",\n    u\"=\\]\":\"Happy\",\n    u\"=\\)\":\"Happy\",\n    u\":‑D\":\"Laughing\",\n    u\":D\":\"Laughing\",\n    u\"8‑D\":\"Laughing\",\n    u\"8D\":\"Laughing\",\n    u\"X‑D\":\"Laughing\",\n    u\"XD\":\"Laughing\",\n    u\"=D\":\"Laughing\",\n    u\"=3\":\"Laughing\",\n    u\"B\\^D\":\"Laughing\",\n    u\":-\\)\\)\":\"Very happy\",\n    u\":‑\\(\":\"sad\",\n    u\":-\\(\":\"sad\",\n    u\":\\(\":\"sad\",\n    u\":‑c\":\"sad\",\n    u\":c\":\"sad,\",\n    u\":‑<\":\"sad,\",\n    u\":<\":\"sad,\",\n    u\":‑\\[\":\"sad\",\n    u\":\\[\":\"sad\",\n    u\":-\\|\\|\":\"sad\",\n    u\">:\\[\":\"sad\",\n    u\":\\{\":\"sad\",\n    u\":@\":\"sad\",\n    u\">:\\(\":\"sad\",\n    u\":'‑\\(\":\"Crying\",\n    u\":'\\(\":\"Crying\",\n    u\":'‑\\)\":\"happiness\",\n    u\":'\\)\":\"happiness\",\n    u\"D‑':\":\"Horror\",\n    u\"D:<\":\"Disgust\",\n    u\"D:\":\"Sadness\",\n    u\"D8\":\"dismay\",\n    u\"D;\":\"dismay\",\n    u\"D=\":\"dismay\",\n    u\"DX\":\"dismay\",\n    u\":‑O\":\"Surprise\",\n    u\":O\":\"Surprise\",\n    u\":‑o\":\"Surprise\",\n    u\":o\":\"Surprise\",\n    u\":-0\":\"Shock\",\n    u\"8‑0\":\"Yawn\",\n    u\">:O\":\"Yawn\",\n    u\":-\\*\":\"Kiss\",\n    u\":\\*\":\"Kiss\",\n    u\":X\":\"Kiss\",\n    u\";‑\\)\":\"Wink\",\n    u\";\\)\":\"Wink\",\n    u\"\\*-\\)\":\"Wink\",\n    u\"\\*\\)\":\"Wink\",\n    u\";‑\\]\":\"Wink\",\n    u\";\\]\":\"Wink\",\n    u\";\\^\\)\":\"Wink\",\n    u\":‑,\":\"Wink\",\n    u\";D\":\"Wink\",\n    u\":‑P\":\"playful\",\n    u\":P\":\"playful\",\n    u\"X‑P\":\"playful\",\n    u\"XP\":\"playful\",\n    u\":‑Þ\":\"playful\",\n    u\":Þ\":\"playful\",\n    u\":b\":\"playful\",\n    u\"d:\":\"playful\",\n    u\"=p\":\"playful\",\n    u\">:P\":\"playful\",\n    u\":‑/\":\"uneasy\",\n    u\":/\":\"uneasy\",\n    u\":-[.]\":\"uneasy\",\n    u\">:[(\\\\\\)]\":\"uneasy\",\n    u\">:/\":\"uneasy\",\n    u\":[(\\\\\\)]\":\"uneasy\",\n    u\"=/\":\"uneasy\",\n    u\"=[(\\\\\\)]\":\"uneasy\",\n    u\":L\":\"uneasy\",\n    u\"=L\":\"uneasy\",\n    u\":S\":\"uneasy\",\n    u\":‑\\|\":\"Straight face\",\n    u\":\\|\":\"Straight face\",\n    u\":$\":\"Embarrassed\",\n    u\":‑x\":\"Sealed lips\",\n    u\":x\":\"Sealed lips\",\n    u\":‑#\":\"Sealed lips\",\n    u\":#\":\"Sealed lips\",\n    u\":‑&\":\"Sealed lips\",\n    u\":&\":\"Sealed lips\",\n    u\"O:‑\\)\":\"innocent\",\n    u\"O:\\)\":\"innocent\",\n    u\"0:‑3\":\"innocent\",\n    u\"0:3\":\"innocent\",\n    u\"0:‑\\)\":\"innocent\",\n    u\"0:\\)\":\"innocent\",\n    u\":‑b\":\"playful\",\n    u\"0;\\^\\)\":\"innocent\",\n    u\">:‑\\)\":\"Evil\",\n    u\">:\\)\":\"Evil\",\n    u\"\\}:‑\\)\":\"Evil\",\n    u\"\\}:\\)\":\"Evil\",\n    u\"3:‑\\)\":\"Evil\",\n    u\"3:\\)\":\"Evil\",\n    u\">;\\)\":\"Evil\",\n    u\"\\|;‑\\)\":\"Cool\",\n    u\"\\|‑O\":\"Bored\",\n    u\":‑J\":\"Tongue-in-cheek\",\n    u\"#‑\\)\":\"Party all night\",\n    u\"%‑\\)\":\"Drunk or confused\",\n    u\"%\\)\":\"Drunk or confused\",\n    u\":-###..\":\"Being sick\",\n    u\":###..\":\"Being sick\",\n    u\"<:‑\\|\":\"Dump\",\n    u\"\\(>_<\\)\":\"Troubled\",\n    u\"\\(>_<\\)>\":\"Troubled\",\n    u\"\\(';'\\)\":\"Baby\",\n    u\"\\(\\^\\^>``\":\"Embarrassed\",\n    u\"\\(\\^_\\^;\\)\":\"Embarrassed\",\n    u\"\\(-_-;\\)\":\"Embarrassed\",\n    u\"\\(~_~;\\) \\(・\\.・;\\)\":\"Embarrassed\",\n    u\"\\(-_-\\)zzz\":\"Sleeping\",\n    u\"\\(\\^_-\\)\":\"Wink\",\n    u\"\\(\\(\\+_\\+\\)\\)\":\"Confused\",\n    u\"\\(\\+o\\+\\)\":\"Confused\",\n    u\"\\(o\\|o\\)\":\"Ultraman\",\n    u\"\\^_\\^\":\"Joyful\",\n    u\"\\(\\^_\\^\\)/\":\"Joyful\",\n    u\"\\(\\^O\\^\\)／\":\"Joyful\",\n    u\"\\(\\^o\\^\\)／\":\"Joyful\",\n    u\"\\(__\\)\":\"respect\",\n    u\"_\\(\\._\\.\\)_\":\"respect\",\n    u\"<\\(_ _\\)>\":\"respect\",\n    u\"<m\\(__\\)m>\":\"respect\",\n    u\"m\\(__\\)m\":\"respect\",\n    u\"m\\(_ _\\)m\":\"respect\",\n    u\"\\('_'\\)\":\"Sad\",\n    u\"\\(/_;\\)\":\"Sad\",\n    u\"\\(T_T\\) \\(;_;\\)\":\"Sad\",\n    u\"\\(;_;\":\"Sad\",\n    u\"\\(;_:\\)\":\"Sad\",\n    u\"\\(;O;\\)\":\"Sad\",\n    u\"\\(:_;\\)\":\"Sad\",\n    u\"\\(ToT\\)\":\"Sad\",\n    u\";_;\":\"Sad\",\n    u\";-;\":\"Sad\",\n    u\";n;\":\"Sad\",\n    u\";;\":\"Sad\",\n    u\"Q\\.Q\":\"Sad\",\n    u\"T\\.T\":\"Sad\",\n    u\"QQ\":\"Sad\",\n    u\"Q_Q\":\"Sad\",\n    u\"\\(-\\.-\\)\":\"Shame\",\n    u\"\\(-_-\\)\":\"Shame\",\n    u\"\\(一一\\)\":\"Shame\",\n    u\"\\(；一_一\\)\":\"Shame\",\n    u\"\\(=_=\\)\":\"Tired\",\n    u\"\\(=\\^\\·\\^=\\)\":\"cat\",\n    u\"\\(=\\^\\·\\·\\^=\\)\":\"cat\",\n    u\"=_\\^=\t\":\"cat\",\n    u\"\\(\\.\\.\\)\":\"Looking down\",\n    u\"\\(\\._\\.\\)\":\"Looking down\",\n    u\"\\^m\\^\":\"Giggling\",\n    u\"\\(\\・\\・?\":\"Confusion\",\n    u\"\\(?_?\\)\":\"Confusion\",\n    u\">\\^_\\^<\":\"Laugh\",\n    u\"<\\^!\\^>\":\"Laugh\",\n    u\"\\^/\\^\":\"Laugh\",\n    u\"\\（\\*\\^_\\^\\*）\" :\"Laugh\",\n    u\"\\(\\^<\\^\\) \\(\\^\\.\\^\\)\":\"Laugh\",\n    u\"\\(^\\^\\)\":\"Laugh\",\n    u\"\\(\\^\\.\\^\\)\":\"Laugh\",\n    u\"\\(\\^_\\^\\.\\)\":\"Laugh\",\n    u\"\\(\\^_\\^\\)\":\"Laugh\",\n    u\"\\(\\^\\^\\)\":\"Laugh\",\n    u\"\\(\\^J\\^\\)\":\"Laugh\",\n    u\"\\(\\*\\^\\.\\^\\*\\)\":\"Laugh\",\n    u\"\\(\\^—\\^\\）\":\"Laugh\",\n    u\"\\(#\\^\\.\\^#\\)\":\"Laugh\",\n    u\"\\（\\^—\\^\\）\":\"Waving\",\n    u\"\\(;_;\\)/~~~\":\"Waving\",\n    u\"\\(\\^\\.\\^\\)/~~~\":\"Waving\",\n    u\"\\(-_-\\)/~~~ \\($\\·\\·\\)/~~~\":\"Waving\",\n    u\"\\(T_T\\)/~~~\":\"Waving\",\n    u\"\\(ToT\\)/~~~\":\"Waving\",\n    u\"\\(\\*\\^0\\^\\*\\)\":\"Excited\",\n    u\"\\(\\*_\\*\\)\":\"amazed\",\n    u\"\\(\\*_\\*;\":\"amazed\",\n    u\"\\(\\+_\\+\\) \\(@_@\\)\":\"amazed\",\n    u\"\\(\\*\\^\\^\\)v\":\"cheerful\",\n    u\"\\(\\^_\\^\\)v\":\"cheerful\",\n    u\"\\(\\(d[-_-]b\\)\\)\":\"music\",\n    u'\\(-\"-\\)':\"Worried\",\n    u\"\\(ーー;\\)\":\"Worried\",\n    u\"\\(\\^0_0\\^\\)\":\"Eyeglasses\",\n    u\"\\(\\＾ｖ\\＾\\)\":\"Happy\",\n    u\"\\(\\＾ｕ\\＾\\)\":\"Happy\",\n    u\"\\(\\^\\)o\\(\\^\\)\":\"Happy\",\n    u\"\\(\\^O\\^\\)\":\"Happy\",\n    u\"\\(\\^o\\^\\)\":\"Happy\",\n    u\"\\)\\^o\\^\\(\":\"Happy\",\n    u\":O o_O\":\"Surprised\",\n    u\"o_0\":\"Surprised\",\n    u\"o\\.O\":\"Surpised\",\n    u\"\\(o\\.o\\)\":\"Surprised\",\n    u\"oO\":\"Surprised\",\n    u\"\\(\\*￣m￣\\)\":\"Dissatisfied\",\n    u\"\\(‘A`\\)\":\"Deflated\"\n}\nchat_words_str = \"\"\"AFAIK=As Far As I Know\nAFK=Away From Keyboard\nASAP=As Soon As Possible\nATK=At The Keyboard\nATM=At The Moment\nA3=Anytime Anywhere Anyplace\nBAK=Back At Keyboard\nBBL=Be Back Later\nBBS=Be Back Soon\nBFN=Bye For Now\nB4N=Bye For Now\nBRB=Be Right Back\nBRT=Be Right There\nBTW=By The Way\nB4=Before\nB4N=Bye For Now\nCU=See You\nCUL8R=See You Later\nCYA=See You\nFAQ=Frequently Asked Questions\nFC=Fingers Crossed\nFWIW=For What It is Worth\nFYI=For Your Information\nGAL=Get A Life\nGG=Good Game\nGN=Good Night\nGMTA=Great Minds Think Alike\nGR8=Great\nG9=Genius\nIC=I See\nICQ=I Seek you (also a chat program)\nILU=ILU: I Love You\nIMHO=In My Honest Opinion\nIMO=In My Opinion\nIOW=In Other Words\nIRL=In Real Life\nKISS=Keep It Simple, Stupid\nLDR=Long Distance Relationship\nLMAO=Laugh My Ass Off\nLOL=Laughing Out Loud\nLTNS=Long Time No See\nL8R=Later\nMTE=My Thoughts Exactly\nM8=Mate\nNRN=No Reply Necessary\nOIC=Oh I See\nPITA=Pain In The Ass\nPRT=Party\nPRW=Parents Are Watching\nROFL=Rolling On The Floor Laughing\nROFLOL=Rolling On The Floor Laughing Out Loud\nROTFLMAO=Rolling On The Floor Laughing My Ass Off\nSK8=Skate\nSTATS=Your sex and age\nASL=Age Sex Location\nTHX=Thank You\nTTFN=Ta Ta For Now\nTTYL=Talk To You Later\nU=You\nU2=You Too\nU4E=Yours For Ever\nWB=Welcome Back\nWTF=What The Fuck\nWTG=Way To Go\nWUF=Where Are You From\nW8=Wait\n7K=Sick:-D Laugher\"\"\"","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-01-12T11:27:46.800740Z","iopub.execute_input":"2022-01-12T11:27:46.801959Z","iopub.status.idle":"2022-01-12T11:27:46.837153Z","shell.execute_reply.started":"2022-01-12T11:27:46.801880Z","shell.execute_reply":"2022-01-12T11:27:46.836407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4>Imports</font>","metadata":{}},{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport re\nfrom tqdm import tqdm\nimport emoji\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.feature_extraction.text import TfidfTransformer, CountVectorizer\nfrom sklearn.metrics import accuracy_score, classification_report, precision_score, roc_auc_score\nfrom wordcloud import WordCloud\nfrom sklearn.neighbors import KNeighborsClassifier\nfrom sklearn.ensemble import RandomForestClassifier\nfrom sklearn.linear_model import LogisticRegression\n\n# For BERT\nimport tensorflow as tf\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\nfrom tokenizers import BertWordPieceTokenizer\n\nimport nltk\nfrom nltk.corpus import stopwords\nimport contractions as ct\nimport spacy\nimport string\nfrom tqdm import tqdm, tqdm_notebook\nfrom gensim.models import KeyedVectors, fasttext\n\ntqdm.pandas()\nnlp = spacy.load('en_core_web_md')\nnltk.download(\"stopwords\")\nstops = set(stopwords.words(\"english\"))\nct_dict = ct.contractions_dict","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-01-12T11:27:47.550555Z","iopub.execute_input":"2022-01-12T11:27:47.550859Z","iopub.status.idle":"2022-01-12T11:27:56.742487Z","shell.execute_reply.started":"2022-01-12T11:27:47.550823Z","shell.execute_reply":"2022-01-12T11:27:56.741416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"stops = stops.union(set([\"hi\", \"hello\"]));","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:27:56.745935Z","iopub.execute_input":"2022-01-12T11:27:56.746427Z","iopub.status.idle":"2022-01-12T11:27:56.751755Z","shell.execute_reply.started":"2022-01-12T11:27:56.746376Z","shell.execute_reply":"2022-01-12T11:27:56.751153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4>Auxiliary functions</font>","metadata":{}},{"cell_type":"code","source":"# Freeze layers from 0 to start and unfreeze next\ndef set_trainable(model, start):\n    layers_number = len(model.layers)\n    for i in range(0,start):\n        model.layers[i].trainable = False\n    for i in range(start, layers_number):\n        model.layers[i].trainable = True\n\n# Create callback to save best model\ndef save_best_callback(path, monitor=\"val_loss\", mode=\"min\"): \n    return tf.keras.callbacks.ModelCheckpoint(filepath=path, save_weights_only=True, monitor=monitor, mode=mode, save_best_only=True)\n\n# Plot history of .fit\ndef plot_history(history, metrics=None):\n    labels = []\n    df = pd.DataFrame(history)\n    ncols = len(df.columns)\n    fig, axs = plt.subplots(ncols=ncols // 2, nrows=1, figsize=(35,10))\n    if metrics is None:\n        metrics = set([i.replace(\"val_\", \"\") for i in history.keys()])\n    for metric in metrics:\n        labels.append([metric, f\"val_{metric}\"])\n    if ncols == 2:\n        sns.lineplot(data=df, ax=axs)\n        plt.xlabel(\"epoch\")\n    else:\n        for pair_ind in range(len(labels)):\n            sns.lineplot(data=df[labels[pair_ind]], ax=axs[pair_ind % ncols])\n            axs[pair_ind % ncols].set_xlabel(\"epoch\")\n    plt.show()\n\ndef plot_wordcloud(text, figsize=(15,15), **kwargs):\n    _, ax = plt.subplots(figsize=(15,15))\n    wordcloud = WordCloud(**kwargs).generate(text)\n    ax.imshow(wordcloud, interpolation='bilinear')\n    plt.axis(\"off\");\n\ndef expand_contractions(text):\n    return \" \".join([ct_dict[w] if w in ct_dict else w for w in text.split()])","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:27:56.753628Z","iopub.execute_input":"2022-01-12T11:27:56.754061Z","iopub.status.idle":"2022-01-12T11:27:56.770507Z","shell.execute_reply.started":"2022-01-12T11:27:56.754027Z","shell.execute_reply":"2022-01-12T11:27:56.769725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4>Load all data</font>","metadata":{}},{"cell_type":"code","source":"test_labels = pd.read_csv(\"../input/jigsaw-multilingual-toxic-comment-classification/test_labels.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:27:56.772002Z","iopub.execute_input":"2022-01-12T11:27:56.772442Z","iopub.status.idle":"2022-01-12T11:27:56.835030Z","shell.execute_reply.started":"2022-01-12T11:27:56.772409Z","shell.execute_reply":"2022-01-12T11:27:56.834360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"toxic_comments_raw_data = pd.read_csv(\"../input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\nunintended_bias_raw_data = pd.read_csv(\"../input/jigsaw-multilingual-toxic-comment-classification/jigsaw-unintended-bias-train.csv\")\nvalidation = pd.read_csv(\"../input/jigsaw-multilingual-toxic-comment-classification/validation.csv\")\ntest_raw_data = pd.read_csv(\"../input/jigsaw-multilingual-toxic-comment-classification/test.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:28:01.221593Z","iopub.execute_input":"2022-01-12T11:28:01.222397Z","iopub.status.idle":"2022-01-12T11:28:31.914913Z","shell.execute_reply.started":"2022-01-12T11:28:01.222353Z","shell.execute_reply":"2022-01-12T11:28:31.913994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#training_data_es = pd.read_csv(\"../input/jigsaw-train-multilingual-coments-google-api/jigsaw-toxic-comment-train-google-es.csv\", index_col=\"id\")\n#training_data_fr = pd.read_csv(\"../input/jigsaw-train-multilingual-coments-google-api/jigsaw-toxic-comment-train-google-fr.csv\", index_col=\"id\")\n#training_data_it = pd.read_csv(\"../input/jigsaw-train-multilingual-coments-google-api/jigsaw-toxic-comment-train-google-it.csv\", index_col=\"id\")\n#training_data_pt = pd.read_csv(\"../input/jigsaw-train-multilingual-coments-google-api/jigsaw-toxic-comment-train-google-pt.csv\", index_col=\"id\")\n#training_data_ru = pd.read_csv(\"../input/jigsaw-train-multilingual-coments-google-api/jigsaw-toxic-comment-train-google-ru.csv\", index_col=\"id\")\n#training_data_tr = pd.read_csv(\"../input/jigsaw-train-multilingual-coments-google-api/jigsaw-toxic-comment-train-google-tr.csv\", index_col=\"id\")","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:28:31.916567Z","iopub.execute_input":"2022-01-12T11:28:31.916829Z","iopub.status.idle":"2022-01-12T11:28:31.922137Z","shell.execute_reply.started":"2022-01-12T11:28:31.916800Z","shell.execute_reply":"2022-01-12T11:28:31.921335Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"translated_test_data = pd.read_csv(\"../input/jigsaw-multilingual-toxic-test-translated/jigsaw_miltilingual_test_translated.csv\", index_col=\"id\")\ntranslated_valid_data = pd.read_csv(\"../input/jigsaw-multilingual-toxic-test-translated/jigsaw_miltilingual_valid_translated.csv\", index_col=\"id\")","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:28:31.923833Z","iopub.execute_input":"2022-01-12T11:28:31.924370Z","iopub.status.idle":"2022-01-12T11:28:33.725103Z","shell.execute_reply.started":"2022-01-12T11:28:31.924302Z","shell.execute_reply":"2022-01-12T11:28:33.724153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4>Data Preprocessing</font>","metadata":{}},{"cell_type":"code","source":"unintended_bias_raw_data[\"toxic\"] = (unintended_bias_raw_data[\"toxic\"] > 0.5).astype(\"int32\")","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:28:36.369070Z","iopub.execute_input":"2022-01-12T11:28:36.370148Z","iopub.status.idle":"2022-01-12T11:28:36.603274Z","shell.execute_reply.started":"2022-01-12T11:28:36.370107Z","shell.execute_reply":"2022-01-12T11:28:36.602237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"toxic_comments_raw_data = toxic_comments_raw_data[[\"id\", \"comment_text\", \"toxic\"]]\nunintended_bias_raw_data = toxic_comments_raw_data[[\"id\", \"comment_text\", \"toxic\"]]","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:28:36.933095Z","iopub.execute_input":"2022-01-12T11:28:36.933469Z","iopub.status.idle":"2022-01-12T11:28:37.396011Z","shell.execute_reply.started":"2022-01-12T11:28:36.933435Z","shell.execute_reply":"2022-01-12T11:28:37.394664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Number of rows in toxic ds: {toxic_comments_raw_data.shape[0]}\")\nprint(f\"Number of rows in unintended bias ds: {unintended_bias_raw_data.shape[0]}\")","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:28:37.399235Z","iopub.execute_input":"2022-01-12T11:28:37.399670Z","iopub.status.idle":"2022-01-12T11:28:37.407793Z","shell.execute_reply.started":"2022-01-12T11:28:37.399632Z","shell.execute_reply":"2022-01-12T11:28:37.405739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"toxic_comments_raw_data[toxic_comments_raw_data[\"comment_text\"].str.len() < 4]","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:28:39.188278Z","iopub.execute_input":"2022-01-12T11:28:39.188955Z","iopub.status.idle":"2022-01-12T11:28:39.502174Z","shell.execute_reply.started":"2022-01-12T11:28:39.188913Z","shell.execute_reply":"2022-01-12T11:28:39.501154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_, axs = plt.subplots(ncols=2, figsize=(15,5))\nsns.histplot(toxic_comments_raw_data[\"toxic\"], ax=axs[0]);\nsns.histplot(toxic_comments_raw_data[\"toxic\"], ax=axs[1]);\n\naxs[0].set_title(\"Toxic comments data\")\naxs[1].set_title(\"Unintended bias data\");","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:28:39.588759Z","iopub.execute_input":"2022-01-12T11:28:39.589231Z","iopub.status.idle":"2022-01-12T11:28:40.590733Z","shell.execute_reply.started":"2022-01-12T11:28:39.589192Z","shell.execute_reply":"2022-01-12T11:28:40.589881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"big_data = pd.concat([toxic_comments_raw_data, unintended_bias_raw_data])","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:28:42.596421Z","iopub.execute_input":"2022-01-12T11:28:42.597458Z","iopub.status.idle":"2022-01-12T11:28:42.631855Z","shell.execute_reply.started":"2022-01-12T11:28:42.597408Z","shell.execute_reply":"2022-01-12T11:28:42.630949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_wordcloud(\" \".join(big_data[big_data[\"toxic\"] == 0].sample(100000)[\"comment_text\"].to_numpy()), width=2560, height=1440, figsize=(25, 25), max_words=300)","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:28:43.496095Z","iopub.execute_input":"2022-01-12T11:28:43.496613Z","iopub.status.idle":"2022-01-12T11:29:24.635854Z","shell.execute_reply.started":"2022-01-12T11:28:43.496560Z","shell.execute_reply":"2022-01-12T11:29:24.634828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_wordcloud(\" \".join(big_data[big_data[\"toxic\"] == 1][\"comment_text\"].to_numpy()), width=2560, height=1440, figsize=(25, 25), max_words=300)","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:29:24.637351Z","iopub.execute_input":"2022-01-12T11:29:24.637979Z","iopub.status.idle":"2022-01-12T11:29:43.440511Z","shell.execute_reply.started":"2022-01-12T11:29:24.637942Z","shell.execute_reply":"2022-01-12T11:29:43.439629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"big_data[\"comment_text\"] = big_data[\"comment_text\"].progress_apply(lambda x: expand_contractions(x.lower()))\nbig_data[\"comment_text\"] = big_data[\"comment_text\"].progress_apply(lambda x: ' '.join([word for word in x.split() if word not in (stops)]))","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:29:43.441660Z","iopub.execute_input":"2022-01-12T11:29:43.441877Z","iopub.status.idle":"2022-01-12T11:29:58.250925Z","shell.execute_reply.started":"2022-01-12T11:29:43.441851Z","shell.execute_reply":"2022-01-12T11:29:58.250112Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"html_pattern = re.compile(r\"<[^<]+?>\")\ndate_pattern = re.compile(r\"\\d\\d[-\\.\\\\\\/]\\d\\d[-\\.\\\\\\/]\\d{0,4}\")\nemoticon_pattern = re.compile((f\"({'|'.join(EMOTICONS.keys())})\"))\nurl_pattern = re.compile(r'https?://\\S+|www\\.\\S+') \nchat_words_map_dict = {pair[0]:pair[1] for pair in [line.split(\"=\") for line in chat_words_str.lower().split(\"\\n\")]}\nchat_words_set = set(chat_words_map_dict.keys())\n\ndef remove_html(text):\n    return html_pattern.sub(\"\", text)\n\ndef remove_dates(text):\n    return date_pattern.sub(\"\", text)\n\ndef remove_punctuation(text):\n    return \"\".join(filter(lambda x: x not in string.punctuation, text))\n\ndef remove_nums(text):\n    return \"\".join(filter(lambda x: not x.isdigit(), text))\n\ndef replace_emoticons(text):\n    return emoticon_pattern.sub(\"\", text)\n\ndef chat_words_conversion(text):\n    words_to_replace = set(text.split()).intersection(chat_words_set)\n    return \" \".join(chat_words_map_dict.get(word, word) for word in text.split())\n                               \ndef replace_emoji_with_word(text):\n    return emoji.demojize(text)\n\ndef clear_text(text):\n    return chat_words_conversion(remove_punctuation(replace_emoji_with_word(replace_emoticons(remove_nums(remove_dates(remove_html(text)))))))","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:30:17.102381Z","iopub.execute_input":"2022-01-12T11:30:17.102690Z","iopub.status.idle":"2022-01-12T11:30:17.121428Z","shell.execute_reply.started":"2022-01-12T11:30:17.102656Z","shell.execute_reply":"2022-01-12T11:30:17.120391Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#cleared_big_data = big_data.copy()\n#cleared_big_data[\"comment_text\"] = big_data[\"comment_text\"].progress_apply(clear_text)\n#cleared_big_data.to_csv(\"cleared_data.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#lemmantized_big_data = cleared_big_data.copy()\n#result = nlp.pipe(lemmantized_big_data[\"comment_text\"], batch_size=32, n_process=4, disable=[\"parser\", \"ner\"])\n#lemmantized_big_data[\"comment_text\"] = [\" \".join([tok.lemma_ for tok in doc]) for doc in result]\n#lemmantized_big_data.to_csv(\"lemmantized_cleared_data.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_several_language_data = np.concatenate([\n#    training_data_es[\"comment_text\"].astype(str), training_data_fr[\"comment_text\"].astype(str), training_data_it[\"comment_text\"].astype(str),\n#    training_data_pt[\"comment_text\"].astype(str), training_data_ru[\"comment_text\"].astype(str), training_data_tr[\"comment_text\"].astype(str),\n#])\n#y_several_language_data = np.concatenate([\n#    training_data_es[\"toxic\"], training_data_fr[\"toxic\"], training_data_it[\"toxic\"],\n#    training_data_pt[\"toxic\"], training_data_ru[\"toxic\"], training_data_tr[\"toxic\"],\n#])","metadata":{"execution":{"iopub.status.busy":"2022-01-04T18:48:58.744702Z","iopub.execute_input":"2022-01-04T18:48:58.745136Z","iopub.status.idle":"2022-01-04T18:48:58.961119Z","shell.execute_reply.started":"2022-01-04T18:48:58.745054Z","shell.execute_reply":"2022-01-04T18:48:58.960142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4>Load preprocessed data</font>","metadata":{}},{"cell_type":"code","source":"cleared_big_data = pd.read_csv(\"../input/jigsaw-preprocessed/cleared_data.csv\")\nlemmantized_big_data = pd.read_csv(\"../input/jigsaw-preprocessed/lemmantized_cleared_data.csv\", index_col=\"id\")\nvalidation_data = pd.read_csv(\"../input/jigsaw-preprocessed/valid_translated_lemmantized.csv\", index_col=\"id\")\ntest_data = pd.read_csv(\"../input/jigsaw-preprocessed/test_translated_lemmantized.csv\", index_col=\"id\")","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:30:27.146755Z","iopub.execute_input":"2022-01-12T11:30:27.147186Z","iopub.status.idle":"2022-01-12T11:30:36.191547Z","shell.execute_reply.started":"2022-01-12T11:30:27.147155Z","shell.execute_reply":"2022-01-12T11:30:36.190671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#X_very_big = np.concatenate([X_several_language_data, cleared_big_data[\"comment_text\"].to_numpy()])\n#y_very_big = np.concatenate([y_several_language_data, cleared_big_data[\"toxic\"].to_numpy()])","metadata":{"execution":{"iopub.status.busy":"2022-01-04T18:49:13.589281Z","iopub.execute_input":"2022-01-04T18:49:13.58961Z","iopub.status.idle":"2022-01-04T18:49:13.689104Z","shell.execute_reply.started":"2022-01-04T18:49:13.589573Z","shell.execute_reply":"2022-01-04T18:49:13.688133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#translated_valid_data[\"comment_text\"] = translated_valid_data[\"comment_text\"].progress_apply(lambda x: expand_contractions(x.lower()))\n#translated_valid_data[\"comment_text\"] = translated_valid_data[\"comment_text\"].progress_apply(lambda x: ' '.join([word for word in x.split() if word not in (stops)]))\n\n#translated_test_data[\"content\"] = translated_test_data[\"content\"].progress_apply(lambda x: expand_contractions(x.lower()))\n#translated_test_data[\"content\"] = translated_test_data[\"content\"].progress_apply(lambda x: ' '.join([word for word in x.split() if word not in (stops)]))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#result = nlp.pipe(translated_valid_data[\"translated\"], batch_size=32, n_process=4, disable=[\"parser\", \"ner\"])\n#translated_valid_data[\"translated\"] = [\" \".join([tok.lemma_ for tok in doc]) for doc in result]\n\n#result = nlp.pipe(translated_test_data[\"translated\"], batch_size=32, n_process=4, disable=[\"parser\", \"ner\"])\n#translated_test_data[\"translated\"] = [\" \".join([tok.lemma_ for tok in doc]) for doc in result]\n\n#translated_valid_data.to_csv(\"valid_translated_lemmantized.csv\")\n#translated_test_data.to_csv(\"test_translated_lemmantized.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"lemmantized_big_data = lemmantized_big_data.drop(\"Unnamed: 0\", axis=1)\nlemmantized_big_data = lemmantized_big_data.dropna()","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:31:35.127442Z","iopub.execute_input":"2022-01-12T11:31:35.128227Z","iopub.status.idle":"2022-01-12T11:31:35.238845Z","shell.execute_reply.started":"2022-01-12T11:31:35.128189Z","shell.execute_reply":"2022-01-12T11:31:35.237843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4>Set up TPU</font>","metadata":{}},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:31:40.838137Z","iopub.execute_input":"2022-01-12T11:31:40.838796Z","iopub.status.idle":"2022-01-12T11:31:46.477209Z","shell.execute_reply.started":"2022-01-12T11:31:40.838747Z","shell.execute_reply":"2022-01-12T11:31:46.476188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<font size=4>BERT</font>","metadata":{}},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\nEPOCHS = 1\nBATCH_SIZE = 32 * strategy.num_replicas_in_sync\nMAX_LENGTH = 128","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:31:46.479378Z","iopub.execute_input":"2022-01-12T11:31:46.480195Z","iopub.status.idle":"2022-01-12T11:31:46.485062Z","shell.execute_reply.started":"2022-01-12T11:31:46.480149Z","shell.execute_reply":"2022-01-12T11:31:46.484074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import AutoTokenizer, TFAutoModelForSequenceClassification\nfrom transformers import DataCollatorWithPadding\nfrom transformers import create_optimizer\n# First load the real tokenizer\ntokenizer = AutoTokenizer.from_pretrained('bert-base-multilingual-uncased')","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:31:46.486639Z","iopub.execute_input":"2022-01-12T11:31:46.487155Z","iopub.status.idle":"2022-01-12T11:31:50.039745Z","shell.execute_reply.started":"2022-01-12T11:31:46.487082Z","shell.execute_reply":"2022-01-12T11:31:50.038777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def tokenize(dataset):\n    return tokenizer(dataset, padding='max_length', max_length=MAX_LENGTH, \n                     truncation=True, return_tensors=\"tf\")","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:31:50.041876Z","iopub.execute_input":"2022-01-12T11:31:50.042128Z","iopub.status.idle":"2022-01-12T11:31:50.048671Z","shell.execute_reply.started":"2022-01-12T11:31:50.042100Z","shell.execute_reply":"2022-01-12T11:31:50.047004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:31:50.049828Z","iopub.execute_input":"2022-01-12T11:31:50.050067Z","iopub.status.idle":"2022-01-12T11:31:50.329664Z","shell.execute_reply.started":"2022-01-12T11:31:50.050038Z","shell.execute_reply":"2022-01-12T11:31:50.328532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_train = tokenize(lemmantized_big_data[\"comment_text\"].astype(str).tolist())\nx_valid = tokenize(validation_data[\"translated\"].astype(str).to_list())\nx_test = tokenize(test_data[\"translated\"].astype(str).to_list())\n\ny_train = lemmantized_big_data[\"toxic\"]\ny_valid = validation_data[\"toxic\"]\ny_test = test_labels[\"toxic\"]","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:33:16.980365Z","iopub.execute_input":"2022-01-12T11:33:16.981234Z","iopub.status.idle":"2022-01-12T11:33:29.889381Z","shell.execute_reply.started":"2022-01-12T11:33:16.981199Z","shell.execute_reply":"2022-01-12T11:33:29.888507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((dict(x_train), y_train))\n    .shuffle(10000, seed=42)\n    .batch(BATCH_SIZE, drop_remainder=True)\n    .cache()\n    .prefetch(AUTO)\n)\n\nvalid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((dict(x_valid), y_valid))\n    .batch(BATCH_SIZE, drop_remainder=True)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(dict(x_test))\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:33:55.599234Z","iopub.execute_input":"2022-01-12T11:33:55.600497Z","iopub.status.idle":"2022-01-12T11:33:55.646004Z","shell.execute_reply.started":"2022-01-12T11:33:55.600437Z","shell.execute_reply":"2022-01-12T11:33:55.645275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_model_new():\n    checkpoint = \"bert-base-multilingual-uncased\"\n    loss = tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True)\n    metrics = [tf.keras.metrics.AUC(name=\"AUC\")]\n    learning_rate = tf.keras.optimizers.schedules.ExponentialDecay(1e-5, 1000, 0.98)\n    optimizer = Adam(learning_rate=learning_rate)\n    \n    input_ids_in = tf.keras.layers.Input(shape=(MAX_LENGTH,), name='input_ids', dtype='int32')\n    input_masks_in = tf.keras.layers.Input(shape=(MAX_LENGTH,), name='attention_mask', dtype='int32') \n    input_token_types_in = tf.keras.layers.Input(shape=(MAX_LENGTH,), name='token_type_ids', dtype='int32') \n    \n    bert_model = TFAutoModelForSequenceClassification.from_pretrained(checkpoint, num_labels=1)\n    \n    embedding_layer = bert_model(input_ids_in, token_type_ids=input_token_types_in, attention_mask=input_masks_in)[0]\n    cls_token = embedding_layer[:, 0, :]\n    \n    X = tf.keras.layers.Dense(50, activation='relu')(cls_token)\n    X = tf.keras.layers.Dropout(0.2)(X)\n    X = tf.keras.layers.Dense(4, activation='sigmoid')(X)\n    \n    model = tf.keras.Model(inputs=[input_ids_in, input_masks_in, input_token_types_in], outputs = X)\n    model.compile(optimizer, loss=loss, metrics=metrics)\n    \n    return model\n\ndef build_model():\n    checkpoint = \"bert-base-multilingual-uncased\"\n    loss = tf.keras.losses.BinaryCrossentropy(from_logits=True)\n    metrics = [tf.keras.metrics.AUC(name=\"AUC\")]\n    learning_rate = tf.keras.optimizers.schedules.ExponentialDecay(1e-5, 1000, 0.98)\n    optimizer = Adam(learning_rate=learning_rate)\n    \n    model = TFAutoModelForSequenceClassification.from_pretrained(checkpoint, num_labels=1)\n    model.compile(optimizer, loss=loss, metrics=metrics)\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:34:00.801557Z","iopub.execute_input":"2022-01-12T11:34:00.802150Z","iopub.status.idle":"2022-01-12T11:34:00.815660Z","shell.execute_reply.started":"2022-01-12T11:34:00.802114Z","shell.execute_reply":"2022-01-12T11:34:00.814792Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nwith strategy.scope():\n    model = build_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:34:05.960093Z","iopub.execute_input":"2022-01-12T11:34:05.960466Z","iopub.status.idle":"2022-01-12T11:34:56.010028Z","shell.execute_reply.started":"2022-01-12T11:34:05.960431Z","shell.execute_reply":"2022-01-12T11:34:56.009066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred = model(tokenize([\"What the hell\", \"Really? Are you dumb?\"]))\nlogits = pred.logits\ntf.math.sigmoid(logits)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_history_1 = model.fit(train_dataset, epochs=1, validation_data=valid_dataset, batch_size=BATCH_SIZE)\n#train_history_1 = model.fit(x_train_ids, np.stack([y_train == 0, y_train == 1], axis=1).astype(int), \n#                            epochs=EPOCHS, \n#                            validation_data=(x_valid_ids, np.stack([y_valid == 0, y_valid == 1], axis=1).astype(int)), \n#                            batch_size=BATCH_SIZE)\n#train_history_1 = model.fit(train_dataset, epochs=EPOCHS, validation_data=valid_dataset, batch_size=BATCH_SIZE)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2022-01-12T11:35:23.436360Z","iopub.execute_input":"2022-01-12T11:35:23.436676Z","iopub.status.idle":"2022-01-12T11:40:14.859500Z","shell.execute_reply.started":"2022-01-12T11:35:23.436646Z","shell.execute_reply":"2022-01-12T11:40:14.858354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_history_2 = model.fit(valid_dataset, epochs=2, batch_size=BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:41:25.181559Z","iopub.execute_input":"2022-01-12T11:41:25.181906Z","iopub.status.idle":"2022-01-12T11:41:29.868513Z","shell.execute_reply.started":"2022-01-12T11:41:25.181868Z","shell.execute_reply":"2022-01-12T11:41:29.867399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bert_pred = model.predict(dict(x_test))\nlogits = bert_pred.logits\nprob = tf.math.sigmoid(logits)","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:41:36.812817Z","iopub.execute_input":"2022-01-12T11:41:36.813701Z","iopub.status.idle":"2022-01-12T11:41:56.280985Z","shell.execute_reply.started":"2022-01-12T11:41:36.813645Z","shell.execute_reply":"2022-01-12T11:41:56.280050Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"ROC AUC: {roc_auc_score(y_test, prob)}\")","metadata":{"execution":{"iopub.status.busy":"2022-01-12T11:41:56.282739Z","iopub.execute_input":"2022-01-12T11:41:56.282997Z","iopub.status.idle":"2022-01-12T11:41:56.362948Z","shell.execute_reply.started":"2022-01-12T11:41:56.282969Z","shell.execute_reply":"2022-01-12T11:41:56.361984Z"},"trusted":true},"execution_count":null,"outputs":[]}]}