{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport nltk\nimport re\nfrom bs4 import BeautifulSoup\n\nfrom tqdm.auto import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-01-20T06:52:40.381349Z","iopub.execute_input":"2022-01-20T06:52:40.381924Z","iopub.status.idle":"2022-01-20T06:52:41.934999Z","shell.execute_reply.started":"2022-01-20T06:52:40.381891Z","shell.execute_reply":"2022-01-20T06:52:41.933969Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_DATA_PATH = \"../input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\"\nVALID_DATA_PATH = \"../input/jigsaw-multilingual-toxic-comment-classification/validation.csv\"\nTEST_DATA_PATH = \"../input/jigsaw-multilingual-toxic-comment-classification/test-processed-seqlen128.csv\"","metadata":{"execution":{"iopub.status.busy":"2022-01-20T06:52:43.901950Z","iopub.execute_input":"2022-01-20T06:52:43.902220Z","iopub.status.idle":"2022-01-20T06:52:43.906438Z","shell.execute_reply.started":"2022-01-20T06:52:43.902189Z","shell.execute_reply":"2022-01-20T06:52:43.905665Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_data(dirs):\n    data = (pd.read_csv(dirs[i]) for i in range(len(dirs)))\n    return data\n\ndf_train,df_valid,df_test = read_data([TRAIN_DATA_PATH,VALID_DATA_PATH,TEST_DATA_PATH])","metadata":{"execution":{"iopub.status.busy":"2022-01-20T06:52:44.807087Z","iopub.execute_input":"2022-01-20T06:52:44.807686Z","iopub.status.idle":"2022-01-20T06:52:50.038005Z","shell.execute_reply.started":"2022-01-20T06:52:44.807635Z","shell.execute_reply":"2022-01-20T06:52:50.037375Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat_mtpl = {'obscene': 0.16, 'toxic': 0.32, 'threat': 1.5, \n            'insult': 0.64, 'severe_toxic': 1.5, 'identity_hate': 1.5}\n\nfor category in cat_mtpl:\n    df_train[category] = df_train[category] * cat_mtpl[category]\n\ndf_train['score'] = df_train.loc[:, 'toxic':'identity_hate'].mean(axis=1)\n\ndf_train['y'] = df_train['score']\n\nmin_len = (df_train['y'] > 0).sum()  # len of toxic comments\ndf_y0_undersample = df_train[df_train['y'] == 0].sample(n=min_len, random_state=41)  # take non toxic comments\ndf_train_new = pd.concat([df_train[df_train['y'] > 0], df_y0_undersample])  # make new df\ndf_train_new","metadata":{"execution":{"iopub.status.busy":"2022-01-20T06:52:50.170214Z","iopub.execute_input":"2022-01-20T06:52:50.170490Z","iopub.status.idle":"2022-01-20T06:52:50.242330Z","shell.execute_reply.started":"2022-01-20T06:52:50.170461Z","shell.execute_reply":"2022-01-20T06:52:50.241644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tokenizers import (\n    decoders,\n    models,\n    normalizers,\n    pre_tokenizers,\n    processors,\n    trainers,\n    Tokenizer,\n)\n\nraw_tokenizer = Tokenizer(models.WordPiece(unk_token=\"[UNK]\"))\nraw_tokenizer.normalizer = normalizers.BertNormalizer(lowercase=True)\nraw_tokenizer.pre_tokenizer = pre_tokenizers.BertPreTokenizer()\nspecial_tokens = [\"[UNK]\", \"[PAD]\", \"[CLS]\", \"[SEP]\", \"[MASK]\"]\ntrainer = trainers.WordPieceTrainer(vocab_size=25000, special_tokens=special_tokens)","metadata":{"execution":{"iopub.status.busy":"2022-01-20T06:52:50.243566Z","iopub.execute_input":"2022-01-20T06:52:50.243738Z","iopub.status.idle":"2022-01-20T06:52:50.249224Z","shell.execute_reply.started":"2022-01-20T06:52:50.243718Z","shell.execute_reply":"2022-01-20T06:52:50.248496Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from datasets import Dataset\n\ndataset = Dataset.from_pandas(df_train_new[['comment_text']])\n\ndef get_training_corpus():\n    for i in range(0, len(dataset), 1000):\n        yield dataset[i : i + 1000][\"comment_text\"]","metadata":{"execution":{"iopub.status.busy":"2022-01-20T06:52:53.070315Z","iopub.execute_input":"2022-01-20T06:52:53.071022Z","iopub.status.idle":"2022-01-20T06:52:54.517848Z","shell.execute_reply.started":"2022-01-20T06:52:53.070986Z","shell.execute_reply":"2022-01-20T06:52:54.517210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"raw_tokenizer.train_from_iterator(get_training_corpus(), trainer=trainer)","metadata":{"execution":{"iopub.status.busy":"2022-01-20T06:52:54.519032Z","iopub.execute_input":"2022-01-20T06:52:54.519197Z","iopub.status.idle":"2022-01-20T06:52:59.797881Z","shell.execute_reply.started":"2022-01-20T06:52:54.519177Z","shell.execute_reply":"2022-01-20T06:52:59.797089Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from transformers import PreTrainedTokenizerFast\n\ntokenizer = PreTrainedTokenizerFast(\n    tokenizer_object=raw_tokenizer,\n    unk_token=\"[UNK]\",\n    pad_token=\"[PAD]\",\n    cls_token=\"[CLS]\",\n    sep_token=\"[SEP]\",\n    mask_token=\"[MASK]\",\n)","metadata":{"execution":{"iopub.status.busy":"2022-01-20T06:52:59.799666Z","iopub.execute_input":"2022-01-20T06:52:59.800054Z","iopub.status.idle":"2022-01-20T06:52:59.891855Z","shell.execute_reply.started":"2022-01-20T06:52:59.799996Z","shell.execute_reply":"2022-01-20T06:52:59.891068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Training","metadata":{}},{"cell_type":"code","source":"from sklearn.feature_extraction.text import TfidfVectorizer\nfrom sklearn.linear_model import LinearRegression","metadata":{"execution":{"iopub.status.busy":"2022-01-20T06:53:00.878016Z","iopub.execute_input":"2022-01-20T06:53:00.878259Z","iopub.status.idle":"2022-01-20T06:53:00.882231Z","shell.execute_reply.started":"2022-01-20T06:53:00.878231Z","shell.execute_reply":"2022-01-20T06:53:00.881471Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def dummy_fun(doc):return doc\n\nlabels = df_train_new['y']\ncomments = df_train_new['comment_text']\ntokenized_comments = tokenizer(comments.to_list())['input_ids']\n\nvectorizer = TfidfVectorizer(\n    analyzer = 'word',\n    tokenizer = dummy_fun,\n    preprocessor = dummy_fun,\n    token_pattern = None)\n\ncomments_tr = vectorizer.fit_transform(tokenized_comments)\n# print(comments_tr)","metadata":{"execution":{"iopub.status.busy":"2022-01-20T06:53:18.686015Z","iopub.execute_input":"2022-01-20T06:53:18.686459Z","iopub.status.idle":"2022-01-20T06:53:26.574949Z","shell.execute_reply.started":"2022-01-20T06:53:18.686427Z","shell.execute_reply":"2022-01-20T06:53:26.574188Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"regressor = LinearRegression()\nregressor.fit(comments_tr, labels)","metadata":{"execution":{"iopub.status.busy":"2022-01-20T06:53:39.521763Z","iopub.execute_input":"2022-01-20T06:53:39.522583Z","iopub.status.idle":"2022-01-20T06:57:44.210874Z","shell.execute_reply.started":"2022-01-20T06:53:39.522531Z","shell.execute_reply":"2022-01-20T06:57:44.209981Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Validation","metadata":{}},{"cell_type":"code","source":"# preprocess val data\nless_toxic_comments = df_valid[df_valid['toxic'] == 0]['comment_text']\nmore_toxic_comments = df_valid[df_valid['toxic'] == 1]['comment_text']\n\nless_toxic_comments = tokenizer(less_toxic_comments.to_list())['input_ids']\nmore_toxic_comments = tokenizer(more_toxic_comments.to_list())['input_ids']\n\nless_toxic = vectorizer.transform(less_toxic_comments)\nmore_toxic = vectorizer.transform(more_toxic_comments)\n\n# make predictions\ny_pred_less = regressor.predict(less_toxic)\ny_pred_more = regressor.predict(more_toxic)\nprint(y_pred_more)","metadata":{"execution":{"iopub.status.busy":"2022-01-20T06:57:46.449453Z","iopub.execute_input":"2022-01-20T06:57:46.449669Z","iopub.status.idle":"2022-01-20T06:57:48.635241Z","shell.execute_reply.started":"2022-01-20T06:57:46.449642Z","shell.execute_reply":"2022-01-20T06:57:48.634381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"texts = df_test['comment_text']\ntexts = tokenizer(texts.to_list())['input_ids']\ntexts = vectorizer.transform(texts)","metadata":{"execution":{"iopub.status.busy":"2022-01-20T06:58:07.888205Z","iopub.execute_input":"2022-01-20T06:58:07.888440Z","iopub.status.idle":"2022-01-20T06:58:25.229495Z","shell.execute_reply.started":"2022-01-20T06:58:07.888413Z","shell.execute_reply":"2022-01-20T06:58:25.228648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test['prediction'] = regressor.predict(texts)\ndf_test = df_test[['comment_text','prediction']]\n\ndf_test['score'] = df_test['prediction']\ndf_test = df_test[['comment_text','score']]","metadata":{"execution":{"iopub.status.busy":"2022-01-20T06:58:25.231063Z","iopub.execute_input":"2022-01-20T06:58:25.231292Z","iopub.status.idle":"2022-01-20T06:58:25.264952Z","shell.execute_reply.started":"2022-01-20T06:58:25.231264Z","shell.execute_reply":"2022-01-20T06:58:25.264069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_test.head(10)","metadata":{"execution":{"iopub.status.busy":"2022-01-20T06:58:25.266279Z","iopub.execute_input":"2022-01-20T06:58:25.266669Z","iopub.status.idle":"2022-01-20T06:58:25.277051Z","shell.execute_reply.started":"2022-01-20T06:58:25.266630Z","shell.execute_reply":"2022-01-20T06:58:25.276353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}}]}