{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.10","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":12500,"databundleVersionId":1375107,"sourceType":"competition"},{"sourceId":19018,"databundleVersionId":2703900,"sourceType":"competition"},{"sourceId":27935,"databundleVersionId":3445671,"sourceType":"competition"},{"sourceId":2798066,"sourceType":"datasetVersion","datasetId":1709138}],"dockerImageVersionId":30146,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"---\n## [Jigsaw Rate Severity of Toxic Comments][1]\n---\n**Comments1**: 'Internet' is required for this Notebook.\n\n**Comments2**: Thanks to previous great Notebooks.\n\n1. [☣️ Jigsaw - Incredibly Simple Naive Bayes [0.768]][2]\n2. [AutoNLP for toxic ratings ;)][3]\n\n\n[1]: https://www.kaggle.com/c/jigsaw-toxic-severity-rating/overview\n[2]: https://www.kaggle.com/julian3833/jigsaw-incredibly-simple-naive-bayes-0-768\n[3]: https://www.kaggle.com/abhishek/autonlp-for-toxic-ratings","metadata":{}},{"cell_type":"markdown","source":"# 0. Settings","metadata":{}},{"cell_type":"code","source":"# Import dependencies libraries\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt \n%matplotlib inline\n\nimport os\nimport pathlib\nimport gc\nimport sys\nimport math \nimport time \nimport tqdm \nfrom tqdm import tqdm \nimport random\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nfrom sklearn.model_selection import KFold \nfrom sklearn.model_selection import StratifiedKFold \n\nimport tensorflow as tf\nimport tensorflow.keras as keras\nfrom tensorflow.keras.layers.experimental import preprocessing\n\nimport transformers \nimport datasets ","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:35:52.374544Z","iopub.execute_input":"2024-12-23T04:35:52.374868Z","iopub.status.idle":"2024-12-23T04:35:59.669795Z","shell.execute_reply.started":"2024-12-23T04:35:52.374787Z","shell.execute_reply":"2024-12-23T04:35:59.668728Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# global config set up\nconfig = {\n    'nfolds': 10,\n    'learning_rate': 1e-4,\n    'num_epochs': 3,\n    'batch_size': 8,\n}\n\nAUTOTUNE = tf.data.experimental.AUTOTUNE\n\n# For reproducible results    \ndef seed_all(s):\n    random.seed(s)\n    np.random.seed(s)\n    tf.random.set_seed(s)\n    os.environ['TF_CUDNN_DETERMINISTIC'] = '1'\n    os.environ['PYTHONHASHSEED'] = str(s) \nglobal_seed = 42\nseed_all(global_seed)","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:47:17.022178Z","iopub.execute_input":"2024-12-23T04:47:17.022457Z","iopub.status.idle":"2024-12-23T04:47:17.027903Z","shell.execute_reply.started":"2024-12-23T04:47:17.022426Z","shell.execute_reply":"2024-12-23T04:47:17.027154Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 1. Data Preprocessing","metadata":{}},{"cell_type":"markdown","source":"### 1. Create train data\n\nFor training data, I used [Toxic Comment Classification Challenge][1] dataset.\n\n[1]: https://www.kaggle.com/c/jigsaw-toxic-comment-classification-challenge/data\n\nI turn it into a binary toxic/ no-toxic classification","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('../input/jigsaw-toxic-comment-classification-challenge/train.csv')\ndf['y'] = (df[['toxic', 'severe_toxic', 'obscene', 'threat', 'insult', 'identity_hate']].sum(axis=1) > 0 ).astype(int)\ndf = df[['comment_text', 'y']].rename(columns={'comment_text': 'text'})\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:47:20.677716Z","iopub.execute_input":"2024-12-23T04:47:20.677991Z","iopub.status.idle":"2024-12-23T04:47:22.671244Z","shell.execute_reply.started":"2024-12-23T04:47:20.677960Z","shell.execute_reply":"2024-12-23T04:47:22.670401Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1.2 Undersampling\n\nThe dataset is very unbalanced. Here we undersample the majority class. Other strategies might work better.","metadata":{}},{"cell_type":"code","source":"df['y'].value_counts(normalize=True)","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:47:25.563219Z","iopub.execute_input":"2024-12-23T04:47:25.563901Z","iopub.status.idle":"2024-12-23T04:47:25.573120Z","shell.execute_reply.started":"2024-12-23T04:47:25.563865Z","shell.execute_reply":"2024-12-23T04:47:25.572362Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"'''\nXử lý data unbalanced bằng undersampling\ncó thể dùng SMOTE\n'''","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"min_len = (df['y'] == 1).sum()\ndf_y0_undersample = df[df['y'] == 0].sample(n=min_len, random_state=global_seed)\ntrain_df = pd.concat([df[df['y'] == 1], df_y0_undersample]).reset_index(drop=True)\ntrain_df['y'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:47:28.042024Z","iopub.execute_input":"2024-12-23T04:47:28.042627Z","iopub.status.idle":"2024-12-23T04:47:28.070274Z","shell.execute_reply.started":"2024-12-23T04:47:28.042597Z","shell.execute_reply":"2024-12-23T04:47:28.069631Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:47:30.550968Z","iopub.execute_input":"2024-12-23T04:47:30.551541Z","iopub.status.idle":"2024-12-23T04:47:30.559060Z","shell.execute_reply.started":"2024-12-23T04:47:30.551502Z","shell.execute_reply":"2024-12-23T04:47:30.558395Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1.3 k-fold","metadata":{}},{"cell_type":"code","source":"n_folds = 10\n\nskf = StratifiedKFold(n_splits=n_folds, shuffle=True, random_state=global_seed)\nfor nfold, (train_index, val_index) in enumerate(skf.split(X=train_df.index,\n                                                           y=train_df.y)):\n    train_df.loc[val_index, 'fold'] = nfold\nprint(train_df.groupby(['fold', train_df.y]).size())","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:47:38.102601Z","iopub.execute_input":"2024-12-23T04:47:38.103324Z","iopub.status.idle":"2024-12-23T04:47:38.129129Z","shell.execute_reply.started":"2024-12-23T04:47:38.103288Z","shell.execute_reply":"2024-12-23T04:47:38.128448Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"p_fold = 0\np_train = train_df.query(f'fold != {p_fold}').reset_index(drop=True)\np_valid = train_df.query(f'fold == {p_fold}').reset_index(drop=True)\n\nprint(len(p_train))\nprint(len(p_valid))\n\np_train.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:47:41.408344Z","iopub.execute_input":"2024-12-23T04:47:41.409041Z","iopub.status.idle":"2024-12-23T04:47:41.431069Z","shell.execute_reply.started":"2024-12-23T04:47:41.408988Z","shell.execute_reply":"2024-12-23T04:47:41.430241Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. DataSet","metadata":{}},{"cell_type":"code","source":"checkpoint = \"bert-large-uncased\"\ntokenizer = transformers.AutoTokenizer.from_pretrained(checkpoint)","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:47:47.545218Z","iopub.execute_input":"2024-12-23T04:47:47.545958Z","iopub.status.idle":"2024-12-23T04:47:49.571781Z","shell.execute_reply.started":"2024-12-23T04:47:47.545907Z","shell.execute_reply":"2024-12-23T04:47:49.571054Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"checkpoint = \"bert-base-uncased\"\ntokenizer = transformers.BertTokenizer.from_pretrained(checkpoint)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T02:18:43.572062Z","iopub.execute_input":"2024-12-23T02:18:43.572324Z","iopub.status.idle":"2024-12-23T02:18:44.209777Z","shell.execute_reply.started":"2024-12-23T02:18:43.572294Z","shell.execute_reply":"2024-12-23T02:18:44.209014Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"tokenizer","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-23T04:47:52.583842Z","iopub.execute_input":"2024-12-23T04:47:52.584084Z","iopub.status.idle":"2024-12-23T04:47:52.588978Z","shell.execute_reply.started":"2024-12-23T04:47:52.584059Z","shell.execute_reply":"2024-12-23T04:47:52.588250Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_ds = datasets.Dataset.from_pandas(p_train)\nvalid_ds = datasets.Dataset.from_pandas(p_valid)\n\nprint(train_ds)\nprint(valid_ds)","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:47:55.279749Z","iopub.execute_input":"2024-12-23T04:47:55.280920Z","iopub.status.idle":"2024-12-23T04:47:55.341631Z","shell.execute_reply.started":"2024-12-23T04:47:55.280870Z","shell.execute_reply":"2024-12-23T04:47:55.340814Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def tokenize_function(example):\n    return tokenizer(example[\"text\"], truncation=True)\n\ntokenized_train_ds = train_ds.map(tokenize_function, batched=True)\ntokenized_valid_ds = valid_ds.map(tokenize_function, batched=True)\n\nprint(tokenized_train_ds)\nprint(tokenized_valid_ds)","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:47:57.420823Z","iopub.execute_input":"2024-12-23T04:47:57.421437Z","iopub.status.idle":"2024-12-23T04:48:01.780771Z","shell.execute_reply.started":"2024-12-23T04:47:57.421407Z","shell.execute_reply":"2024-12-23T04:48:01.779993Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"data_collator = transformers.DataCollatorWithPadding(tokenizer=tokenizer)\n\ntf_train_ds = tokenized_train_ds.to_tf_dataset(\n    columns=[\"attention_mask\", \"input_ids\", \"token_type_ids\"],\n    label_cols=[\"y\"],\n    shuffle=True,\n    collate_fn=data_collator,\n    batch_size=config['batch_size'],\n)\n\ntf_valid_ds = tokenized_valid_ds.to_tf_dataset(\n    columns=[\"attention_mask\", \"input_ids\", \"token_type_ids\"],\n    label_cols=[\"y\"],\n    shuffle=False,\n    collate_fn=data_collator,\n    batch_size=config['batch_size'],\n)\n\nprint(len(tf_train_ds))\nprint(len(tf_valid_ds))","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:48:05.728138Z","iopub.execute_input":"2024-12-23T04:48:05.728402Z","iopub.status.idle":"2024-12-23T04:48:11.396698Z","shell.execute_reply.started":"2024-12-23T04:48:05.728372Z","shell.execute_reply":"2024-12-23T04:48:11.395777Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Model Training","metadata":{}},{"cell_type":"code","source":"from transformers import TFAutoModelForSequenceClassification\n\nmodel = TFAutoModelForSequenceClassification.from_pretrained(checkpoint, num_labels=2)","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:48:15.433179Z","iopub.execute_input":"2024-12-23T04:48:15.433412Z","iopub.status.idle":"2024-12-23T04:48:45.389988Z","shell.execute_reply.started":"2024-12-23T04:48:15.433388Z","shell.execute_reply":"2024-12-23T04:48:45.389273Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"num_epochs = 2\nnum_train_steps = len(tf_train_ds) * num_epochs\n\nlr_scheduler = tf.keras.optimizers.schedules.PolynomialDecay(\n    initial_learning_rate=5e-5, end_learning_rate=0.0, decay_steps=num_train_steps\n)\n\nmodel.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=lr_scheduler),\n              loss=tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n              metrics=['accuracy'])\n\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:48:49.177451Z","iopub.execute_input":"2024-12-23T04:48:49.177729Z","iopub.status.idle":"2024-12-23T04:48:49.210379Z","shell.execute_reply.started":"2024-12-23T04:48:49.177697Z","shell.execute_reply":"2024-12-23T04:48:49.209717Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fit_history = model.fit(tf_train_ds,\n                        epochs=num_epochs,\n                        validation_data=tf_valid_ds,\n                        verbose=1)","metadata":{"execution":{"iopub.status.busy":"2024-12-23T04:48:52.270529Z","iopub.execute_input":"2024-12-23T04:48:52.270813Z","iopub.status.idle":"2024-12-23T04:49:30.877562Z","shell.execute_reply.started":"2024-12-23T04:48:52.270774Z","shell.execute_reply":"2024-12-23T04:49:30.876434Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Prediction & Submit","metadata":{}},{"cell_type":"code","source":"test_df = pd.read_csv(\"../input/jigsaw-toxic-severity-rating/comments_to_score.csv\")\ntest_ds = datasets.Dataset.from_pandas(test_df)\ntokenized_test_ds = test_ds.map(tokenize_function, batched=True)\ntf_test_ds = tokenized_test_ds.to_tf_dataset(\n    columns=[\"attention_mask\", \"input_ids\", \"token_type_ids\"],\n    shuffle=False,\n    collate_fn=data_collator,\n    batch_size=config['batch_size'],\n)","metadata":{"execution":{"iopub.status.busy":"2024-12-23T02:57:05.160662Z","iopub.execute_input":"2024-12-23T02:57:05.160944Z","iopub.status.idle":"2024-12-23T02:57:16.995707Z","shell.execute_reply.started":"2024-12-23T02:57:05.160914Z","shell.execute_reply":"2024-12-23T02:57:16.995143Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"raw_result = model.predict(tf_test_ds)\nresult = tf.sigmoid(raw_result.logits)\n\ntest_df['score'] = result.numpy()[:, 0]\nsubmission_df = test_df[['comment_id', 'score']]\n\n# submission_df.to_csv(\"submission.csv\", index=False) \nsubmission_df","metadata":{"execution":{"iopub.status.busy":"2024-12-23T02:57:56.828008Z","iopub.execute_input":"2024-12-23T02:57:56.828266Z","iopub.status.idle":"2024-12-23T02:59:40.941255Z","shell.execute_reply.started":"2024-12-23T02:57:56.828237Z","shell.execute_reply":"2024-12-23T02:59:40.940561Z"},"trusted":true},"outputs":[],"execution_count":null}]}