{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom tqdm.auto import tqdm\nfrom bs4 import BeautifulSoup\nfrom sklearn.feature_extraction.text import TfidfVectorizer\nimport gc\nimport re \nfrom scipy import sparse\nimport time\nimport warnings\nwarnings.filterwarnings(\"ignore\")\npd.options.display.max_colwidth=300\npd.options.display.max_columns = 100\n\nfrom sklearn.linear_model import Ridge\n","metadata":{"papermill":{"duration":2.036616,"end_time":"2021-12-31T12:05:46.476168","exception":false,"start_time":"2021-12-31T12:05:44.439552","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2022-02-06T05:12:57.667849Z","iopub.execute_input":"2022-02-06T05:12:57.668779Z","iopub.status.idle":"2022-02-06T05:12:59.003695Z","shell.execute_reply.started":"2022-02-06T05:12:57.668671Z","shell.execute_reply":"2022-02-06T05:12:59.002833Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport random\nimport torch\nimport torch.nn as nn\nfrom torch.nn import Parameter\nimport torch.nn.functional as F\nfrom torch.optim import Adam, SGD, AdamW\nfrom torch.utils.data import DataLoader, Dataset\n\nimport transformers\nfrom transformers import AutoTokenizer, AutoModel, AutoConfig\nfrom transformers import get_linear_schedule_with_warmup, get_cosine_schedule_with_warmup","metadata":{"execution":{"iopub.status.busy":"2022-02-06T05:12:59.005270Z","iopub.execute_input":"2022-02-06T05:12:59.005508Z","iopub.status.idle":"2022-02-06T05:13:05.971553Z","shell.execute_reply.started":"2022-02-06T05:12:59.005477Z","shell.execute_reply":"2022-02-06T05:13:05.970767Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Validation data \n\ndf_val = pd.read_csv(\"../input/jigsaw-toxic-severity-rating/validation_data.csv\")\nprint(df_val.shape)\n\ndf_more = df_val[['more_toxic']].rename(columns={'more_toxic': 'text'}).reset_index(drop=True)\ndf_less = df_val[['less_toxic']].rename(columns={'less_toxic': 'text'}).reset_index(drop=True)\n\ndf_val_unique = pd.concat([df_more, df_less]).drop_duplicates(subset='text', keep='first')\n\nprint(df_val_unique.shape)\ndf_val_unique.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-06T05:13:05.972817Z","iopub.execute_input":"2022-02-06T05:13:05.973040Z","iopub.status.idle":"2022-02-06T05:13:06.615315Z","shell.execute_reply.started":"2022-02-06T05:13:05.973012Z","shell.execute_reply":"2022-02-06T05:13:06.614407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_val.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/jigsaw4-previous-data-preprocessing/jig1_no_jig4_dup_df.csv')\n\nprint(df.shape)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-06T05:13:11.536454Z","iopub.execute_input":"2022-02-06T05:13:11.536762Z","iopub.status.idle":"2022-02-06T05:13:15.392032Z","shell.execute_reply.started":"2022-02-06T05:13:11.536726Z","shell.execute_reply":"2022-02-06T05:13:15.391197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['severe_toxic'] = df['severe_toxic'] * 2\ndf['y'] = (df[['toxic', 'severe_toxic', 'obscene',\n       'identity_hate', 'insult', 'threat', ]].sum(axis=1) ).astype(int)\n\ndf['y'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-02-06T05:14:14.314080Z","iopub.execute_input":"2022-02-06T05:14:14.314372Z","iopub.status.idle":"2022-02-06T05:14:14.347836Z","shell.execute_reply.started":"2022-02-06T05:14:14.314334Z","shell.execute_reply":"2022-02-06T05:14:14.347040Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dfs = []\nfor i in range(0, 8):\n    dfs.append(df.query(f'y == {i}')['text'])","metadata":{"execution":{"iopub.status.busy":"2022-02-06T05:14:23.852940Z","iopub.execute_input":"2022-02-06T05:14:23.853225Z","iopub.status.idle":"2022-02-06T05:14:23.944846Z","shell.execute_reply.started":"2022-02-06T05:14:23.853185Z","shell.execute_reply":"2022-02-06T05:14:23.944066Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pair_df = pd.DataFrame()\npair_df['more_toxic'] = pd.concat([\n    dfs[7], dfs[7], dfs[7], dfs[7], dfs[7], dfs[7], dfs[7],\n    dfs[6], dfs[6], dfs[6], dfs[6], dfs[6], dfs[6],\n    dfs[5], dfs[5], dfs[5], dfs[5], dfs[5],\n    dfs[4], dfs[4], dfs[4], dfs[4],\n    dfs[3], dfs[3],\n    dfs[2], dfs[2], \n    dfs[1]\n]).reset_index(drop=True)\n\npair_df['less_toxic'] = pd.concat([\n    dfs[6].sample(len(dfs[7]), random_state=42),\n    dfs[5].sample(len(dfs[7]), random_state=42),\n    dfs[4].sample(len(dfs[7]), random_state=42),\n    dfs[3].sample(len(dfs[7]), random_state=42),\n    dfs[2].sample(len(dfs[7]), random_state=42),\n    dfs[1].sample(len(dfs[7]), random_state=42),\n    dfs[0].sample(len(dfs[7]), random_state=42),\n    \n    dfs[5].sample(len(dfs[6]), random_state=42),\n    dfs[4].sample(len(dfs[6]), random_state=42),\n    dfs[3].sample(len(dfs[6]), random_state=42),\n    dfs[2].sample(len(dfs[6]), random_state=42),\n    dfs[1].sample(len(dfs[6]), random_state=42),\n    dfs[0].sample(len(dfs[6]), random_state=42),\n    \n    dfs[4].sample(len(dfs[5]), random_state=42),\n    dfs[3].sample(len(dfs[5]), random_state=42),\n    dfs[2].sample(len(dfs[5]), random_state=42),\n    dfs[1].sample(len(dfs[5]), random_state=42),\n    dfs[0].sample(len(dfs[5]), random_state=42),\n    \n    dfs[3].sample(len(dfs[4]), random_state=42),\n    dfs[2].sample(len(dfs[4]), random_state=42),\n    dfs[1].sample(len(dfs[4]), random_state=42),\n    dfs[0].sample(len(dfs[4]), random_state=42),\n    \n    dfs[1].sample(len(dfs[3]), random_state=42),\n    dfs[0].sample(len(dfs[3]), random_state=42),\n    \n    dfs[1].sample(len(dfs[2]), random_state=42),\n    dfs[0].sample(len(dfs[2]), random_state=42),\n    \n    dfs[0].sample(len(dfs[1]), random_state=42)\n]).reset_index(drop=True)\n\npair_df['worker'] = 9999\n\nprint(pair_df.shape)\npair_df.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-06T05:15:08.349147Z","iopub.execute_input":"2022-02-06T05:15:08.349413Z","iopub.status.idle":"2022-02-06T05:15:08.474144Z","shell.execute_reply.started":"2022-02-06T05:15:08.349384Z","shell.execute_reply":"2022-02-06T05:15:08.473364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pair_df.to_csv('jigsaw4_additional_pairs_from_jigsaw1.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-06T05:15:18.266051Z","iopub.execute_input":"2022-02-06T05:15:18.267021Z","iopub.status.idle":"2022-02-06T05:15:19.348913Z","shell.execute_reply.started":"2022-02-06T05:15:18.266979Z","shell.execute_reply":"2022-02-06T05:15:19.348029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold, GroupKFold, KFold\n\ntrain = pd.read_csv(\"../input/jigsaw-toxic-severity-rating/validation_data.csv\")\n\n\nFold = StratifiedKFold(n_splits=5, shuffle=True, random_state=2021)\nfor n, (trn_index, val_index) in enumerate(Fold.split(train, train['worker'])):\n    train.loc[val_index, 'kfold'] = int(n)\ntrain['kfold'] = train['kfold'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-02-06T05:15:22.388820Z","iopub.execute_input":"2022-02-06T05:15:22.389067Z","iopub.status.idle":"2022-02-06T05:15:22.693296Z","shell.execute_reply.started":"2022-02-06T05:15:22.389038Z","shell.execute_reply":"2022-02-06T05:15:22.692447Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"external = pair_df.copy()\n\nFold = KFold(n_splits=5, shuffle=True, random_state=2021)\nfor n, (trn_index, val_index) in enumerate(Fold.split(external, external['worker'])):\n    external.loc[val_index, 'kfold'] = int(n)\nexternal['kfold'] = external['kfold'].astype(int)","metadata":{"execution":{"iopub.status.busy":"2022-02-06T05:15:23.384769Z","iopub.execute_input":"2022-02-06T05:15:23.385892Z","iopub.status.idle":"2022-02-06T05:15:23.409268Z","shell.execute_reply.started":"2022-02-06T05:15:23.385839Z","shell.execute_reply":"2022-02-06T05:15:23.408356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fold = 0\n\nfor fold in range(5):\n    print(f' *** {fold} ***')\n    trn_df = train[train.kfold != fold].reset_index(drop=True)\n    val_df = train[train.kfold == fold].reset_index(drop=True)\n\n    # join external pair \n    trn_ex = external[external.kfold != fold].reset_index(drop=True)\n\n    trn_ex = trn_ex.reset_index()\n    print(len(trn_df), len(val_df), len(trn_ex))\n\n    idx1 = list(pd.merge(val_df, trn_ex, on=['less_toxic'], how='inner').index)\n    idx2 = list(pd.merge(val_df, trn_ex, on=['more_toxic'], how='inner').index)\n    duplicated_index = list(set(idx1 + idx2))\n    print(f'num dup : ', len(duplicated_index))\n\n    trn_ex = trn_ex[~trn_ex['index'].isin(duplicated_index)]\n\n    trn_df = pd.concat([trn_df, trn_ex]).reset_index(drop=True)\n    print(f'merged : ', len(trn_df))","metadata":{"execution":{"iopub.status.busy":"2022-02-06T05:15:24.262502Z","iopub.execute_input":"2022-02-06T05:15:24.263124Z","iopub.status.idle":"2022-02-06T05:15:24.589905Z","shell.execute_reply.started":"2022-02-06T05:15:24.263075Z","shell.execute_reply":"2022-02-06T05:15:24.588198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}