{"cells":[{"metadata":{},"cell_type":"markdown","source":"","execution_count":null},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import os\nimport json\n\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nfrom tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom kaggle_datasets import KaggleDatasets\nimport transformers\nfrom transformers import TFAutoModel, AutoTokenizer,BertTokenizer\nfrom tqdm.notebook import tqdm\nfrom tokenizers import Tokenizer, models, pre_tokenizers, decoders, processors\nfrom sklearn.model_selection import KFold\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Helper Functions","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"def fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):\n    \"\"\"\n    https://www.kaggle.com/xhlulu/jigsaw-tpu-distilbert-with-huggingface-and-keras\n    \"\"\"\n    tokenizer.enable_truncation(max_length=maxlen)\n    tokenizer.enable_padding(max_length=maxlen)\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist()\n        encs = tokenizer.encode_batch(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n    \n    return np.array(all_ids)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def build_model(transformer, max_len=512):\n    \"\"\"\n    https://www.kaggle.com/xhlulu/jigsaw-tpu-distilbert-with-huggingface-and-keras\n    \"\"\"\n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\")\n    sequence_output = transformer(input_word_ids)[0]\n    cls_token = sequence_output[:, 0, :]\n    out = Dense(6, activation='softmax')(cls_token)\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='categorical_crossentropy', metrics=['accuracy'])\n    \n    return model","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## TPU Configs","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n\n# Data access\n# GCS_DS_PATH = KaggleDatasets().get_gcs_path()\n\n# Configuration\nEPOCHS = 2\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nMAX_LEN = 192\n","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Create fast tokenizer","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# tokenizer = AutoTokenizer.from_pretrained(\"voidful/albert_chinese_xlarge\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# MODEL = 'jplu/tf-xlm-roberta-large'\nMODEL = \"hfl/chinese-roberta-wwm-ext\"\nMODEL = 'hfl/chinese-roberta-wwm-ext-large'\nMODEL = '../input/chinese-roberta-wwm-ext-l12-h768-a12'\n# MODEL = '../input/hflchineserobertawwmext'\nMODEL = 'bert-base-chinese'\n# First load the real tokenizer\ntokenizer = AutoTokenizer.from_pretrained(MODEL)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Load text data into memory","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# read weibo data\nusual_train = pd.read_excel('../input/weibo-sentiment-analysis/usual_train.xlsx')\nvirus_train = pd.read_excel('../input/weibo-sentiment-analysis/virus_train.xlsx')\nusual_test = pd.read_excel('../input/weibo-sentiment-analysis/usual_eval.xlsx')\nvirus_test = pd.read_excel('../input/weibo-sentiment-analysis/virus_eval.xlsx')\n\nusual_train['type'] = 'usual'\nusual_test['type'] = 'usual'\nvirus_train['type'] = 'virus'\nvirus_test['type'] = 'virus'\nprint(usual_train.head())\nprint(virus_train.head())\ndata = usual_train.append(virus_train)\ndata['文本'] = data['文本'].astype('str')\n# 打乱数据\ndata = data.sample(frac=1,random_state=2020).reset_index(drop=True)\nprint(data.shape)\nlabel_dict = {}\nfor label in data['情绪标签'].unique():\n    label_dict[label] = len(label_dict)\nprint(label_dict)\ndata['情绪标签'] = data['情绪标签'].map(label_dict)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data['文本'] = data.apply(lambda x:'[疫情]'+x['文本'] if x.type=='virus' else '[普通]'+x['文本'],axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"data\ndef regular_encode(texts, tokenizer, data_type, maxlen=512):\n    enc_di = tokenizer.batch_encode_plus(\n        texts, \n        return_attention_masks=False, \n        return_token_type_ids=False,\n        pad_to_max_length=True,\n        max_length=maxlen\n    )\n    \n    res = np.array(enc_di['input_ids'])\n#     res2 = [np.insert(res[index],1,1) if data_type[index]=='usual' else np.insert(res[index],1,2) for index in range(len(texts))]\n    return np.array(res)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"kfold = KFold(n_splits=4, shuffle=True, random_state=2019)\nfor train_index, test_index in kfold.split(np.zeros(len(data))):\n    train = data.loc[train_index,:].reset_index()\n    val = data.loc[test_index,:].reset_index()\n    x_train = regular_encode(train['文本'].values, tokenizer, train['type'].values,maxlen=MAX_LEN)\n    x_valid = regular_encode(val['文本'].values, tokenizer, val['type'].values, maxlen=MAX_LEN)\n    y_train = train['情绪标签'].values\n    y_valid = val['情绪标签'].values\n    train_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, tf.keras.utils.to_categorical(y_train)))\n    .repeat()\n    .shuffle(2048)\n    .batch(BATCH_SIZE)\n    .prefetch(AUTO)\n)\n    valid_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, tf.keras.utils.to_categorical(y_valid)))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n#     %%time\n    with strategy.scope():\n        transformer_layer = TFAutoModel.from_pretrained(MODEL)\n        model = build_model(transformer_layer, max_len=MAX_LEN)\n#     model.summary()\n    n_steps = x_train.shape[0] // BATCH_SIZE\n    train_history = model.fit(\n        train_dataset,\n        steps_per_epoch=n_steps,\n        validation_data=valid_dataset,\n        epochs=EPOCHS+1\n    )\n    val['pred'] = model.predict(valid_dataset, verbose=1).argmax(axis=1)\n    val['match'] = val['pred']==val['情绪标签']\n    print(val['match'].mean())\n    print(val[val.type=='usual']['match'].mean())\n    print(val[val.type=='virus']['match'].mean())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"1\n0.7713877281724214\n0.7661302015369001\n0.7878925807919891\n2\n0.7790851110622389\n0.7720525264059378\n0.8026819923371648\n3\n0.7819201583635764\n0.7797515886770653\n0.7888427846934071\n4\n0.7890685142417244\n0.787350525860827\n0.7946096654275093","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"x_test_usual = regular_encode(usual_test['文本'].values, tokenizer,usual_test['type'].values, maxlen=MAX_LEN)\nx_test_virus = regular_encode(virus_test['文本'].values, tokenizer,virus_test['type'].values, maxlen=MAX_LEN)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# %%time \n\n# # x_train = regular_encode(train.comment_text.values, tokenizer, maxlen=MAX_LEN)\n# # x_valid = regular_encode(valid.comment_text.values, tokenizer, maxlen=MAX_LEN)\n# x_test = regular_encode(test.content.values, tokenizer, maxlen=MAX_LEN)\n\n# y_train = train.toxic.values\n# y_valid = valid.toxic.values","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Build datasets objects","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"usual_test_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test_usual)\n    .batch(BATCH_SIZE)\n)\nvirus_test_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test_virus)\n    .batch(BATCH_SIZE)\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n_steps = x_valid.shape[0] // BATCH_SIZE\ntrain_history_2 = model.fit(\n    valid_dataset.repeat(),\n    steps_per_epoch=n_steps,\n    epochs=EPOCHS\n)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Submission","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"usual_pred = model.predict(usual_test_dataset, verbose=1).argmax(axis=1)\nvirus_pred = model.predict(virus_test_dataset, verbose=1).argmax(axis=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"id_label = {}\nfor key,value in label_dict.items():\n    id_label[value] = key","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"usual_result = []\nfor index, id in enumerate(usual_test['数据编号']):\n    line = {}\n    line['id'] = id\n    line['label'] = id_label[usual_pred[index]]\n    usual_result.append(line)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"virus_result = []\nfor index, id in enumerate(virus_test['数据编号']):\n    line = {}\n    line['id'] = id\n    line['label'] = id_label[virus_pred[index]]\n    virus_result.append(line)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with open('usual_result.txt', 'w', encoding='utf-8') as f:\n    json.dump(usual_result, f)\nwith open('virus_result.txt', 'w', encoding='utf-8') as f:\n    json.dump(virus_result, f)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}