{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"from tensorflow.keras.layers import Dense, Input\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import ModelCheckpoint #導入tensorflow\n# from tensorflow import keras\nfrom tensorflow.keras.callbacks import ModelCheckpoint\n\nfrom kaggle_datasets import KaggleDatasets #採用Kaggle資料集\nimport transformers\n\nfrom tokenizers import BertWordPieceTokenizer #分詞器\nfrom tqdm import tqdm #進度條顯示\nimport numpy as np\n\n#!pip install wandb\n\n#基本模型導入\nimport os, time\nimport gc\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow_hub as hub\nfrom kaggle_datasets import KaggleDatasets\n\n!pip install bert-tensorflow\nimport bert.tokenization\n\nprint(tf.version.VERSION) #tensorflow版本輸出","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(transformers.__version__) #tensorflow版本輸出","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:42:03.846095Z","iopub.execute_input":"2021-08-11T06:42:03.846401Z","iopub.status.idle":"2021-08-11T06:42:03.855547Z","shell.execute_reply.started":"2021-08-11T06:42:03.846369Z","shell.execute_reply":"2021-08-11T06:42:03.854734Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# TPU 檢測. \ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu) #TPU的連接\nelse:\n    \n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)\n\n#在TPU上針對Kaggle用戶運行Bert模型","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:42:03.857124Z","iopub.execute_input":"2021-08-11T06:42:03.857417Z","iopub.status.idle":"2021-08-11T06:42:09.81812Z","shell.execute_reply.started":"2021-08-11T06:42:03.857388Z","shell.execute_reply":"2021-08-11T06:42:09.817002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SEQUENCE_LENGTH = 128 #一個輸入字串長度為128的list\n\n#設置Kaggle數據的訪問路徑\nDATA_PATH =  KaggleDatasets().get_gcs_path('jigsaw-multilingual-toxic-comment-classification')\nBERT_PATH = KaggleDatasets().get_gcs_path('bert-multi')\nBERT_PATH_SAVEDMODEL = BERT_PATH + \"/bert_multi_from_tfhub\"\nWEIGHTS_PATH = '../input/jigsaw-weights'\n\n\nOUTPUT_PATH = \"/kaggle/working\"","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2021-08-11T06:42:09.819917Z","iopub.execute_input":"2021-08-11T06:42:09.820491Z","iopub.status.idle":"2021-08-11T06:42:10.445355Z","shell.execute_reply.started":"2021-08-11T06:42:09.820448Z","shell.execute_reply":"2021-08-11T06:42:10.44418Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train1 = pd.read_csv(\"/kaggle/input/jigsaw-multilingual-toxic-comment-classification/jigsaw-toxic-comment-train.csv\")\ntrain2 = pd.read_csv(\"/kaggle/input/jigsawch/train-ch.csv\")\nvalid = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/validation.csv')\ntest = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/test.csv')\nsub = pd.read_csv('/kaggle/input/jigsaw-multilingual-toxic-comment-classification/sample_submission.csv')\nsub2 = pd.read_csv('../input/ensemble/submission.csv')","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:42:10.448854Z","iopub.execute_input":"2021-08-11T06:42:10.449171Z","iopub.status.idle":"2021-08-11T06:42:14.540354Z","shell.execute_reply.started":"2021-08-11T06:42:10.44914Z","shell.execute_reply":"2021-08-11T06:42:14.539084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test.tail())\nprint(type(test))\nprint(test.columns)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:42:14.541904Z","iopub.execute_input":"2021-08-11T06:42:14.542318Z","iopub.status.idle":"2021-08-11T06:42:14.561597Z","shell.execute_reply.started":"2021-08-11T06:42:14.542275Z","shell.execute_reply":"2021-08-11T06:42:14.560777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# BERT Tokenizer","metadata":{}},{"cell_type":"code","source":"#把文字切割並轉成BERT所需要的編碼\n\ndef get_tokenizer(bert_path=BERT_PATH_SAVEDMODEL):\n    bert_layer = tf.saved_model.load(bert_path)\n    bert_layer = hub.KerasLayer(bert_layer, trainable=False)\n    vocab_file = bert_layer.resolved_object.vocab_file.asset_path.numpy() \n    cased = bert_layer.resolved_object.do_lower_case.numpy()\n    tf.gfile = tf.io.gfile  \n    tokenizer = bert.tokenization.FullTokenizer(vocab_file, cased)\n  \n    return tokenizer\n\ntokenizer = get_tokenizer()","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:42:14.573815Z","iopub.execute_input":"2021-08-11T06:42:14.574079Z","iopub.status.idle":"2021-08-11T06:42:32.68698Z","shell.execute_reply.started":"2021-08-11T06:42:14.574054Z","shell.execute_reply":"2021-08-11T06:42:32.686073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preprocessing","metadata":{}},{"cell_type":"code","source":"#編碼器，用於將文本編碼為整數序列，以進行BERT輸入\n\ndef fast_encode(texts, tokenizer, chunk_size=256, maxlen=512):#批次上傳256，最長序列512\n    \n    tokenizer.enable_truncation(max_length=maxlen)\n    tokenizer.enable_padding(length=maxlen) #最大長度為512，不足會自動補0\n    all_ids = []\n    \n    for i in tqdm(range(0, len(texts), chunk_size)):\n        text_chunk = texts[i:i+chunk_size].tolist() #將數據轉換為最接近Python的類型\n        encs = tokenizer.encode_batch(text_chunk)\n        #print(text_chunk)\n        all_ids.extend([enc.ids for enc in encs])\n        \n    \n    return np.array(all_ids)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:42:32.692092Z","iopub.execute_input":"2021-08-11T06:42:32.69248Z","iopub.status.idle":"2021-08-11T06:42:32.701808Z","shell.execute_reply.started":"2021-08-11T06:42:32.692438Z","shell.execute_reply":"2021-08-11T06:42:32.700733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#用於配置的IMP數據\n\nAUTO = tf.data.experimental.AUTOTUNE\n\n# 配置\nEPOCHS = 5 #定義訓練過程數據輪5次\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync  #資料集大小\nMAX_LEN = 192","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:42:32.702964Z","iopub.execute_input":"2021-08-11T06:42:32.703296Z","iopub.status.idle":"2021-08-11T06:42:32.714961Z","shell.execute_reply.started":"2021-08-11T06:42:32.703267Z","shell.execute_reply":"2021-08-11T06:42:32.713872Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = transformers.DistilBertTokenizer.from_pretrained('distilbert-base-multilingual-cased')# 使用分詞器加載DistilBERT\n\ntokenizer.save_pretrained('.') #儲存\n\nfast_tokenizer = BertWordPieceTokenizer('vocab.txt', lowercase=False)\nfast_tokenizer  #利用 huggingface tokenizers庫 重新加載詞向量，lowercase=False:詞向量皆為大寫","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:42:32.717115Z","iopub.execute_input":"2021-08-11T06:42:32.717423Z","iopub.status.idle":"2021-08-11T06:42:35.507466Z","shell.execute_reply.started":"2021-08-11T06:42:32.717396Z","shell.execute_reply":"2021-08-11T06:42:35.506677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#快速編碼\n\nx_train = fast_encode(train1.comment_text.astype(str), fast_tokenizer, maxlen=MAX_LEN)\nx_train_2 = fast_encode(train2.comment_text.astype(str), fast_tokenizer, maxlen=MAX_LEN)\nx_valid = fast_encode(valid.comment_text.astype(str), fast_tokenizer, maxlen=MAX_LEN)\nx_test = fast_encode(test.content.astype(str), fast_tokenizer, maxlen=MAX_LEN)\n\ny_train = train1.toxic.values\ny_train_2 = train2.toxic.values\ny_valid = valid.toxic.values","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:42:35.508535Z","iopub.execute_input":"2021-08-11T06:42:35.509068Z","iopub.status.idle":"2021-08-11T06:42:35.514556Z","shell.execute_reply.started":"2021-08-11T06:42:35.509021Z","shell.execute_reply":"2021-08-11T06:42:35.513644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#轉化成數據集 生成對應的Dataset\n\ntrain_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train, y_train))\n    .repeat() #重複數據集count次數\n    .shuffle(2048) #隨機混洗數據集多元素\n    .batch(BATCH_SIZE) #將數據集多連續元素合成批次\n    .prefetch(AUTO)#將一部分內存加載到cache裡面\n)\ntrain_dataset_2 = (\n    tf.data.Dataset\n    .from_tensor_slices((x_train_2, y_train_2))\n    .repeat() #重複數據集count次數\n    .shuffle(2048) #隨機混洗數據集多元素\n    .batch(BATCH_SIZE) #將數據集多連續元素合成批次\n    .prefetch(AUTO)#將一部分內存加載到cache裡面\n)\nvalid_dataset =(\n    tf.data.Dataset\n    .from_tensor_slices((x_valid, y_valid))\n    .batch(BATCH_SIZE)\n    .cache()\n    .prefetch(AUTO)\n)\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:42:35.515925Z","iopub.execute_input":"2021-08-11T06:42:35.516222Z","iopub.status.idle":"2021-08-11T06:42:35.529381Z","shell.execute_reply.started":"2021-08-11T06:42:35.516191Z","shell.execute_reply":"2021-08-11T06:42:35.528356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#訓練BERT模型\n\ndef build_model(transformer, max_len=512):  #建立模型，輸入句子最大長度512\n    \n    input_word_ids = Input(shape=(max_len,), dtype=tf.int32, name=\"input_word_ids\") #dtype=tf.int 返回數據元素的數據類型int\n    sequence_output = transformer(input_word_ids)[0] #BERT模型的輸出 \n    cls_token = sequence_output[:, 0, :]\n    \n    #激活函數\n    out = tf.keras.layers.Dense(300, activation='relu')(cls_token)\n    out = tf.keras.layers.Dense(128, activation='relu')(out)\n    out = tf.keras.layers.Dense(128, activation='relu')(out)\n    out = Dense(1, activation='sigmoid')(out) #relu線性函數激活 sigmoid非線性激活函數\n    \n    model = Model(inputs=input_word_ids, outputs=out)\n    model.compile(Adam(lr=1e-5), loss='binary_crossentropy', metrics=['accuracy']) #損失函數的用法，Adam是優化器，loss：計算損失\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:42:35.53112Z","iopub.execute_input":"2021-08-11T06:42:35.531793Z","iopub.status.idle":"2021-08-11T06:42:35.540955Z","shell.execute_reply.started":"2021-08-11T06:42:35.531757Z","shell.execute_reply":"2021-08-11T06:42:35.539859Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"checkpoint_filepath = OUTPUT_PATH +\"/checkpoint\"\nmodel_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n    filepath=checkpoint_filepath,\n    save_weights_only=True,\n    monitor='val_accuracy',\n    mode='max',\n    save_best_only=True)","metadata":{"execution":{"iopub.status.busy":"2021-08-09T09:50:34.480312Z","iopub.status.idle":"2021-08-09T09:50:34.480726Z"}}},{"cell_type":"code","source":"%%time\nwith strategy.scope(): #表明分散式執行的程式碼區塊\n    transformer_layer = (\n        transformers.TFDistilBertModel\n        .from_pretrained('distilbert-base-multilingual-cased')\n    )\n    model = build_model(transformer_layer, max_len=MAX_LEN)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:42:35.542673Z","iopub.execute_input":"2021-08-11T06:42:35.543377Z","iopub.status.idle":"2021-08-11T06:43:25.312124Z","shell.execute_reply.started":"2021-08-11T06:42:35.54334Z","shell.execute_reply":"2021-08-11T06:43:25.310883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary() #輸出各層的輸出情況","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:43:25.313927Z","iopub.execute_input":"2021-08-11T06:43:25.314311Z","iopub.status.idle":"2021-08-11T06:43:25.335465Z","shell.execute_reply.started":"2021-08-11T06:43:25.314266Z","shell.execute_reply":"2021-08-11T06:43:25.334368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model.load_weights(WEIGHTS_PATH+\"/weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:43:25.336888Z","iopub.execute_input":"2021-08-11T06:43:25.337276Z","iopub.status.idle":"2021-08-11T06:43:43.613744Z","shell.execute_reply.started":"2021-08-11T06:43:25.337193Z","shell.execute_reply":"2021-08-11T06:43:43.612785Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# n_steps = x_train.shape[0] // BATCH_SIZE #讀取矩陣第一維度的長度\n# train_history = model.fit(\n#     train_dataset,\n#     steps_per_epoch=n_steps,\n#     validation_data=valid_dataset,\n#     epochs=EPOCHS,\n# ) # 使用model.fit()執行訓練過程","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:43:43.61509Z","iopub.execute_input":"2021-08-11T06:43:43.615357Z","iopub.status.idle":"2021-08-11T06:43:43.621742Z","shell.execute_reply.started":"2021-08-11T06:43:43.615331Z","shell.execute_reply":"2021-08-11T06:43:43.620286Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = x_train.shape[0] // BATCH_SIZE #讀取矩陣第一維度的長度\ntrain_history_3 = model.fit(\n    train_dataset_2,\n    steps_per_epoch=n_steps,\n    validation_data=valid_dataset,\n    epochs=EPOCHS,\n) # 使用model.fit()執行訓練過程","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_steps = x_valid.shape[0] // BATCH_SIZE\ntrain_history_2 = model.fit(\n    valid_dataset.repeat(),\n    steps_per_epoch=n_steps,\n    epochs=EPOCHS*2,\n)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:43:43.623435Z","iopub.execute_input":"2021-08-11T06:43:43.623876Z","iopub.status.idle":"2021-08-11T06:43:43.64031Z","shell.execute_reply.started":"2021-08-11T06:43:43.623833Z","shell.execute_reply":"2021-08-11T06:43:43.638233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model.save_weights(\"weights.h5\")","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:43:43.64364Z","iopub.execute_input":"2021-08-11T06:43:43.644296Z","iopub.status.idle":"2021-08-11T06:43:43.651166Z","shell.execute_reply.started":"2021-08-11T06:43:43.644247Z","shell.execute_reply":"2021-08-11T06:43:43.650136Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(model)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:43:43.652981Z","iopub.execute_input":"2021-08-11T06:43:43.653395Z","iopub.status.idle":"2021-08-11T06:43:43.669234Z","shell.execute_reply.started":"2021-08-11T06:43:43.653342Z","shell.execute_reply":"2021-08-11T06:43:43.667423Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_test = fast_encode(test.content.astype(str), fast_tokenizer, maxlen=MAX_LEN)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:43:43.683156Z","iopub.execute_input":"2021-08-11T06:43:43.683651Z","iopub.status.idle":"2021-08-11T06:43:57.023879Z","shell.execute_reply.started":"2021-08-11T06:43:43.68362Z","shell.execute_reply":"2021-08-11T06:43:57.022659Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model.predict()返回值是數值,表示樣本屬於toxic類別的概率\n\ntest_dataset = (\n    tf.data.Dataset\n    .from_tensor_slices(x_test)\n    .batch(BATCH_SIZE)\n)\n\nsub['toxic'] = model.predict(test_dataset, verbose=1)\n\nsub1 = sub[['id', 'toxic']]","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:43:57.026621Z","iopub.execute_input":"2021-08-11T06:43:57.027194Z","iopub.status.idle":"2021-08-11T06:44:16.268224Z","shell.execute_reply.started":"2021-08-11T06:43:57.027139Z","shell.execute_reply":"2021-08-11T06:44:16.267257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 保存架構\n#config = model.get_config()\n#new_model = keras.Model.from_config(config)\n\n#json_config = model.to_json()\n#new_model = keras.models.model_from_json(json_config)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-08-11T06:44:16.275289Z","iopub.execute_input":"2021-08-11T06:44:16.275769Z","iopub.status.idle":"2021-08-11T06:44:16.286212Z","shell.execute_reply.started":"2021-08-11T06:44:16.275663Z","shell.execute_reply":"2021-08-11T06:44:16.285475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# New in Tensorflow 2.4: models can be save locally from TPUs in Tensorflow's SavedModel format\n\n# TPUs need this extra setting to save to local disk, otherwise, they can only save models to GCS (Google Cloud Storage).\n# The setting instructs Tensorflow to retrieve all parameters from the TPU then do the saving from the local VM, not the TPU.\n# This setting does nothing on GPUs.\n#save_locally = tf.saved_model.SaveOptions(experimental_io_device='/job:localhost')\n#model.save('./model', options=save_locally) # saving in Tensorflow's \"saved model\" format","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:44:16.287124Z","iopub.execute_input":"2021-08-11T06:44:16.287373Z","iopub.status.idle":"2021-08-11T06:44:16.298828Z","shell.execute_reply.started":"2021-08-11T06:44:16.287348Z","shell.execute_reply":"2021-08-11T06:44:16.297735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model.save('saved_model', save_format='tf')\n#with strategy.scope():\n#    load_locally = tf.saved_model.LoadOptions(experimental_io_device='/job:localhost')\n#    model = tf.keras.models.load_model(OUTPUT_PATH+\"/model\", options=load_locally) # loading in Tensorflow's \"SavedModel\" format","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:44:16.300396Z","iopub.execute_input":"2021-08-11T06:44:16.300985Z","iopub.status.idle":"2021-08-11T06:44:16.310746Z","shell.execute_reply.started":"2021-08-11T06:44:16.300943Z","shell.execute_reply":"2021-08-11T06:44:16.309666Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#checkpoint_1 = ModelCheckpoint(OUTPUT_PATH, monitor='train_history_2', verbose=1, save_best_only=False, save_weights_only=False, mode='auto', period=1)\ncheckpoint = ModelCheckpoint(OUTPUT_PATH, monitor='history = model.fit()', verbose=1, save_best_only=True, mode='max')\nprint(checkpoint)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:44:16.312164Z","iopub.execute_input":"2021-08-11T06:44:16.312436Z","iopub.status.idle":"2021-08-11T06:44:16.322812Z","shell.execute_reply.started":"2021-08-11T06:44:16.31241Z","shell.execute_reply":"2021-08-11T06:44:16.32196Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(sub1)\nprint(sub2)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:44:16.324001Z","iopub.execute_input":"2021-08-11T06:44:16.324265Z","iopub.status.idle":"2021-08-11T06:44:16.34428Z","shell.execute_reply.started":"2021-08-11T06:44:16.32424Z","shell.execute_reply":"2021-08-11T06:44:16.343275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub1.rename(columns={'toxic':'toxic1'}, inplace=True)\nsub2.rename(columns={'toxic':'toxic2'}, inplace=True) #命名文件或目錄\nsub3 = pd.merge(sub1, sub2, how='left', on='id')\n\nsub3['toxic'] = (sub3['toxic1'] * 0.1) + (sub3['toxic2'] * 0.9)\nsub3['toxic'] = (sub3['toxic2'] * 0.39) + (sub3['toxic'] * 0.61)\n\nsub3[['id', 'toxic']].to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-08-11T06:44:16.345467Z","iopub.execute_input":"2021-08-11T06:44:16.346025Z","iopub.status.idle":"2021-08-11T06:44:16.650159Z","shell.execute_reply.started":"2021-08-11T06:44:16.345984Z","shell.execute_reply":"2021-08-11T06:44:16.649299Z"},"trusted":true},"execution_count":null,"outputs":[]}]}