{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Feedback Prize - BERT\n\nThis is a **training** notebook. The inference notebook can be found [Feedback Prize - BERT Inference](https://www.kaggle.com/code/morodertobias/feedback-prize-bert-inference/notebook).\n\nWe fine-tune a pretrained BERT base model and start with a classification acting on ``discourse_type [SEP] discourse_text`` only.\n\n\n## References\n\nWe mainly used the following references, also corresponding to former challenges:\n\n- [Semantic Similarity with BERT](https://keras.io/examples/nlp/semantic_similarity_with_bert/)\n- [US Phrase Matching: TF-Keras Train [TPU]](https://www.kaggle.com/code/mohamadmerchant/us-phrase-matching-tf-keras-train-tpu/notebook)\n- [TensorFlow - LongFormer - NER - [CV 0.633]](https://www.kaggle.com/code/cdeotte/tensorflow-longformer-ner-cv-0-633/notebook)\n- [【Tensorflow】FeedBack BERT-Baseline](https://www.kaggle.com/code/imvision12/tensorflow-feedback-bert-baseline/notebook)\n- [TFRecord Experiments - Upsample and Coarse Dropout](https://www.kaggle.com/code/cdeotte/tfrecord-experiments-upsample-and-coarse-dropout)\n- [Tez for feedback v2.0](https://www.kaggle.com/code/abhishek/tez-for-feedback-v2-0)","metadata":{}},{"cell_type":"code","source":"!pip install -qU scikit-learn 2> /dev/null\n!pip install -q transformers==4.18.0 2> /dev/null","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:42:22.715545Z","iopub.execute_input":"2022-07-18T21:42:22.716239Z","iopub.status.idle":"2022-07-18T21:42:56.216714Z","shell.execute_reply.started":"2022-07-18T21:42:22.716128Z","shell.execute_reply":"2022-07-18T21:42:56.215590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport pathlib\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nimport transformers","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:42:58.416748Z","iopub.execute_input":"2022-07-18T21:42:58.417085Z","iopub.status.idle":"2022-07-18T21:43:06.387146Z","shell.execute_reply.started":"2022-07-18T21:42:58.417046Z","shell.execute_reply":"2022-07-18T21:43:06.386017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tf.__version__)\nprint(transformers.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:43:06.389018Z","iopub.execute_input":"2022-07-18T21:43:06.389316Z","iopub.status.idle":"2022-07-18T21:43:06.394926Z","shell.execute_reply.started":"2022-07-18T21:43:06.389285Z","shell.execute_reply":"2022-07-18T21:43:06.393832Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Device:', tpu.master())\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept:\n    print(\"TPU failed!\")\n    tpu = None\n    strategy = tf.distribute.get_strategy()\nprint('Number of replicas:', strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:43:06.396725Z","iopub.execute_input":"2022-07-18T21:43:06.397084Z","iopub.status.idle":"2022-07-18T21:43:12.813986Z","shell.execute_reply.started":"2022-07-18T21:43:06.397041Z","shell.execute_reply":"2022-07-18T21:43:12.813054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config\n\nLet us use a config object holding all parameters, settings and configs.","metadata":{}},{"cell_type":"code","source":"class Config:\n    seed = 887\n    model_name = \"tpu_bert_v15\"\n    n_fold = 5\n    one_fold = False\n    # inputs\n    input_dir = pathlib.Path(\"/kaggle/input/feedback-prize-effectiveness/\")\n    path_train = input_dir / \"train.csv\"\n    train_dir = input_dir / \"train\"\n    path_test = input_dir / \"test.csv\"\n    test_dir = input_dir / \"test\"\n    path_submission = input_dir / \"sample_submission.csv\"\n    labels = [\"Ineffective\", \"Adequate\", \"Effective\"]\n    label_dict = {v: i for i, v in enumerate(labels)}\n    num_classes = len(labels)\n    id_col = \"discourse_id\"\n    # model\n    pretrained = \"bert-base-uncased\"\n    pretrained_dir = pathlib.Path(\"/kaggle/working/pretrained\")\n    max_len = 512\n    dropout = 0.4\n    # train\n    label_smoothing = 0.1\n    learning_rate = 3e-6\n    batch_size = 128\n    steps_per_epoch = 100\n    epochs = 100  # 50\n    patience = 8\n    verbose = 1\n    \ncfg = Config()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:43:12.825878Z","iopub.execute_input":"2022-07-18T21:43:12.826291Z","iopub.status.idle":"2022-07-18T21:43:12.839955Z","shell.execute_reply.started":"2022-07-18T21:43:12.826260Z","shell.execute_reply":"2022-07-18T21:43:12.839270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Download pretrained files\n\nLet us follow the ideas of [TensorFlow - LongFormer - NER - [CV 0.633]](https://www.kaggle.com/code/cdeotte/tensorflow-longformer-ner-cv-0-633/notebook) and download  tokenizer and model, and store them into a notebook output folder in order to use them directly in the inference notebook or for creating a versioned dataset.","metadata":{}},{"cell_type":"code","source":"tokenizer = transformers.AutoTokenizer.from_pretrained(cfg.pretrained)\ntokenizer.save_pretrained(cfg.pretrained_dir)\nconfig = transformers.AutoConfig.from_pretrained(cfg.pretrained)\nconfig.save_pretrained(cfg.pretrained_dir)\nbase_model = transformers.TFAutoModel.from_pretrained(cfg.pretrained, config=config, from_pt=True)\nbase_model.save_pretrained(cfg.pretrained_dir)\nos.listdir(cfg.pretrained_dir)","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:43:13.305410Z","iopub.execute_input":"2022-07-18T21:43:13.306132Z","iopub.status.idle":"2022-07-18T21:43:49.558059Z","shell.execute_reply.started":"2022-07-18T21:43:13.306076Z","shell.execute_reply":"2022-07-18T21:43:49.556627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data Preparation\n\nLoad data and prepare the text ``discourse_type [SEP] discourse_text`` used in classification. \n\nThe tokenizer will encode this into ``[CLS] discourse_type [SEP] discourse_text [SEP]`` as can be seen by the example below.","metadata":{}},{"cell_type":"code","source":"data = pd.read_csv(cfg.path_train)\ndata[\"label\"] = data[\"discourse_effectiveness\"].map(cfg.label_dict)\ndata","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:43:56.648205Z","iopub.execute_input":"2022-07-18T21:43:56.649381Z","iopub.status.idle":"2022-07-18T21:43:57.030647Z","shell.execute_reply.started":"2022-07-18T21:43:56.649321Z","shell.execute_reply":"2022-07-18T21:43:57.029630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tokenizer = transformers.AutoTokenizer.from_pretrained(cfg.pretrained_dir)\ntokenizer","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:45:41.328891Z","iopub.execute_input":"2022-07-18T21:45:41.329196Z","iopub.status.idle":"2022-07-18T21:45:41.369168Z","shell.execute_reply.started":"2022-07-18T21:45:41.329166Z","shell.execute_reply":"2022-07-18T21:45:41.368357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data[\"text\"] = data[\"discourse_type\"] + tokenizer.sep_token + data[\"discourse_text\"]\ndata","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:45:42.092616Z","iopub.execute_input":"2022-07-18T21:45:42.093123Z","iopub.status.idle":"2022-07-18T21:45:42.133511Z","shell.execute_reply.started":"2022-07-18T21:45:42.093077Z","shell.execute_reply":"2022-07-18T21:45:42.132717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Look at a random example","metadata":{"execution":{"iopub.status.busy":"2022-06-16T12:23:45.382576Z","iopub.execute_input":"2022-06-16T12:23:45.383666Z","iopub.status.idle":"2022-06-16T12:23:45.389863Z","shell.execute_reply.started":"2022-06-16T12:23:45.383624Z","shell.execute_reply":"2022-06-16T12:23:45.389014Z"}}},{"cell_type":"code","source":"rec = data.sample(n=1).iloc[0].to_dict()\nrec","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:45:43.422085Z","iopub.execute_input":"2022-07-18T21:45:43.422808Z","iopub.status.idle":"2022-07-18T21:45:43.444405Z","shell.execute_reply.started":"2022-07-18T21:45:43.422767Z","shell.execute_reply":"2022-07-18T21:45:43.443386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"original:\", rec[\"text\"])\nprint(\"tokenized:\", tokenizer.tokenize(rec[\"text\"]))\nprint(\"encode_plus:\", tokenizer.encode_plus(rec[\"text\"]))\nprint(\"decoded:\", tokenizer.decode(tokenizer.encode_plus(rec[\"text\"])['input_ids']))","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:45:43.925825Z","iopub.execute_input":"2022-07-18T21:45:43.926134Z","iopub.status.idle":"2022-07-18T21:45:43.938256Z","shell.execute_reply.started":"2022-07-18T21:45:43.926105Z","shell.execute_reply":"2022-07-18T21:45:43.937151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset\n\nWe like to map ``tokenizer.encode_plus`` to all texts in the dataset; this does not seem to directly work on a tensorflow dataset, hence we compute the results beforehand... as done in the reference notebooks.","metadata":{}},{"cell_type":"code","source":"options = tf.data.Options()\noptions.experimental_distribute.auto_shard_policy = tf.data.experimental.AutoShardPolicy.OFF\n\n\ndef encode_text(text):\n    \"\"\"Encode text with tokenizer and return dictionary of numpy results.\"\"\"\n    encoded = tokenizer.batch_encode_plus(\n        text,\n        max_length=cfg.max_len,\n        padding='max_length',\n        truncation=True,\n        return_attention_mask=True,\n        return_token_type_ids=True,\n        return_tensors=\"tf\",\n    )\n    return {\n        \"input_ids\": encoded[\"input_ids\"].numpy(),\n        \"attention_masks\": encoded[\"attention_mask\"].numpy(),\n        \"token_type_ids\": encoded[\"token_type_ids\"].numpy(),\n    }\n\n\ndef get_dataset(data, batch_size=cfg.batch_size, shuffle=False, repeat=False, include_label=True):\n    \"\"\"Get dataset\"\"\"\n    encoded_text = encode_text(data['text'].to_list())\n    tensor_slices = encoded_text\n    if include_label:\n        label = tf.one_hot(data[\"label\"].to_list(), cfg.num_classes)\n        tensor_slices = (encoded_text, label)\n    ds = tf.data.Dataset.from_tensor_slices(tensor_slices)\n    ds = ds.with_options(options)\n    if repeat:\n        ds = ds.repeat()\n    if shuffle:\n        ds = ds.shuffle(2048)\n    ds = ds.batch(batch_size)\n    ds = ds.prefetch(tf.data.AUTOTUNE)\n    return ds","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:45:45.073000Z","iopub.execute_input":"2022-07-18T21:45:45.073419Z","iopub.status.idle":"2022-07-18T21:45:45.085575Z","shell.execute_reply.started":"2022-07-18T21:45:45.073366Z","shell.execute_reply":"2022-07-18T21:45:45.084437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Verify dataset creation","metadata":{}},{"cell_type":"code","source":"data.iloc[:5]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:45:46.118172Z","iopub.execute_input":"2022-07-18T21:45:46.118517Z","iopub.status.idle":"2022-07-18T21:45:46.135653Z","shell.execute_reply.started":"2022-07-18T21:45:46.118482Z","shell.execute_reply":"2022-07-18T21:45:46.134482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds = get_dataset(data.iloc[:5], batch_size=2)\nds.element_spec","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:45:46.543499Z","iopub.execute_input":"2022-07-18T21:45:46.544482Z","iopub.status.idle":"2022-07-18T21:45:46.567946Z","shell.execute_reply.started":"2022-07-18T21:45:46.544441Z","shell.execute_reply":"2022-07-18T21:45:46.567010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"elem = next(iter(ds))\nelem","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:45:47.351840Z","iopub.execute_input":"2022-07-18T21:45:47.352144Z","iopub.status.idle":"2022-07-18T21:45:47.374848Z","shell.execute_reply.started":"2022-07-18T21:45:47.352116Z","shell.execute_reply":"2022-07-18T21:45:47.373791Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Model\n\nThe model is straightforward, i.e., inputs > base_model > output head.","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras import Model, layers, losses, optimizers, metrics, callbacks, backend","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:45:48.639587Z","iopub.execute_input":"2022-07-18T21:45:48.640381Z","iopub.status.idle":"2022-07-18T21:45:48.646398Z","shell.execute_reply.started":"2022-07-18T21:45:48.640330Z","shell.execute_reply":"2022-07-18T21:45:48.645322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# idea from https://www.kaggle.com/code/abhishek/tez-for-feedback-v2-0\n# careful, own implementation following https://www.tensorflow.org/guide/keras/masking_and_padding\n\n\nclass MeanPooler(layers.Layer):\n    def call(self, inputs, mask=None):\n        broadcast_float_mask = tf.expand_dims(tf.cast(mask, \"float32\"), -1)\n        masked_inputs = inputs * broadcast_float_mask\n        inputs_sum = tf.reduce_sum(masked_inputs, axis=1)\n        mask_sum = tf.reduce_sum(broadcast_float_mask, axis=1)\n        mask_sum = tf.math.maximum(mask_sum, 1e-9)\n        return inputs_sum / mask_sum","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:45:48.944425Z","iopub.execute_input":"2022-07-18T21:45:48.944777Z","iopub.status.idle":"2022-07-18T21:45:48.952408Z","shell.execute_reply.started":"2022-07-18T21:45:48.944739Z","shell.execute_reply":"2022-07-18T21:45:48.951277Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_model():\n    # inputs\n    input_ids = layers.Input(shape=(cfg.max_len,), dtype=\"int32\", name=\"input_ids\")\n    attention_masks = layers.Input(shape=(cfg.max_len,), dtype=\"int32\", name=\"attention_masks\")\n    token_type_ids = layers.Input(shape=(cfg.max_len,), dtype=\"int32\", name=\"token_type_ids\")\n    # base_model\n    base_model_config = transformers.AutoConfig.from_pretrained(\n        cfg.pretrained_dir / \"config.json\"\n    )\n    base_model = transformers.TFAutoModel.from_pretrained(\n        cfg.pretrained_dir / \"tf_model.h5\", config=base_model_config\n    )\n    # base_model.trainable = False\n    base_model_output = base_model(\n        input_ids, attention_mask=attention_masks, token_type_ids=token_type_ids\n    )\n    # x = base_model_output.last_hidden_state[:, 0, :]\n    x = MeanPooler()(base_model_output.last_hidden_state, mask=attention_masks)\n    # head\n    x = layers.Dropout(cfg.dropout)(x)\n    output = layers.Dense(cfg.num_classes, activation='softmax')(x)\n    model = Model(\n        inputs=[input_ids, attention_masks, token_type_ids],\n        outputs=output,\n        name=cfg.model_name,\n    )\n    # compile\n    model.compile(\n        optimizer=optimizers.Adam(cfg.learning_rate),\n        loss=losses.CategoricalCrossentropy(label_smoothing=cfg.label_smoothing),\n        metrics=[\"acc\", metrics.CategoricalCrossentropy(name='xentropy')],\n    )\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:45:49.423826Z","iopub.execute_input":"2022-07-18T21:45:49.425112Z","iopub.status.idle":"2022-07-18T21:45:49.436460Z","shell.execute_reply.started":"2022-07-18T21:45:49.424964Z","shell.execute_reply":"2022-07-18T21:45:49.435551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Verify model creation","metadata":{}},{"cell_type":"code","source":"backend.clear_session()\nwith strategy.scope():\n    model = create_model()\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:45:50.409381Z","iopub.execute_input":"2022-07-18T21:45:50.409757Z","iopub.status.idle":"2022-07-18T21:46:10.623147Z","shell.execute_reply.started":"2022-07-18T21:45:50.409721Z","shell.execute_reply":"2022-07-18T21:46:10.621949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.predict(elem[0]), elem[1]","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:46:10.624918Z","iopub.execute_input":"2022-07-18T21:46:10.625179Z","iopub.status.idle":"2022-07-18T21:46:20.155910Z","shell.execute_reply.started":"2022-07-18T21:46:10.625151Z","shell.execute_reply":"2022-07-18T21:46:20.154674Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Training helper functions","metadata":{}},{"cell_type":"code","source":"import sklearn.metrics as sk_metrics","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:46:20.158793Z","iopub.execute_input":"2022-07-18T21:46:20.159586Z","iopub.status.idle":"2022-07-18T21:46:20.964270Z","shell.execute_reply.started":"2022-07-18T21:46:20.159544Z","shell.execute_reply":"2022-07-18T21:46:20.963290Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_callbacks(filepath):\n    \"\"\"Create callbacks for training\"\"\"\n    return [\n        callbacks.ModelCheckpoint(\n            filepath=filepath, monitor=\"val_xentropy\", save_best_only=True, save_weights_only=True, verbose=1\n        ),\n        callbacks.EarlyStopping(\n            patience=cfg.patience, monitor=\"val_xentropy\", restore_best_weights=False, verbose=1\n        ),\n    ]\n\n\ndef show_history(history):\n    \"\"\"Show history\"\"\"\n    history_df = pd.DataFrame(history.history)\n    history_df.index = pd.Index(history.epoch, name=\"epoch\")\n    display(\n        history_df.style.highlight_min(\n            color=\"green\", subset=[\"val_loss\", \"val_xentropy\"]\n        ).highlight_max(color=\"green\", subset=[\"val_acc\"])\n    )\n    fig, ax = plt.subplots(1, 2, figsize=(16, 8))\n    history_df[[\"loss\", \"val_loss\", \"xentropy\", \"val_xentropy\"]].plot(ax=ax[0], title=\"loss/xentropy\")\n    history_df[[\"acc\", \"val_acc\"]].plot(ax=ax[1], title=\"acc\")\n    plt.tight_layout()\n    plt.show()\n    \n    \ndef compute_oof(model, valid):\n    \"\"\"Compute OOF\"\"\"\n    valid_ds = get_dataset(valid)\n    pred = model.predict(valid_ds, verbose=0)\n    oof = pd.DataFrame(pred, columns=cfg.labels, index=valid[cfg.id_col])\n    oof[\"label\"] = valid.set_index(cfg.id_col)[\"label\"]\n    return oof    \n    \n\ndef compute_score(x):\n    \"\"\"Compute score\"\"\"\n    return sk_metrics.log_loss(y_true=x[\"label\"], y_pred=x[cfg.labels])\n\n\ndef create_lr_scheduler(train_ds):\n    \"\"\"Create learning rate scheduler\"\"\"\n    lr_scheduler = optimizers.schedules.PolynomialDecay(\n        initial_learning_rate=cfg.init_learning_rate,\n        decay_steps=(len(train_ds) * cfg.decay_epochs),\n        end_learning_rate=cfg.end_learning_rate\n    )\n    print(\"lr_scheduler.get_config():\", lr_scheduler.get_config())\n    return lr_scheduler","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:46:22.266198Z","iopub.execute_input":"2022-07-18T21:46:22.266530Z","iopub.status.idle":"2022-07-18T21:46:22.284331Z","shell.execute_reply.started":"2022-07-18T21:46:22.266502Z","shell.execute_reply":"2022-07-18T21:46:22.283199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_training(train, valid, filename):\n    \"\"\"Run training\"\"\"\n    # https://www.kaggle.com/code/cdeotte/tfrecord-experiments-upsample-and-coarse-dropout\n    if tpu:\n        tf.tpu.experimental.initialize_tpu_system()\n    # create datasets\n    train_ds = get_dataset(train, repeat=True, shuffle=True)\n    valid_ds = get_dataset(valid)\n    # create model\n    backend.clear_session()\n    with strategy.scope():\n        model = create_model()\n    # fit\n    hist = model.fit(\n        train_ds,\n        epochs=cfg.epochs,\n        steps_per_epoch=cfg.steps_per_epoch,\n        validation_data=valid_ds,\n        callbacks=create_callbacks(filename),\n        verbose=cfg.verbose,\n    )\n    model.load_weights(filename)\n    # oof\n    oof = compute_oof(model, valid)\n    return hist, oof","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:46:22.918215Z","iopub.execute_input":"2022-07-18T21:46:22.918561Z","iopub.status.idle":"2022-07-18T21:46:22.927439Z","shell.execute_reply.started":"2022-07-18T21:46:22.918526Z","shell.execute_reply":"2022-07-18T21:46:22.925972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Run training\n\nWe use stratified splitting in creating training and validation folds.","metadata":{"execution":{"iopub.status.busy":"2022-06-16T15:41:56.057774Z","iopub.execute_input":"2022-06-16T15:41:56.058314Z","iopub.status.idle":"2022-06-16T15:41:56.063774Z","shell.execute_reply.started":"2022-06-16T15:41:56.058279Z","shell.execute_reply":"2022-06-16T15:41:56.063005Z"}}},{"cell_type":"code","source":"from sklearn.model_selection import StratifiedGroupKFold","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:46:24.818254Z","iopub.execute_input":"2022-07-18T21:46:24.819163Z","iopub.status.idle":"2022-07-18T21:46:24.829584Z","shell.execute_reply.started":"2022-07-18T21:46:24.819125Z","shell.execute_reply":"2022-07-18T21:46:24.828728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sgkf = StratifiedGroupKFold(n_splits=cfg.n_fold, shuffle=True, random_state=cfg.seed)\ndata[\"stratified_label\"] = data[\"discourse_type\"] + data[\"label\"].astype('str')\nsgkf, data[\"stratified_label\"].value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:49:39.062969Z","iopub.execute_input":"2022-07-18T21:49:39.063744Z","iopub.status.idle":"2022-07-18T21:49:39.152294Z","shell.execute_reply.started":"2022-07-18T21:49:39.063688Z","shell.execute_reply":"2022-07-18T21:49:39.151270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\nd_oof = {}\nfor fold, (iloc_train, iloc_valid) in enumerate(sgkf.split(data, data[\"stratified_label\"], data['essay_id'])):\n    print(f\"fold: {fold}\")\n    train = data.iloc[iloc_train]\n    valid = data.iloc[iloc_valid]\n    model_filepath = f\"weights__{cfg.model_name}__fold-{fold}.h5\"\n    print(f\"#train: {len(train)},  #valid: {len(valid)} \")\n    print(f\"model_filepath: {model_filepath}\")\n    hist, oof = run_training(train, valid, model_filepath)\n    print(\"OOF score:\", compute_score(oof))\n    show_history(hist)\n    d_oof[fold] = oof\n    if cfg.one_fold:\n        break    ","metadata":{"execution":{"iopub.status.busy":"2022-07-18T21:49:48.483732Z","iopub.execute_input":"2022-07-18T21:49:48.484576Z","iopub.status.idle":"2022-07-18T21:53:23.935203Z","shell.execute_reply.started":"2022-07-18T21:49:48.484531Z","shell.execute_reply":"2022-07-18T21:53:23.933996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# OOF\n\nFinalize OOF prediction and score summary.","metadata":{}},{"cell_type":"code","source":"oof = pd.concat(d_oof, names=['fold']).reset_index('fold')\noof.to_csv(\"oof.csv\")\nscore_by_fold = oof.groupby('fold').apply(compute_score)\ndisplay(score_by_fold)\nscore = compute_score(oof)\nprint(f\"\\nOOF score: {score:.6f}\")","metadata":{"execution":{"iopub.status.busy":"2022-06-17T19:49:56.08034Z","iopub.execute_input":"2022-06-17T19:49:56.080846Z","iopub.status.idle":"2022-06-17T19:49:56.467762Z","shell.execute_reply.started":"2022-06-17T19:49:56.080805Z","shell.execute_reply":"2022-06-17T19:49:56.466719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}