{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"### Based on the Tensorflow Keras baseline notebook from https://www.kaggle.com/code/nadare/tensorflow-baseline-lb-0-681\n\n- Configurable hyperparameters are placed on top for easy tuning.\n- Seed is added for reproducible results\n\n--------------------\n\n### For CSC532/621 Machine Learning class:\n- Take note of the differences between this current notebook and the source\n- Document your notebook thoroughly (use ChatGPT/search engines as necessary but put the content in quotations and cite the source)\n- Only share and discuss with your teammates who are officially on your Kaggle team\n- Use the following to keep track of your runs (feel free to add additional columns):\n\n## Changelog\n\n|Version | Description | LB | Note |\n| --: | -- | -- | -- |\n| V0 | Soure [Notebook](https://www.kaggle.com/code/nadare/tensorflow-baseline-lb-0-681) | `0.681` |baseline (TF Keras DCN V2)|\n| V1 |Some finetuning of hyperparameters and added seed for reproducible results | `0.682` ||\n| V2 |Added 1 num_linear layer and used the original 'gelu' activation function  | `0.682` |LB slightly lower than V1|\n| V3 |Quick save |  ||\n| V4 |Quick save |  |changed title to reflect CSC532 Machine Learning class|\n| V5 |Epochs to 60 & Num_linear to 3|  | |","metadata":{}},{"cell_type":"code","source":"import gc\nimport os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nimport random\n\nfrom tqdm.notebook import tqdm","metadata":{"ExecuteTime":{"end_time":"2023-02-19T14:18:49.457013Z","start_time":"2023-02-19T14:18:47.989756Z"},"execution":{"iopub.status.busy":"2023-03-07T02:08:33.779694Z","iopub.execute_input":"2023-03-07T02:08:33.780452Z","iopub.status.idle":"2023-03-07T02:08:42.547760Z","shell.execute_reply.started":"2023-03-07T02:08:33.780354Z","shell.execute_reply":"2023-03-07T02:08:42.546777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Config\nEPOCHS = 60 # baseline was 10\nSEED = 42 # baseline model was stochastic\nthreshold = 100\nLR = 1.e-3\nepsilon = 1.e-9\n\nFEAT_DIM = 8\nOUT_DIM = 32\nNUM_CROSS = 5\nNUM_LINEAR = 3 # baseline was 0\nBATCH = 128\nACT_FUN = \"gelu\"","metadata":{"execution":{"iopub.status.busy":"2023-03-07T02:08:42.549426Z","iopub.execute_input":"2023-03-07T02:08:42.550202Z","iopub.status.idle":"2023-03-07T02:08:42.555109Z","shell.execute_reply.started":"2023-03-07T02:08:42.550169Z","shell.execute_reply":"2023-03-07T02:08:42.554280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Seed all random number generators\ndef seed_everything(seed=SEED):\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    random.seed(seed)\n    np.random.seed(seed)\n    tf.random.set_seed(seed)\n\nseed_everything()","metadata":{"execution":{"iopub.status.busy":"2023-03-07T02:08:42.556414Z","iopub.execute_input":"2023-03-07T02:08:42.556919Z","iopub.status.idle":"2023-03-07T02:08:42.567228Z","shell.execute_reply.started":"2023-03-07T02:08:42.556888Z","shell.execute_reply":"2023-03-07T02:08:42.566246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"feature_columns = [\"event_name\", \"name\", \"level\", \"fqid\", \"room_fqid\", \"text_fqid\", \"text\"]\nfeature_ix_columns = [col + \"_ix\" for col in feature_columns]\nusecols = [\"session_id\", \"level_group\"] + feature_columns\n\ntrain_df = pd.read_csv(\"../input/predict-student-performance-from-game-play/train.csv\", usecols=usecols)","metadata":{"ExecuteTime":{"end_time":"2023-02-19T14:19:02.846725Z","start_time":"2023-02-19T14:18:49.458018Z"},"execution":{"iopub.status.busy":"2023-03-07T02:08:42.569729Z","iopub.execute_input":"2023-03-07T02:08:42.570277Z","iopub.status.idle":"2023-03-07T02:09:36.162668Z","shell.execute_reply.started":"2023-03-07T02:08:42.570244Z","shell.execute_reply":"2023-03-07T02:09:36.161448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# category to int\nfeature_dicts = {}\nfor col in tqdm(feature_columns):\n    vc = train_df[col].fillna(\"nan_value\").astype(str).value_counts()\n    vc = vc[vc.values >= threshold]\n    feature_dict = {k: i for i, k in enumerate(vc.index, start=1)}\n    train_df[col + \"_ix\"] = np.vectorize(lambda x: feature_dict.get(x, 0))(train_df[col].fillna(\"nan_value\").astype(str)).astype(np.int32)\n    feature_dicts[col] = feature_dict\n    del train_df[col]\n    gc.collect()\nfeature_num_vocabs = [len(feature_dicts[k]) + 1 for k in feature_columns]\n\nfor col, num_vocab in zip(feature_columns, feature_num_vocabs):\n    print(col, num_vocab)","metadata":{"ExecuteTime":{"end_time":"2023-02-19T14:19:24.566707Z","start_time":"2023-02-19T14:19:02.847734Z"},"execution":{"iopub.status.busy":"2023-03-07T02:09:36.164076Z","iopub.execute_input":"2023-03-07T02:09:36.164395Z","iopub.status.idle":"2023-03-07T02:10:40.256455Z","shell.execute_reply.started":"2023-03-07T02:09:36.164366Z","shell.execute_reply":"2023-03-07T02:10:40.255672Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df[\"level_group\"] = train_df[\"level_group\"].map({k: v for v, k in enumerate(['0-4', '5-12', '13-22'])})","metadata":{"execution":{"iopub.status.busy":"2023-03-07T02:10:40.257544Z","iopub.execute_input":"2023-03-07T02:10:40.258276Z","iopub.status.idle":"2023-03-07T02:10:41.462788Z","shell.execute_reply.started":"2023-03-07T02:10:40.258244Z","shell.execute_reply":"2023-03-07T02:10:41.461696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.info()","metadata":{"execution":{"iopub.status.busy":"2023-03-07T02:10:41.464220Z","iopub.execute_input":"2023-03-07T02:10:41.464555Z","iopub.status.idle":"2023-03-07T02:10:41.479985Z","shell.execute_reply.started":"2023-03-07T02:10:41.464502Z","shell.execute_reply":"2023-03-07T02:10:41.478700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# aggregate unique_feature\nunique_train_df = train_df[feature_ix_columns].drop_duplicates()\nunique_train_df[\"unique_ix\"] = np.arange(unique_train_df.shape[0])\n\ntrain_df = train_df.merge(unique_train_df,\n                          on=feature_ix_columns,\n                          how=\"left\")\nunique_feature = tf.convert_to_tensor(unique_train_df[feature_ix_columns].values)\n\nfor col in feature_ix_columns:\n    del train_df[col]\n    gc.collect()\n\nunique_feature.shape","metadata":{"ExecuteTime":{"end_time":"2023-02-19T14:19:29.917509Z","start_time":"2023-02-19T14:19:24.568708Z"},"execution":{"iopub.status.busy":"2023-03-07T02:10:41.481577Z","iopub.execute_input":"2023-03-07T02:10:41.481932Z","iopub.status.idle":"2023-03-07T02:10:48.301586Z","shell.execute_reply.started":"2023-03-07T02:10:41.481900Z","shell.execute_reply":"2023-03-07T02:10:48.300581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# session to index\ntrain_session_map = {k: i for i, k in enumerate(train_df[\"session_id\"].unique())}\ntrain_df[\"session_ix\"] = train_df[\"session_id\"].map(train_session_map)\nnum_session = train_df[\"session_ix\"].max() + 1\nnum_session","metadata":{"ExecuteTime":{"end_time":"2023-02-19T14:19:30.105616Z","start_time":"2023-02-19T14:19:29.91951Z"},"execution":{"iopub.status.busy":"2023-03-07T02:10:48.302964Z","iopub.execute_input":"2023-03-07T02:10:48.303269Z","iopub.status.idle":"2023-03-07T02:10:48.619307Z","shell.execute_reply.started":"2023-03-07T02:10:48.303241Z","shell.execute_reply":"2023-03-07T02:10:48.618278Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# history to ragged tensor\ntrain_histories = []\nfor i in range(3):\n    tmp_df = train_df[[\"session_ix\", \"unique_ix\", \"level_group\"]].query(f\"level_group == {i}\")\n    train_histories.append(tf.RaggedTensor.from_value_rowids(values=tf.convert_to_tensor(tmp_df[\"unique_ix\"].astype(np.int64)),\n                                                             value_rowids=tf.convert_to_tensor(tmp_df[\"session_ix\"].astype(np.int64)),\n                                                             nrows=num_session))","metadata":{"ExecuteTime":{"end_time":"2023-02-19T14:19:31.689464Z","start_time":"2023-02-19T14:19:30.106617Z"},"execution":{"iopub.status.busy":"2023-03-07T02:10:48.623138Z","iopub.execute_input":"2023-03-07T02:10:48.623529Z","iopub.status.idle":"2023-03-07T02:10:50.509732Z","shell.execute_reply.started":"2023-03-07T02:10:48.623477Z","shell.execute_reply":"2023-03-07T02:10:50.508751Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del train_df\ngc.collect()","metadata":{"execution":{"iopub.status.busy":"2023-03-07T02:10:50.513560Z","iopub.execute_input":"2023-03-07T02:10:50.513904Z","iopub.status.idle":"2023-03-07T02:10:50.701572Z","shell.execute_reply.started":"2023-03-07T02:10:50.513875Z","shell.execute_reply":"2023-03-07T02:10:50.700365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# label to tensor\ntrain_label_df = pd.read_csv(\"../input/predict-student-performance-from-game-play/train_labels.csv\")\ntrain_label_df[\"session_ix\"] = train_label_df[\"session_id\"].map(lambda x: train_session_map[int(x.split(\"_\")[0])])\ntrain_label_df[\"question_ix\"] = train_label_df[\"session_id\"].map(lambda x: int(x.split(\"_\")[1][1:]))\n\ntrain_label = train_label_df.pivot_table(columns=\"question_ix\",\n                                         index=\"session_ix\",\n                                         values=\"correct\",\n                                         aggfunc=\"first\").loc[np.arange(num_session)].values.astype(np.float32)\ntrain_labels = [tf.convert_to_tensor(train_label[:, :3]), tf.convert_to_tensor(train_label[:, 3:13]), tf.convert_to_tensor(train_label[:, 13:])]\n\ndel train_label_df\ngc.collect()","metadata":{"ExecuteTime":{"end_time":"2023-02-19T14:19:31.953894Z","start_time":"2023-02-19T14:19:31.690973Z"},"execution":{"iopub.status.busy":"2023-03-07T02:10:50.703115Z","iopub.execute_input":"2023-03-07T02:10:50.703470Z","iopub.status.idle":"2023-03-07T02:10:51.616206Z","shell.execute_reply.started":"2023-03-07T02:10:50.703440Z","shell.execute_reply":"2023-03-07T02:10:51.615327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# define dataset\nclass DataLoader():\n\n    def __init__(self, train_histories, train_labels, train_unique_feataures):\n        self.train_histories = train_histories\n        self.train_labels = train_labels\n        self.train_unique_feataures = train_unique_feataures\n\n    def call(self, inputs):\n        session_ix = inputs\n        raw_histories = [tf.gather(self.train_histories[i], session_ix) for i in range(3)]\n        historiy_lengths = [tf.shape(h.values)[0] for h in raw_histories]\n        unique_ix, unique_idx = tf.unique(tf.concat([h.values for h in raw_histories], axis=0))\n        history_values = tf.split(unique_idx, historiy_lengths, axis=0)\n        histories = [tf.RaggedTensor.from_value_rowids(values=history_values[i],\n                                                       value_rowids=raw_histories[i].value_rowids(),\n                                                       nrows=raw_histories[i].nrows()) for i in range(3)]\n        inputs = {}\n        for i in range(3):\n            inputs[f\"history_{i}\"] = histories[i]\n            label = tf.gather(self.train_labels[i], session_ix)\n            inputs[f\"label_{i}\"] = label\n        inputs[\"unique_feature\"] = tf.gather(self.train_unique_feataures, unique_ix)\n        return inputs\n\n\n# define model\nclass DCNV2Model(tf.keras.Model):\n\n    def __init__(self, feature_num_vocabs, feat_dim, out_dim, num_cross, num_linear):\n        super(DCNV2Model, self).__init__()\n        self.num_features = len(feature_num_vocabs)\n\n        self.feature_num_vocabs = feature_num_vocabs\n        self.feat_dim = feat_dim\n        self.out_dim = out_dim\n        self.num_cross = num_cross\n        self.num_linear = num_linear\n\n        self.input_dim = feat_dim * self.num_features\n        self.embedding_layers = [tf.keras.layers.Embedding(feature_num_vocabs[i], feat_dim) for i in range(self.num_features)]\n        self.cross_in_layers = [tf.keras.layers.Dense(self.feat_dim) for _ in range(self.num_cross)]\n        self.cross_out_layers = [tf.keras.layers.Dense(self.input_dim) for _ in range(self.num_cross)]\n        self.linear_layers = [tf.keras.layers.Dense(self.input_dim, activation=ACT_FUN) for _ in range(self.num_linear)]\n        self.out_layer = tf.keras.layers.Dense(self.out_dim)\n        \n    def call(self, inputs):\n        X = []\n        for i in range(self.num_features):\n            X.append(self.embedding_layers[i](tf.gather(inputs, i, axis=1)))\n        X = tf.concat(X, axis=1)\n        X0 = tf.identity(X)\n\n        for i in range(self.num_cross):\n            X = X0 * self.cross_out_layers[i](self.cross_in_layers[i](X)) + X\n\n        for i in range(self.num_linear):\n            X = self.linear_layers[i](X)\n\n        X = self.out_layer(X)\n        return X\n\nclass Predictor(tf.keras.Model):\n\n    def __init__(self, out_dim):\n        super(Predictor, self).__init__()\n        self.out_layer = tf.keras.layers.Dense(out_dim)\n\n    def call(self, inputs):\n        return self.out_layer(inputs)\n\n\nclass Trainer(tf.keras.Model):\n\n    def __init__(self, emb_model, predictors):\n        super(Trainer, self).__init__()\n        self.emb_model = emb_model\n        self.predictors = predictors\n        self.eps = epsilon\n\n    def call(self, inputs):\n        unique_emb = self.emb_model(inputs[\"unique_feature\"])\n        loss_sum = 0.\n        pos_true_positive = 0.\n        pos_false_positive = 0.\n        pos_false_negative = 0.\n        neg_true_positive = 0.\n        neg_false_positive = 0.\n        neg_false_negative = 0.\n        correct = 0.\n        for i in range(3):\n            pred_emb = tf.reduce_sum(tf.gather(unique_emb, inputs[f\"history_{i}\"]), axis=1)\n            pred_val = tf.clip_by_value(tf.math.sigmoid(self.predictors[i](tf.nn.l2_normalize(pred_emb, axis=1))), self.eps, 1. - self.eps)\n            # Binary Cross Entropy\n            loss = -inputs[f\"label_{i}\"] * tf.math.log(pred_val) - (1. - inputs[f\"label_{i}\"]) * tf.math.log(1. - pred_val)\n            loss_sum += tf.reduce_sum(tf.reduce_mean(loss, axis=0))\n\n            # F1-macro\n            pred_label = pred_val > .5\n            bool_label = inputs[f\"label_{i}\"] == 1.\n            correct += tf.cast(tf.math.count_nonzero(pred_label == bool_label), \"float32\")\n            pos_true_positive += tf.cast(tf.math.count_nonzero(tf.math.logical_and(bool_label, pred_label)), \"float32\")\n            pos_false_positive += tf.cast(tf.math.count_nonzero(tf.math.logical_and(tf.math.logical_not(bool_label), pred_label)), \"float32\")\n            pos_false_negative += tf.cast(tf.math.count_nonzero(tf.math.logical_and(bool_label, tf.math.logical_not(pred_label))), \"float32\")\n\n            \n            pred_label = pred_val < .5\n            bool_label = inputs[f\"label_{i}\"] == 0.\n            neg_true_positive += tf.cast(tf.math.count_nonzero(tf.math.logical_and(bool_label, pred_label)), \"float32\")\n            neg_false_positive += tf.cast(tf.math.count_nonzero(tf.math.logical_and(tf.math.logical_not(bool_label), pred_label)), \"float32\")\n            neg_false_negative += tf.cast(tf.math.count_nonzero(tf.math.logical_and(bool_label, tf.math.logical_not(pred_label))), \"float32\")\n\n            \n        accuracy = correct / tf.cast(tf.shape(inputs[f\"label_0\"])[0] * 18, \"float32\")\n        pos_recall = pos_true_positive / tf.maximum(self.eps, pos_true_positive + pos_false_negative)\n        pos_precision = pos_true_positive / tf.maximum(self.eps, pos_true_positive + pos_false_positive)\n        pos_f1 = 2*pos_recall*pos_precision / tf.maximum(self.eps, pos_recall + pos_precision)\n        \n        neg_recall = neg_true_positive / tf.maximum(self.eps, neg_true_positive + neg_false_negative)\n        neg_precision = neg_true_positive / tf.maximum(self.eps, neg_true_positive + neg_false_positive)\n        neg_f1 = 2*neg_recall*neg_precision / tf.maximum(self.eps, neg_recall + neg_precision)\n        \n        return loss_sum, (pos_f1 + neg_f1)/2., accuracy\n\n    def predict_proba(self, inputs):\n        unique_emb = self.emb_model(inputs[\"unique_feature\"])\n        labels = []\n        pred_vals = []\n        for i in range(3):\n            pred_emb = tf.reduce_sum(tf.gather(unique_emb, inputs[f\"history_{i}\"]), axis=1)\n            pred_val = tf.clip_by_value(tf.math.sigmoid(self.predictors[i](tf.nn.l2_normalize(pred_emb, axis=1))), self.eps, 1. - self.eps)\n            pred_vals.append(pred_val)\n            labels.append(inputs[f\"label_{i}\"])\n        return tf.concat(pred_vals, axis=1), tf.concat(labels, axis=1)\n","metadata":{"ExecuteTime":{"end_time":"2023-02-19T14:25:52.879446Z","start_time":"2023-02-19T14:25:52.86024Z"},"execution":{"iopub.status.busy":"2023-03-07T02:10:51.617746Z","iopub.execute_input":"2023-03-07T02:10:51.618531Z","iopub.status.idle":"2023-03-07T02:10:51.652706Z","shell.execute_reply.started":"2023-03-07T02:10:51.618483Z","shell.execute_reply":"2023-03-07T02:10:51.651803Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# evaluate model\nfrom sklearn.model_selection import train_test_split\n\n# define model and dataset\ndata_loader = DataLoader(train_histories, train_labels, unique_feature)\nemb_model = DCNV2Model(feature_num_vocabs, feat_dim=FEAT_DIM, out_dim=OUT_DIM, num_cross=NUM_CROSS, num_linear=NUM_LINEAR)\npredictors = [Predictor(3), Predictor(10), Predictor(5)]\ntrainer = Trainer(emb_model, predictors)\n\noptimizer = tfa.optimizers.LazyAdam(learning_rate=LR)\nloss_metric = tf.keras.metrics.Mean()\nf1_metric = tf.keras.metrics.Mean()\nacc_metric = tf.keras.metrics.Mean()\n\ntrain_sessions = np.arange(num_session)\ndev_sessions, val_sessions = train_test_split(train_sessions, test_size=2000, shuffle=True)\ndev_dataset = tf.data.Dataset.from_tensor_slices(tf.convert_to_tensor(dev_sessions))\\\n                             .shuffle(num_session, reshuffle_each_iteration=True)\\\n                             .batch(BATCH)\\\n                             .map(data_loader.call)\\\n                             .prefetch(tf.data.AUTOTUNE)","metadata":{"ExecuteTime":{"end_time":"2023-02-19T14:25:53.553354Z","start_time":"2023-02-19T14:25:53.527075Z"},"execution":{"iopub.status.busy":"2023-03-07T02:10:51.654105Z","iopub.execute_input":"2023-03-07T02:10:51.654787Z","iopub.status.idle":"2023-03-07T02:10:52.774887Z","shell.execute_reply.started":"2023-03-07T02:10:51.654749Z","shell.execute_reply":"2023-03-07T02:10:52.773623Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training\n\n@tf.function(experimental_relax_shapes=True)\ndef forward_step(batch_inputs):\n    with tf.GradientTape() as tape:\n        loss, f1, acc = trainer(batch_inputs, training=True)\n    gradients = tape.gradient(loss, trainer.trainable_variables)\n    optimizer.apply_gradients(zip(gradients, trainer.trainable_variables))\n    return loss, f1, acc\n\nwith tf.device(\"CPU: 0\"):\n    for epoch in range(EPOCHS):\n        with tqdm(total=len(dev_dataset)) as pbar:\n            for batch_inputs in dev_dataset:\n                loss, f1, acc = forward_step(batch_inputs)\n                loss_metric(loss)\n                f1_metric(f1)\n                acc_metric(acc)\n                progress = {\"BCE\": loss_metric.result().numpy(), \"f1\": f1_metric.result().numpy(), \"accuracy\": acc_metric.result().numpy()}\n                pbar.set_postfix(progress)\n                pbar.update(1)\n        print(\"epoch:\", epoch+1)\n        print(\"dev:\", loss_metric.result().numpy(), f1_metric.result().numpy())\n        val_loss, val_f1, val_acc = trainer(data_loader.call(val_sessions))\n        print(\"val:\", val_loss.numpy(), val_f1.numpy(), val_acc.numpy())\n        loss_metric.reset_states()\n        f1_metric.reset_states()","metadata":{"ExecuteTime":{"end_time":"2023-02-19T14:27:10.314161Z","start_time":"2023-02-19T14:25:57.61061Z"},"execution":{"iopub.status.busy":"2023-03-07T02:10:52.777090Z","iopub.execute_input":"2023-03-07T02:10:52.778139Z","iopub.status.idle":"2023-03-07T02:30:42.434245Z","shell.execute_reply.started":"2023-03-07T02:10:52.778099Z","shell.execute_reply":"2023-03-07T02:30:42.432597Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom sklearn.metrics import f1_score\ndev_pred, dev_label = trainer.predict_proba(data_loader.call(dev_sessions))\nval_pred, val_label = trainer.predict_proba(data_loader.call(val_sessions))\n\ndev_f1_scores = []\nval_f1_scores = []\nfor th in np.linspace(0.1, 0.9, 80):\n    dev_f1_scores.append(f1_score(dev_label.numpy().reshape(-1), (dev_pred > th).numpy().reshape(-1), average=\"macro\"))\n    val_f1_scores.append(f1_score(val_label.numpy().reshape(-1), (val_pred > th).numpy().reshape(-1), average=\"macro\"))\n\nplt.plot(np.linspace(0.1, 0.9, 80), dev_f1_scores, label=\"dev\")\nplt.plot(np.linspace(0.1, 0.9, 80), val_f1_scores, label=\"val\")\nplt.ylim(0, 1)\n\nprint(max(dev_f1_scores), max(val_f1_scores))","metadata":{"execution":{"iopub.status.busy":"2023-03-07T02:30:42.436106Z","iopub.execute_input":"2023-03-07T02:30:42.436484Z","iopub.status.idle":"2023-03-07T02:30:54.535982Z","shell.execute_reply.started":"2023-03-07T02:30:42.436441Z","shell.execute_reply":"2023-03-07T02:30:54.534109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(max(dev_f1_scores), max(val_f1_scores))","metadata":{"execution":{"iopub.status.busy":"2023-03-07T02:30:54.537216Z","iopub.execute_input":"2023-03-07T02:30:54.537578Z","iopub.status.idle":"2023-03-07T02:30:54.547554Z","shell.execute_reply.started":"2023-03-07T02:30:54.537546Z","shell.execute_reply":"2023-03-07T02:30:54.546314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# use all_data for train\n# define model and dataset\ndata_loader = DataLoader(train_histories, train_labels, unique_feature)\nemb_model = DCNV2Model(feature_num_vocabs, feat_dim=FEAT_DIM, out_dim=OUT_DIM, num_cross=NUM_CROSS, num_linear=NUM_LINEAR)\npredictors = [Predictor(3), Predictor(10), Predictor(5)]\ntrainer = Trainer(emb_model, predictors)\n\noptimizer = tfa.optimizers.LazyAdam(learning_rate=LR)\nloss_metric = tf.keras.metrics.Mean()\nf1_metric = tf.keras.metrics.Mean()\nacc_metric = tf.keras.metrics.Mean()\n\ntrain_sessions = np.arange(num_session)\ntrain_dataset = tf.data.Dataset.from_tensor_slices(tf.convert_to_tensor(train_sessions))\\\n                               .shuffle(num_session, reshuffle_each_iteration=True)\\\n                               .batch(BATCH)\\\n                               .map(data_loader.call)\\\n                               .prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2023-03-07T02:30:54.549486Z","iopub.execute_input":"2023-03-07T02:30:54.550164Z","iopub.status.idle":"2023-03-07T02:30:55.562500Z","shell.execute_reply.started":"2023-03-07T02:30:54.550127Z","shell.execute_reply":"2023-03-07T02:30:55.561481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training\n\n@tf.function(experimental_relax_shapes=True)\ndef forward_step(batch_inputs):\n    with tf.GradientTape() as tape:\n        loss, f1, acc = trainer(batch_inputs, training=True)\n    gradients = tape.gradient(loss, trainer.trainable_variables)\n    optimizer.apply_gradients(zip(gradients, trainer.trainable_variables))\n    return loss, f1, acc\n\nwith tf.device(\"CPU: 0\"):\n    for epoch in range(EPOCHS):\n        with tqdm(total=len(train_dataset)) as pbar:\n            for batch_inputs in train_dataset:\n                loss, f1, acc = forward_step(batch_inputs)\n                loss_metric(loss)\n                f1_metric(f1)\n                acc_metric(acc)\n                progress = {\"BCE\": loss_metric.result().numpy(), \"f1\": f1_metric.result().numpy(), \"accuracy\": acc_metric.result().numpy()}\n                pbar.set_postfix(progress)\n                pbar.update(1)\n        print(\"epoch:\", epoch+1)\n        print(\"train:\", loss_metric.result().numpy(), f1_metric.result().numpy())\n        loss_metric.reset_states()\n        f1_metric.reset_states()","metadata":{"execution":{"iopub.status.busy":"2023-03-07T02:30:55.564846Z","iopub.execute_input":"2023-03-07T02:30:55.565481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_pred, train_label = trainer.predict_proba(data_loader.call(train_sessions))\n\ntrain_f1_scores = []\nfor th in np.linspace(0.1, 0.9, 80):\n    train_f1_scores.append(f1_score(train_label.numpy().reshape(-1), (train_pred > th).numpy().reshape(-1), average=\"macro\"))\n    \nplt.plot(np.linspace(0.1, 0.9, 80), train_f1_scores, label=\"dev\")\nplt.ylim(0, 1)\n\nthreshold = np.linspace(0.1, 0.9, 80)[np.argmax(train_f1_scores)]\nprint(threshold, max(train_f1_scores))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"embeddings = emb_model(unique_feature).numpy()\n\ndel train_histories, train_label, train_labels\ngc.collect()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder\nenv = jo_wilder.make_env()\niter_test = env.iter_test()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(unique_train_df.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# predict\neps = epsilon\nfor (sample_submission, test) in iter_test:\n    level_group = ['0-4', '5-12', '13-22'].index(test[\"level_group\"].values[0])\n    for col in feature_columns:\n        test[col + \"_ix\"] = np.vectorize(lambda x: feature_dicts[col].get(x, 0))(test[col].fillna(\"nan_value\").astype(str)).astype(np.int32)\n\n    # update embedding table\n    test_feats = test[feature_ix_columns].merge(unique_train_df,\n                                                on=feature_ix_columns,\n                                                how=\"left\")\n    unseen_test_feats = test_feats[test_feats[\"unique_ix\"].isna()].drop_duplicates(feature_ix_columns)[feature_ix_columns]\n    new_embedding = emb_model(unseen_test_feats.values).numpy()\n\n    if new_embedding.shape[0]:\n        unseen_test_feats[\"unique_ix\"] = list(range(unique_train_df.shape[0], unique_train_df.shape[0] + new_embedding.shape[0]))\n        unique_train_df = pd.concat([unique_train_df, unseen_test_feats], axis=0)\n        embeddings = np.concatenate([embeddings, new_embedding], axis=0)\n\n    # predict\n    test_feats = test[feature_ix_columns].merge(unique_train_df,\n                                                on=feature_ix_columns,\n                                                how=\"left\")\n    pred_emb = embeddings[test_feats[\"unique_ix\"].values].sum(axis=0).reshape([1, -1])\n    pred_emb /= np.maximum(eps, np.linalg.norm(pred_emb, ord=2))\n    sample_submission[\"correct\"] = tf.cast(tf.math.sigmoid(predictors[level_group](pred_emb)) > threshold, \"int32\").numpy()[0]\n\n    env.predict(sample_submission)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(unique_train_df.shape)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check prediction\n\ndf = pd.read_csv('submission.csv')\nprint( df.shape )\ndf.head(20)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(df.correct.mean())","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}