{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Version v7b\n* Similar to version v6e\n* Same aggregated features for all folds","metadata":{"papermill":{"duration":0.009347,"end_time":"2023-05-24T02:34:29.083872","exception":false,"start_time":"2023-05-24T02:34:29.074525","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"# Config","metadata":{"papermill":{"duration":0.007744,"end_time":"2023-05-24T02:34:29.099817","exception":false,"start_time":"2023-05-24T02:34:29.092073","status":"completed"},"tags":[]}},{"cell_type":"code","source":"%%writefile Config.py\nimport os\n\nclass cfg:\n    # General settings\n    comp_name = 'PSP'\n    vers = ['v5c', 'v7b']\n    env = 'kaggle'\n    seed = 314\n    use_polar = False    # If False, use pandas instead\n    device = 'cpu'\n    # Data and training\n    apex = True\n    nfolds = 5\n    training_folds = [0, 1, 2, 3, 4]\n    # Paths\n    if env == 'colab':\n        comp_data_dir = f'/content/drive/My Drive/Kaggle competitions/{comp_name}/comp_data'\n        ext_data_dir = f'/content/drive/My Drive/Kaggle competitions/{comp_name}/ext_data'\n        model_dir = f'/content/drive/My Drive/Kaggle competitions/{comp_name}/model'\n        os.makedirs(os.path.join(model_dir, ver[:-1], ver[-1]), exist_ok = True)\n    elif env == 'kaggle':\n        comp_data_dir = '/kaggle/input/predict-student-performance-from-game-play'\n        ext_data_dir = ...\n        model_dirs = [f'/kaggle/input/psp{ver}' for ver in vers]\n    elif env in ['paperspace', 'vastai']:\n        comp_data_dir = 'data'\n        ext_data_dir = 'ext_data'\n        model_dir = 'model'\n        os.makedirs(os.path.join(model_dir, ver[:-1], ver[-1]), exist_ok = True)","metadata":{"papermill":{"duration":0.028072,"end_time":"2023-05-24T02:34:29.135997","exception":false,"start_time":"2023-05-24T02:34:29.107925","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-13T10:33:10.797514Z","iopub.execute_input":"2023-06-13T10:33:10.798349Z","iopub.status.idle":"2023-06-13T10:33:10.825903Z","shell.execute_reply.started":"2023-06-13T10:33:10.798308Z","shell.execute_reply":"2023-06-13T10:33:10.824974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Package","metadata":{"papermill":{"duration":0.007926,"end_time":"2023-05-24T02:34:29.152333","exception":false,"start_time":"2023-05-24T02:34:29.144407","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os, pickle, gc\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport torch\nfrom torch.cuda.amp import autocast\n\nimport xgboost as xgb\nimport lightgbm as lgbm\n\nfrom Config import cfg\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport sys\nsys.path.insert(1, f'/kaggle/input/{cfg.comp_name.lower()}{cfg.vers[-1]}')\n\nfrom custom_config import set_random_seed\nfrom feature_engineering import process_data, add_columns_pl, feature_engineering_pl\nfrom utils import build_log, print_log\nfrom model import DEFAULT_LGBM_PARAMS, DEFAULT_XGB_PARAMS, BoostingModel\n\nDEFAULT_XGB_PARAMS['tree_method'] = 'hist'\nDEFAULT_XGB_PARAMS['gpu_id'] = -1\n\nQUESTION2LEVEL = {}    # This dictionary contains pairs of {k: v}, which means, to predict question k, we need data from group v\n    \nfor i in range(1, 19):\n    # From https://www.kaggle.com/code/cdeotte/xgboost-baseline-0-676\n    # There are 18 questions from 1 to 18\n    if i <= 3:\n        QUESTION2LEVEL[f'q{i}'] = '0-4'\n    elif i <= 13:\n        QUESTION2LEVEL[f'q{i}'] = '5-12'\n    else:\n        QUESTION2LEVEL[f'q{i}'] = '13-22'\n        \nLEVEL2QUESTION = {\n    '0-4': ['q1', 'q2', 'q3'],\n    '5-12': ['q4', 'q5', 'q6', 'q7', 'q8', 'q9', 'q10', 'q11', 'q12', 'q13'],\n    '13-22': ['q14', 'q15', 'q16', 'q17', 'q18'],\n}\n\nLEVEL_MAP = {\n    '0-4': 0,\n    '5-12': 1,\n    '13-22': 2,\n}\n\nNUM_COLS = ['index', 'time_diff', 'room_coor_x', 'room_coor_y', 'screen_coor_x', 'screen_coor_y', 'hover_duration']\nTXT_COLS = ['level', 'event_name', 'name', 'text', 'fqid', 'room_fqid', 'text_fqid']\n\nset_random_seed(cfg.seed)","metadata":{"papermill":{"duration":6.314771,"end_time":"2023-05-24T02:34:35.475124","exception":false,"start_time":"2023-05-24T02:34:29.160353","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-13T10:33:10.852032Z","iopub.execute_input":"2023-06-13T10:33:10.852808Z","iopub.status.idle":"2023-06-13T10:33:16.452316Z","shell.execute_reply.started":"2023-06-13T10:33:10.852768Z","shell.execute_reply":"2023-06-13T10:33:16.451011Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load the feature lists","metadata":{"papermill":{"duration":0.008208,"end_time":"2023-05-24T02:34:35.492176","exception":false,"start_time":"2023-05-24T02:34:35.483968","status":"completed"},"tags":[]}},{"cell_type":"code","source":"feature_lists = {}\nfeature_cols = {}\nfor level in ['0-4', '5-12', '13-22']:\n    with open(os.path.join(cfg.model_dirs[1], 'best_model', f\"feature_list_level_{level.replace('-', '_')}.pkl\"), 'rb') as f:\n        feature_list = pickle.load(f)\n    feature_lists[f\"level_{level.replace('-', '_')}\"] = [c for c in feature_list if c not in ['session_id']]\n\nfor fold in cfg.training_folds:\n    for level in ['0-4', '5-12', '13-22']:\n        with open(os.path.join(cfg.model_dirs[1], 'best_model', f\"feature_cols_level_{level.replace('-', '_')}_fold_{fold}.pkl\"), 'rb') as f:\n            feature_col = pickle.load(f)\n        feature_cols[f\"level_{level.replace('-', '_')}_fold_{fold}\"] = [c for c in feature_col if c not in ['session_id']]","metadata":{"papermill":{"duration":0.145509,"end_time":"2023-05-24T02:34:35.646102","exception":false,"start_time":"2023-05-24T02:34:35.500593","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-13T10:33:16.457894Z","iopub.execute_input":"2023-06-13T10:33:16.458284Z","iopub.status.idle":"2023-06-13T10:33:16.621261Z","shell.execute_reply.started":"2023-06-13T10:33:16.458248Z","shell.execute_reply":"2023-06-13T10:33:16.620238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# For the Neural-network Models","metadata":{"papermill":{"duration":0.008149,"end_time":"2023-05-24T02:34:35.663364","exception":false,"start_time":"2023-05-24T02:34:35.655215","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### Model architecture","metadata":{"papermill":{"duration":0.007933,"end_time":"2023-05-24T02:34:35.679597","exception":false,"start_time":"2023-05-24T02:34:35.671664","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os, pickle, copy, math\nimport torch\nfrom torch import nn\nimport torch.nn.functional as F\n\n# Custom transformer-encoder layer with attention weights\ndef clones(module, N):\n    \"Produce N identical layers.\"\n    return nn.ModuleList([copy.deepcopy(module) for _ in range(N)])\n\ndef attention(query, key, value, attn_weight = None, mask = None, dropout = None):\n    \"Compute 'Scaled Dot Product Attention'\"\n    batch_size, seq_len, d_k = query.shape[0], query.shape[2], query.shape[-1]\n    scores = torch.matmul(query, key.transpose(-2, -1)) \\\n             / math.sqrt(d_k)\n    if mask is not None:\n        mask = mask.view(batch_size, 1, 1, seq_len)\n        scores = scores.masked_fill(mask, float('-inf'))    # Masking at mask == True\n    if attn_weight is not None:\n        # Check the dimmension\n        assert len(attn_weight.shape) == 4, \"'attn_weight' should have 4 dimensions, (batch_size, num_head, T, T).\"\n        scores -= attn_weight\n    p_attn = F.softmax(scores, dim = -1)\n    if dropout is not None:\n        p_attn = dropout(p_attn)\n    return torch.matmul(p_attn, value), p_attn\n\nclass WeightedMultiHeadedAttention(nn.Module):\n    def __init__(self, d_model, nhead, dropout = 0.1):\n        \"Take in model size and number of heads.\"\n        super(WeightedMultiHeadedAttention, self).__init__()\n        assert d_model % nhead == 0\n        # We assume d_v always equals d_k\n        self.d_k = d_model // nhead\n        self.nhead = nhead\n        self.linears = clones(nn.Linear(d_model, d_model, bias = False), 4) # Q, K, V, last\n        self.attn = None\n        self.dropout = nn.Dropout(p = dropout)\n\n    def forward(self, query, key, value, attn_weight = None, mask = None):\n        \"Implements Figure 2\"\n        if mask is not None:\n            # Same mask applied to all h heads.\n            mask = mask.unsqueeze(1)\n        nbatches = query.size(0)\n\n        # 1) Do all the linear projections in batch from d_model => h x d_k\n        query, key, value = \\\n            [l(x).view(nbatches, -1, self.nhead, self.d_k).transpose(1, 2)\n             for l, x in zip(self.linears, (query, key, value))]\n        \n        # 2) Apply attention on all the projected vectors in batch.\n        x, self.attn = attention(query, key, value, attn_weight = attn_weight, mask = mask, dropout = self.dropout)\n\n        # 3) \"Concat\" using a view and apply a final linear.\n        x = x.transpose(1, 2).contiguous() \\\n            .view(nbatches, -1, self.nhead * self.d_k)\n        return self.linears[-1](x)\n\nclass PositionwiseFeedForward(nn.Module):\n    \"Implements FFN equation.\"\n    def __init__(self, d_model, d_ff, dropout = 0.1):\n        super(PositionwiseFeedForward, self).__init__()\n        self.w_1 = nn.Linear(d_model, d_ff)\n        self.w_2 = nn.Linear(d_ff, d_model)\n        self.dropout = nn.Dropout(dropout)\n\n    def forward(self, x):\n        return self.w_2(self.dropout(F.relu(self.w_1(x))))\n    \nclass TransformerEncoderLayer(nn.Module):\n    \"\"\"\n    Single Encoder Block\n    \"\"\"\n    def __init__(self, d_model, nhead, dim_feedforward = 1024, dropout = 0.1, batch_first = False):\n        super(TransformerEncoderLayer, self).__init__()\n        self._self_attn = WeightedMultiHeadedAttention(d_model, nhead, dropout)\n        self._ffn = PositionwiseFeedForward(d_model, dim_feedforward, dropout)\n        self._layernorms = clones(nn.LayerNorm(d_model, eps = 1e-6), 2)\n        self._dropout = nn.Dropout(dropout)\n\n    def forward(self, src, attn_weight = None, src_key_padding_mask = None):\n        \"\"\"\n        query: question embeddings\n        key: interaction embeddings\n        \"\"\"\n        # self-attention block\n        src2 = self._self_attn(query = src, key = src, value = src, attn_weight = attn_weight, mask = src_key_padding_mask)\n        src = src + self._dropout(src2)\n        src = self._layernorms[0](src)\n        src2 = self._ffn(src)\n        src = src + self._dropout(src2)\n        src = self._layernorms[1](src)\n        return src\n    \nclass TransformerEncoder(nn.Module):\n    \"\"\"Stack of N single transformer-encoder blocks\"\"\"\n    def __init__(self, encoder_layer, num_layers):\n        super(TransformerEncoder, self).__init__()\n        self.encoder_layers = clones(encoder_layer, num_layers)\n        \n    def forward(self, src, attn_weight = None, src_key_padding_mask = None):\n        for layer in self.encoder_layers:\n            src = layer(src, attn_weight = attn_weight, src_key_padding_mask = src_key_padding_mask)\n        return src\n        \n# Pooling\nclass MeanPooling(nn.Module):\n    def __init__(self):\n        super().__init__()\n        \n    def forward(self, x, mask = None):\n        if mask is None:\n            return torch.mean(x, dim = 1)    # Mean pooling over the time\n        else:\n            return (x * (~mask).unsqueeze(-1)).sum(dim = 1) / (~mask).unsqueeze(-1).sum(dim = 1)\n\n# Main model\nclass PSPModel(nn.Module):\n    def __init__(self, cfg, TXT_COL_MAPS, level = '0-4', hidden_size = 128):\n        super().__init__()\n        self.cfg = cfg\n        self.level = level\n        # Categorical variables embeddings\n        self.txt_embeddings = {}\n        TXT_DIM = 0\n        for col, maps in TXT_COL_MAPS.items():\n            self.txt_embeddings[col] = nn.Embedding(num_embeddings = len(maps), \n                                                    embedding_dim = 8)\n            TXT_DIM += 8\n        self.txt_embeddings = nn.ModuleDict(self.txt_embeddings)\n        \n        # Numerical normalization\n        self.num_batch_norm = nn.BatchNorm1d(len(NUM_COLS) - 1)\n        \n        # The first layer\n        self.gru = nn.GRU(input_size = len(NUM_COLS) + TXT_DIM - 1,\n                          hidden_size = hidden_size // 2,\n                          num_layers = 3,\n                          batch_first = True,\n                          dropout = 0.2,\n                          bidirectional = True)\n        \n        # Transformer-encoder\n        self.attn_weight = nn.Parameter(torch.randn(hidden_size // 32).view(hidden_size // 32, 1, 1).pow(2).to(cfg.device))\n        \n        encoder = TransformerEncoderLayer(d_model = hidden_size, \n                                          nhead = hidden_size // 32, \n                                          dim_feedforward = hidden_size, \n                                          batch_first = True)\n        self.encoder = TransformerEncoder(encoder, num_layers = 3)\n               \n        \n        # Pooling and output\n        self.pooler = MeanPooling()\n        self.classifier = nn.Sequential(\n            nn.Linear(hidden_size, hidden_size * 4),\n            nn.ReLU(),\n            nn.Linear(hidden_size * 4, hidden_size * 4),\n            nn.ReLU(),\n        )\n        \n        self.output = nn.Linear(hidden_size * 4, len(LEVEL2QUESTION[level]))\n        self.aux_output = nn.Linear(hidden_size * 4, 18 - len(LEVEL2QUESTION[level]))\n        if level != '13-22':\n            self.pred_answering_time = nn.Linear(hidden_size * 4, 1)\n        \n    def _get_time_attn_weight(self, elapsed_time):\n        time_diff = elapsed_time.unsqueeze(-1) - elapsed_time.unsqueeze(1)\n        time_diff = torch.clip(time_diff, min = 1e-6, max = 3.6e6)    # Clip to 1 hour between actions\n        time_diff /= 60 * 1e3\n        time_diff = torch.log(time_diff)\n        return time_diff.unsqueeze(1)    # Shape: (batch_size, 1, seq_len, seq_len)\n        \n    def loss_fn(self, pred, true, aux_pred = None, aux_true = None, pred_answering_time = None, true_answering_time = None):\n        if pred_answering_time is not None:\n            return nn.BCEWithLogitsLoss()(pred, true) + nn.BCEWithLogitsLoss()(aux_pred, aux_true) + nn.MSELoss()(pred_answering_time, true_answering_time)\n        return nn.BCEWithLogitsLoss()(pred, true) + nn.BCEWithLogitsLoss()(aux_pred, aux_true)\n        \n    def forward(self, inputs, mask = None, label = None, aux_label = None):\n        # Embed the features\n        num_features = torch.cat([inputs[col].unsqueeze(-1) for col in NUM_COLS if col != 'index'], dim = -1)\n        num_features = self.num_batch_norm(num_features.permute(0, 2, 1)).permute(0, 2, 1)\n        txt_features = torch.cat([self.txt_embeddings[col](inputs[col]) for col in TXT_COLS], dim = -1)\n        \n        features = torch.cat([num_features, txt_features], dim = -1)\n        \n        # Trimming the sequences\n        if mask is not None:\n            local_len = (~mask).sum(axis = 1).max().item()\n            features = features[:,:local_len]\n            time_diff = inputs['time_diff'][:,:local_len]\n            mask = mask[:,:local_len]\n            lengths = (~mask).sum(axis = 1).cpu()\n        else:\n            time_diff = inputs['time_diff']\n        \n        if mask is not None:\n            features = nn.utils.rnn.pack_padded_sequence(features, lengths = lengths, batch_first = True, enforce_sorted = False)\n            features, _ = self.gru(features)\n            features, _ = nn.utils.rnn.pad_packed_sequence(features, batch_first = True, padding_value = -1.)\n        else:\n            features, _ = self.gru(features)\n        \n        # Time attention weight\n        time_attn_weight = self._get_time_attn_weight(time_diff) * self.attn_weight\n        \n        # Transformer-encoder\n        features = self.encoder(features, attn_weight = time_attn_weight, src_key_padding_mask = mask)\n        \n        # Pooling and output\n        features = self.pooler(features, mask = mask)\n        pooled_features = features.contiguous()\n        features = self.classifier(features)\n        output = self.output(features)\n        \n        aux_output = self.aux_output(features)\n        \n        if self.level != '13-22':\n            pred_answering_time = self.pred_answering_time(features).unsqueeze(-1)\n        else:\n            pred_answering_time = None\n        \n        if label is not None:\n            loss = self.loss_fn(output, label, aux_output, aux_label, pred_answering_time, inputs['answering_time'])\n        else:\n            loss = None\n        return loss, output, pooled_features","metadata":{"papermill":{"duration":0.081609,"end_time":"2023-05-24T02:34:35.769405","exception":false,"start_time":"2023-05-24T02:34:35.687796","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-13T10:33:16.622586Z","iopub.execute_input":"2023-06-13T10:33:16.623146Z","iopub.status.idle":"2023-06-13T10:33:16.691227Z","shell.execute_reply.started":"2023-06-13T10:33:16.623111Z","shell.execute_reply":"2023-06-13T10:33:16.689966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Load the trained models","metadata":{"papermill":{"duration":0.008399,"end_time":"2023-05-24T02:34:35.786327","exception":false,"start_time":"2023-05-24T02:34:35.777928","status":"completed"},"tags":[]}},{"cell_type":"code","source":"with open(os.path.join(cfg.model_dirs[0], 'best_model', 'TXT_COL_MAPS.pkl'), 'rb') as f:\n    TXT_COL_MAPS = pickle.load(f)\n\nNN_models = {}\nfor fold in cfg.training_folds:\n    for level in ['0-4', '5-12', '13-22']:\n        ckp = os.path.join(cfg.model_dirs[0], 'best_model', f\"nn_level_{level.replace('-', '_')}_fold_{fold}.pt\")\n        model = PSPModel(cfg, TXT_COL_MAPS, level = level).to(cfg.device)\n        \n        print_log(f'Loading the model weights from {ckp}')\n        model.load_state_dict(torch.load(ckp, map_location = cfg.device))\n        NN_models[f\"level_{level.replace('-', '_')}_fold_{fold}\"] = model","metadata":{"papermill":{"duration":1.958282,"end_time":"2023-05-24T02:34:37.754082","exception":false,"start_time":"2023-05-24T02:34:35.795800","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-13T10:33:16.693762Z","iopub.execute_input":"2023-06-13T10:33:16.694689Z","iopub.status.idle":"2023-06-13T10:33:18.011875Z","shell.execute_reply.started":"2023-06-13T10:33:16.694650Z","shell.execute_reply":"2023-06-13T10:33:18.010814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Inferring functions","metadata":{"papermill":{"duration":0.00861,"end_time":"2023-05-24T02:34:37.771842","exception":false,"start_time":"2023-05-24T02:34:37.763232","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def _convert_dataframe_to_dict(cfg, data, level = '0-4'):\n    '''\n    This function converts a dataframe to a dictionary. Currently, only convert numerical columns (NUM_COLS)\n    '''\n    data_dict = {}\n    data = data.sort_index('index')\n    \n    for col in NUM_COLS + TXT_COLS:\n        seq = data[col].tolist()\n        if level == '0-4':\n            max_len = 512\n        elif level == '5-12':\n            max_len = 1024\n        else:\n            max_len = 1536\n            \n        # Trimming and padding\n        mask = [0] * max_len\n        seq = seq[-max_len:]\n            \n        if col in NUM_COLS:\n            data_dict[col] = torch.tensor(seq, dtype = torch.float).to(cfg.device).unsqueeze(0)\n        else:\n            data_dict[col] = torch.tensor(seq, dtype = torch.long).to(cfg.device).unsqueeze(0)\n        mask = torch.tensor(mask, dtype = torch.bool).unsqueeze(0)\n    return data_dict, mask\n\ndef nn_infer_fn(cfg, model, data, level = '0-4'):\n    model.eval()\n    data_dict, mask = _convert_dataframe_to_dict(cfg, data, level = level)\n    with torch.no_grad():\n        with autocast(enabled = cfg.apex):\n            _, output, embeddings = model(data_dict)\n        output = output.sigmoid().detach().cpu().numpy().flatten()\n        embeddings = embeddings.detach().cpu().numpy()\n    return output, embeddings","metadata":{"papermill":{"duration":0.029109,"end_time":"2023-05-24T02:34:37.809797","exception":false,"start_time":"2023-05-24T02:34:37.780688","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-13T10:33:18.013258Z","iopub.execute_input":"2023-06-13T10:33:18.013627Z","iopub.status.idle":"2023-06-13T10:33:18.030415Z","shell.execute_reply.started":"2023-06-13T10:33:18.013591Z","shell.execute_reply":"2023-06-13T10:33:18.029110Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# For the Boosting-tree Models","metadata":{"papermill":{"duration":0.008668,"end_time":"2023-05-24T02:34:37.827723","exception":false,"start_time":"2023-05-24T02:34:37.819055","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"### Load the trained models","metadata":{"papermill":{"duration":0.008531,"end_time":"2023-05-24T02:34:37.845316","exception":false,"start_time":"2023-05-24T02:34:37.836785","status":"completed"},"tags":[]}},{"cell_type":"code","source":"tree_models = {}\nfor fold in cfg.training_folds:\n    for level in ['0-4', '5-12', '13-22']:\n        _feature_cols = [c for c in feature_cols[f\"level_{level.replace('-', '_')}_fold_{fold}\"] if c not in ['session_id']]\n        with open(os.path.join(cfg.model_dirs[1], 'best_model', f\"xgb_level_{level.replace('-', '_')}_fold_{fold}.pkl\"), 'rb') as f:\n            model = pickle.load(f)\n        tree_models[f\"level_{level.replace('-', '_')}_fold_{fold}\"] = model","metadata":{"papermill":{"duration":0.910441,"end_time":"2023-05-24T02:34:38.764907","exception":false,"start_time":"2023-05-24T02:34:37.854466","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-13T10:33:18.031877Z","iopub.execute_input":"2023-06-13T10:33:18.032515Z","iopub.status.idle":"2023-06-13T10:33:18.977853Z","shell.execute_reply.started":"2023-06-13T10:33:18.032474Z","shell.execute_reply":"2023-06-13T10:33:18.976972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Inferring functions","metadata":{"papermill":{"duration":0.008924,"end_time":"2023-05-24T02:34:38.783319","exception":false,"start_time":"2023-05-24T02:34:38.774395","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def tree_infer_fn(cfg, model, data):\n    prediction = model.predict(xgb.DMatrix(data))\n    return prediction.flatten()","metadata":{"papermill":{"duration":0.02057,"end_time":"2023-05-24T02:34:38.813210","exception":false,"start_time":"2023-05-24T02:34:38.792640","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-13T10:33:18.982180Z","iopub.execute_input":"2023-06-13T10:33:18.984437Z","iopub.status.idle":"2023-06-13T10:33:18.990215Z","shell.execute_reply.started":"2023-06-13T10:33:18.984376Z","shell.execute_reply":"2023-06-13T10:33:18.989320Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Utils","metadata":{"papermill":{"duration":0.009104,"end_time":"2023-05-24T02:34:38.831907","exception":false,"start_time":"2023-05-24T02:34:38.822803","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def process_data(df, only_drop = False):\n    df = df.drop(['fullscreen', 'hq', 'music'], axis = 1)\n    df['page'] = df['page'].astype('float32')\n        \n    if not only_drop:\n        # Fill NaN\n        for col in df.columns:\n            if df[col].dtype == float:\n                df[col] = df[col].fillna(-999.)\n            elif df[col].dtype == object:\n                df[col] = df[col].fillna(f'no {col}')\n            else:   # Maybe integers\n                df[col] = df[col].fillna(-999)\n    return df\n\ndef drop_reset_sections(df: pd.DataFrame, local: bool = True) -> pd.DataFrame:\n    '''\n    There are some sections that the users reset their game play.\n    Particularly, they start the game as usual, at level 0-4, then 5-12, then 13-22, but then they continue with 0-4 again.\n    This function will drop the repeated parts. \n    Credit to https://www.kaggle.com/code/abaojiang/lb-0-694-tconv-with-4-features-training-part\n    \n    Parameters:\n        df: pd.DataFrame\n    \n    Return:\n        df: pd.DataFrame with events occurring at the first game play only\n    '''\n    if local:\n        df['lv_diff'] = df.groupby('session_id').apply(lambda x: x['encoded_level'].diff().fillna(0)).values\n    else:\n        df['lv_diff'] = df['encoded_level'].diff().fillna(0)\n    df.loc[df['lv_diff'] >= 0, 'lv_diff'] = 0\n    if local:\n        df['multi_game_flag'] = df.groupby('session_id')['lv_diff'].cumsum()\n    else:\n        df['multi_game_flag'] = df['lv_diff'].cumsum()\n    multi_game_rows = df[df['multi_game_flag'] < 0].index\n    print_log(f\"Dropping {len(multi_game_rows)} observations, which is corresponding to {df.loc[multi_game_rows, 'session_id'].nunique()}/{df.session_id.nunique()} sections\")\n    df = df.drop(multi_game_rows).reset_index(drop = True)\n    df.drop(['lv_diff', 'multi_game_flag'], axis = 1, inplace = True)\n    return df","metadata":{"papermill":{"duration":0.035148,"end_time":"2023-05-24T02:34:38.876566","exception":false,"start_time":"2023-05-24T02:34:38.841418","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-13T10:33:18.993335Z","iopub.execute_input":"2023-06-13T10:33:18.994138Z","iopub.status.idle":"2023-06-13T10:33:19.011414Z","shell.execute_reply.started":"2023-06-13T10:33:18.994101Z","shell.execute_reply":"2023-06-13T10:33:19.010325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{"papermill":{"duration":0.009238,"end_time":"2023-05-24T02:34:38.895335","exception":false,"start_time":"2023-05-24T02:34:38.886097","status":"completed"},"tags":[]}},{"cell_type":"code","source":"from jo_wilder import make_env\n# make_env.__called__ = False\nenv = make_env()\n\nCACHE_DATA_NN = {}\nCACHE_DATA_TREE = {}\n\nfor _test, sample_submission in env.iter_test():\n    _test = _test.sort_values('index')\n    sample_submission['questions'] = sample_submission['session_id'].str.split('_q').apply(lambda x: int(x[-1]))\n    sample_submission = sample_submission.sort_values('questions')\n    sample_submission = sample_submission[['session_id', 'correct']]\n    \n    session_id = _test.session_id.values[0]\n    level = _test.level_group.values[0]\n    \n    # Cache the data\n    if session_id in CACHE_DATA_NN.keys():\n        current_test = pd.concat([CACHE_DATA_NN[session_id], _test]).reset_index(drop = True)\n    else:\n        current_test = _test\n    \n    CACHE_DATA_NN[session_id] = current_test\n    \n    # Process the data for or the neural network models\n    nn_test = process_data(current_test, only_drop = False)\n    \n    for col in TXT_COL_MAPS:\n        nn_test[col] = nn_test[col].map(TXT_COL_MAPS[col])\n        \n    nn_test['encoded_level'] = nn_test['level_group'].map(LEVEL_MAP)\n    nn_test = drop_reset_sections(nn_test, local = False)\n    \n    nn_test['time_diff'] = nn_test['elapsed_time'].diff().fillna(0).clip(lower = 0).values\n    nn_test.fillna(0, inplace = True)\n    \n    # Process the data for the tree models\n    tree_test = process_data(_test, only_drop = True)\n    tree_test = pl.from_pandas(tree_test).with_columns(add_columns_pl(tree_test))\n    \n    # Feature engineering\n    agg_test, _ = feature_engineering_pl(tree_test, group = level, use_extra_features = True, feature_suffix = '', \n                                         remaining_features = feature_lists[f\"level_{level.replace('-', '_')}\"])\n    agg_test = agg_test.set_index('session_id')\n    agg_test = agg_test[feature_lists[f\"level_{level.replace('-', '_')}\"]].values\n    \n    fold_data = {}\n    \n    for i, fold in enumerate(cfg.training_folds):\n        # Prediction - Neural network models\n        nn_prediction, nn_embeddings = nn_infer_fn(cfg, NN_models[f\"level_{level.replace('-', '_')}_fold_{fold}\"], \n                                                   nn_test, level = level)\n        _agg_test = agg_test\n        if level == '5-12':\n            past_levels = ['0-4']\n            # Load the features of the previous levels\n            for past_level in past_levels:\n                _agg_test = np.hstack((_agg_test, CACHE_DATA_TREE[session_id][past_level][fold]))\n        elif level == '13-22':\n            past_levels = ['0-4', '5-12']\n            # Load the features of the previous levels\n            for past_level in past_levels:\n                _agg_test = np.hstack((_agg_test, CACHE_DATA_TREE[session_id][past_level][fold]))\n        \n        # Add the neural network embeddings\n        _agg_test = np.hstack((_agg_test, nn_embeddings))\n        \n        tree_prediction = tree_infer_fn(cfg, tree_models[f\"level_{level.replace('-', '_')}_fold_{fold}\"], \n                                        pd.DataFrame(_agg_test, columns = feature_cols[f\"level_{level.replace('-', '_')}_fold_{fold}\"]))\n        \n        _agg_test = np.hstack((_agg_test, tree_prediction.reshape(1, -1)))\n        \n        fold_data[fold] = _agg_test\n        \n        if level != '13-22':\n            if session_id not in CACHE_DATA_TREE:\n                sub_data_dict = {}\n                sub_data_dict[level] = fold_data\n                CACHE_DATA_TREE[session_id] = sub_data_dict\n            else:\n                CACHE_DATA_TREE[session_id][level] = fold_data\n                \n        if i == 0:\n            final_prediction = tree_prediction / len(cfg.training_folds)\n        else:\n            final_prediction += tree_prediction / len(cfg.training_folds)\n            \n    sample_submission['correct'] = (final_prediction > 0.63).astype(int)\n    \n    env.predict(sample_submission)","metadata":{"papermill":{"duration":14.522035,"end_time":"2023-05-24T02:34:53.427653","exception":false,"start_time":"2023-05-24T02:34:38.905618","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-13T10:33:19.012988Z","iopub.execute_input":"2023-06-13T10:33:19.013306Z","iopub.status.idle":"2023-06-13T10:33:33.329527Z","shell.execute_reply.started":"2023-06-13T10:33:19.013276Z","shell.execute_reply":"2023-06-13T10:33:33.328220Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"display(pd.read_csv('/kaggle/working/submission.csv'))","metadata":{"papermill":{"duration":0.041295,"end_time":"2023-05-24T02:34:53.479496","exception":false,"start_time":"2023-05-24T02:34:53.438201","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2023-06-13T10:33:33.330878Z","iopub.execute_input":"2023-06-13T10:33:33.331255Z","iopub.status.idle":"2023-06-13T10:33:33.350158Z","shell.execute_reply.started":"2023-06-13T10:33:33.331218Z","shell.execute_reply":"2023-06-13T10:33:33.349200Z"},"trusted":true},"execution_count":null,"outputs":[]}]}