{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install /kaggle/input/xgb162/xgboost-1.6.2-py3-none-manylinux2014_x86_64.whl\n!pip install /kaggle/input/polars0165/polars-0.16.5-cp37-abi3-manylinux_2_17_x86_64.manylinux2014_x86_64.whl","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:50:49.040841Z","iopub.execute_input":"2023-06-28T10:50:49.041567Z","iopub.status.idle":"2023-06-28T10:52:00.217554Z","shell.execute_reply.started":"2023-06-28T10:50:49.041525Z","shell.execute_reply":"2023-06-28T10:52:00.215170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport sys\n\nimport numpy as np\nimport pandas as pd\nimport polars as pl\nimport pickle\nimport xgboost as xgb\nfrom catboost import CatBoostClassifier, Pool","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-06-28T10:52:00.220242Z","iopub.execute_input":"2023-06-28T10:52:00.220710Z","iopub.status.idle":"2023-06-28T10:52:01.179303Z","shell.execute_reply.started":"2023-06-28T10:52:00.220671Z","shell.execute_reply":"2023-06-28T10:52:01.177814Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc\nimport os\nfrom typing import Any, Dict, List, Optional, Tuple\nimport pickle\nimport warnings\nwarnings.simplefilter(\"ignore\")\n\nimport numpy as np\nimport pandas as pd\nimport yaml\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch import Tensor","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:01.181353Z","iopub.execute_input":"2023-06-28T10:52:01.181826Z","iopub.status.idle":"2023-06-28T10:52:03.237006Z","shell.execute_reply.started":"2023-06-28T10:52:01.181786Z","shell.execute_reply":"2023-06-28T10:52:03.235385Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"exp_dump_path = \"/kaggle/input/public-nn-v32\"\ncat2code_path = \"/kaggle/input/public-nn-v32\"\nmodel_path1 = \"/kaggle/input/public-nn-v32\"\nmodel_path2 = \"/kaggle/input/public-nn-v47\"\nmodel_path3 = \"/kaggle/input/public-nn-v51\"","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:03.240925Z","iopub.execute_input":"2023-06-28T10:52:03.241851Z","iopub.status.idle":"2023-06-28T10:52:03.247367Z","shell.execute_reply.started":"2023-06-28T10:52:03.241803Z","shell.execute_reply":"2023-06-28T10:52:03.246268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CAT_FEATS = [\"event_comb_code\", \"room_fqid_code\",\n             \"page_code\", \"text_fqid_code\",\n             \"level_code\"]\nCAT_FEAT_SIZE = {\n    \"event_comb_code\": 19,\n    \"room_fqid_code\": 19,\n    \"page_code\": 8,\n    \"text_fqid_code\": 127,\n    \"level_code\": 23,\n}\nDEVICVE = torch.device(\"cpu\")\n\n# Load model configuration\n\nmodel_cfg = {\"cat_feats\": [\"event_comb_code\", \"room_fqid_code\",\n                           \"page_code\", \"text_fqid_code\",\n                           \"level_code\"]}\nt_window = 512","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:03.248995Z","iopub.execute_input":"2023-06-28T10:52:03.249638Z","iopub.status.idle":"2023-06-28T10:52:03.268242Z","shell.execute_reply.started":"2023-06-28T10:52:03.249603Z","shell.execute_reply":"2023-06-28T10:52:03.267148Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Model_v32(out_dim, **model_cfg):\n\n    class GEGLU(nn.Module):\n        \"\"\"\n        References:\n            Shazeer et al., \"GLU Variants Improve Transformer,\" 2020.\n            https://arxiv.org/abs/2002.05202\n        \"\"\"\n        def __init__(self, dim=1):\n            super(GEGLU, self).__init__()\n            self.dim=dim\n\n        def geglu(self, x: Tensor) -> Tensor:\n            assert x.shape[self.dim] % 2 == 0\n            a, b = x.chunk(2, dim=self.dim)\n            return a * F.gelu(b)\n\n        def forward(self, x: Tensor) -> Tensor:\n            return self.geglu(x)\n\n\n    class Conformer(nn.Module):\n        \"\"\"Dilated temporal convolution layer.\n\n        Considering the time cost, I currently disable dilation.\n        \"\"\"\n\n        def __init__(\n            self,\n            d_model: int,\n            nheads: int,\n            dropout: float = 0.1,\n            kernel_size=5\n        ):\n            super(Conformer, self).__init__()\n\n            self.norm1 = nn.LayerNorm(d_model)\n            self.conv1 = nn.Sequential(nn.Conv1d(d_model,2*d_model,kernel_size=kernel_size,padding='same'),\n                                      GEGLU(),\n                                      nn.Dropout(dropout),\n                                      nn.Conv1d(d_model,d_model,kernel_size=kernel_size,padding='same'),\n                                      nn.Dropout(dropout))\n\n            self.norm2 = nn.LayerNorm(d_model)\n            self.attn = nn.MultiheadAttention(d_model,nheads,dropout=dropout,batch_first=True)\n\n            self.norm3 = nn.LayerNorm(d_model)\n            self.conv2 = nn.Sequential(nn.Conv1d(d_model,2*d_model,kernel_size=kernel_size,padding='same'),\n                                      GEGLU(),\n                                      nn.Dropout(dropout),\n                                      nn.Conv1d(d_model,d_model,kernel_size=kernel_size,padding='same'),\n                                      nn.Dropout(dropout))\n\n        def forward(self, x: Tensor, mask: Tensor) -> Tensor:\n            \"\"\"Forward pass.\n\n            Shape:\n                x: (B, L, D)\n                mask: (B, L)\n            \"\"\"\n            x2 = self.norm1(x)\n            x2 = x2.permute(0,2,1)\n            x2 = self.conv1(x2)\n            x2 = x2.permute(0,2,1)\n            x = x + 0.5 * x2\n\n            x2 = self.norm2(x)\n            x2 = self.attn(x2[:,[-1]],x2,x2,key_padding_mask=mask)[0]\n            x = x + x2\n\n            x2 = self.norm3(x)\n            x2 = x2.permute(0,2,1)\n            x2 = self.conv2(x2)\n            x2 = x2.permute(0,2,1)\n            x = x + 0.5 * x2\n            return x\n\n\n    class EventAwareEncoder(nn.Module):\n        \"\"\"Event-aware encoder based on 1D-Conv.\"\"\"\n\n        def __init__(\n            self,\n            d_model: int = 32,\n            out_dim: int = 128,\n            readout: bool = True,\n            cat_feats: List[str] = [\"event_comb_code\", \"room_fqid_code\"]\n        ):\n            super(EventAwareEncoder, self).__init__()\n\n            # Network parameters\n            self.d_model = d_model\n            self.out_dim = out_dim\n            self.cat_feats = cat_feats\n            d_hidden = 128\n            nheads = 2\n            dropout = 0.2\n\n            # Model blocks\n            # Categorical embeddings\n            self.embs = nn.ModuleList()\n            for cat_feat in cat_feats:\n                self.embs.append(nn.Embedding(CAT_FEAT_SIZE[cat_feat] + 1, d_model, padding_idx=0))\n            self.emb_enc = nn.Sequential(\n                nn.Linear(d_model*len(cat_feats), 2*d_model*len(cat_feats)),\n                nn.GELU(),\n                nn.Linear(2*d_model*len(cat_feats), d_model*len(cat_feats)),\n                nn.GELU(),\n            )\n            self.dropout = nn.Dropout(0.2)\n            self.dense_layer = nn.Sequential(nn.Linear(1,2*d_model*len(cat_feats)), nn.GLU())\n\n            # Feature extractor        \n            self.combine_layer = nn.Linear(d_model*len(cat_feats), d_hidden)\n\n            self.encoder1 = Conformer(d_hidden, nheads, dropout, 5)\n            self.encoder2 = Conformer(d_hidden, nheads, dropout, 10)\n            self.encoder3 = Conformer(d_hidden, nheads, dropout, 5)\n\n            self.post_norm = nn.LayerNorm(d_hidden)\n\n            self.rnn1 = nn.LSTM(d_hidden,d_hidden,1,batch_first=True, bidirectional=True)\n            self.rnn2 = nn.LSTM(2*d_hidden,d_hidden,1,batch_first=True, bidirectional=False)\n\n            # Readout layer\n            self.readout = nn.Sequential(\n                nn.BatchNorm1d(4*d_hidden),\n                nn.Linear(4*d_hidden, out_dim),\n                nn.GELU(),\n                nn.Dropout(0.1)\n            )\n\n        def forward(self, x: Tensor, x_cat: Tensor) -> Tensor:\n            \"\"\"Forward pass.\n\n            Shape:\n                x: (B, P, C)\n                x_cat: (B, P, M)\n            \"\"\"\n            mask = torch.all(x_cat==-1,dim=-1)\n            # Categorical embeddings\n            x_cat = x_cat + 1\n            x_emb = []\n            for i in range(len(self.cat_feats)):\n                x_emb.append(self.embs[i](x_cat[..., i]))  # (B, P, emb_dim)\n            x_emb = torch.cat(x_emb, dim=-1)  # (B, P, C')\n            x_emb = self.emb_enc(x_emb) + x_emb  # (B, P, C')\n            x = self.dense_layer(x) * x_emb  # (B, P, C')\n\n            x = self.combine_layer(x)\n\n            x = self.encoder1(x,mask)\n            x = self.encoder2(x,mask)\n            x = self.encoder3(x,mask)\n\n            x = self.post_norm(x)\n\n            x = self.rnn1(x)[0]\n            x = self.rnn2(x)[0]\n\n            x_std = torch.std(x, dim=1)\n            x_sum = torch.sum(x, dim=1)\n            x_max = torch.max(x, dim=1).values\n            x_last = x[:,-1]\n\n            x = torch.cat([x_std, x_sum, x_max, x_last], dim=1)\n            x = self.readout(x)  # (B, out_dim)\n\n            return x\n\n\n    class EventConvSimple(nn.Module):\n\n        def __init__(self, out_dim: int, **model_cfg: Any):\n            self.name = self.__class__.__name__\n            super(EventConvSimple, self).__init__()\n\n            enc_out_dim = 512\n\n            # Network parameters\n            self.out_dim = out_dim\n            self.cat_feats = model_cfg[\"cat_feats\"]\n\n            self.encoder = EventAwareEncoder(d_model=32, out_dim=enc_out_dim, cat_feats=self.cat_feats)\n            self.clf = nn.Linear(enc_out_dim, out_dim)\n\n        def forward(self, x: Tensor, x_cat: Tensor) -> Tensor:\n            \"\"\"Forward pass.\n\n            Shape:\n                x: (B, P, C)\n                x_cat: (B, P, M)\n            \"\"\"\n            x = self.encoder(x, x_cat)\n            x = self.clf(x)\n\n            return x\n    return EventConvSimple(out_dim, **model_cfg)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:03.270324Z","iopub.execute_input":"2023-06-28T10:52:03.270700Z","iopub.status.idle":"2023-06-28T10:52:03.316544Z","shell.execute_reply.started":"2023-06-28T10:52:03.270668Z","shell.execute_reply":"2023-06-28T10:52:03.314430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Model_v47(out_dim, **model_cfg):\n    class GEGLU(nn.Module):\n        \"\"\"\n        References:\n            Shazeer et al., \"GLU Variants Improve Transformer,\" 2020.\n            https://arxiv.org/abs/2002.05202\n        \"\"\"\n        def __init__(self, dim=1):\n            super(GEGLU, self).__init__()\n            self.dim=dim\n\n        def geglu(self, x: Tensor) -> Tensor:\n            assert x.shape[self.dim] % 2 == 0\n            a, b = x.chunk(2, dim=self.dim)\n            return a * F.gelu(b)\n\n        def forward(self, x: Tensor) -> Tensor:\n            return self.geglu(x)\n\n\n    class Conformer(nn.Module):\n        \"\"\"Dilated temporal convolution layer.\n\n        Considering the time cost, I currently disable dilation.\n        \"\"\"\n\n        def __init__(\n            self,\n            d_model: int,\n            nheads: int,\n            dropout: float = 0.1,\n            kernel_size=5\n        ):\n            super(Conformer, self).__init__()\n\n            self.norm1 = nn.LayerNorm(d_model)\n            self.conv = nn.Sequential(nn.Conv1d(d_model,2*d_model,\n                                                groups=d_model,\n                                                kernel_size=kernel_size,padding='same'),\n                                      GEGLU(),\n                                      nn.Dropout(dropout),\n                                      nn.Conv1d(d_model,d_model,\n                                                groups=d_model,\n                                                kernel_size=kernel_size,padding='same'),\n                                      nn.Dropout(dropout))\n\n            self.norm2 = nn.LayerNorm(d_model)\n            self.attn = nn.MultiheadAttention(d_model,nheads,dropout=dropout,batch_first=True)\n\n            self.norm3 = nn.LayerNorm(d_model)\n            self.mlp = nn.Sequential(nn.Linear(d_model,4*d_model),\n                                      GEGLU(dim=-1),\n                                      nn.Dropout(dropout),\n                                      nn.Linear(2*d_model,d_model),\n                                      nn.Dropout(dropout))\n\n        def forward(self, x: Tensor, mask: Tensor) -> Tensor:\n            \"\"\"Forward pass.\n\n            Shape:\n                x: (B, L, D)\n                mask: (B, L)\n            \"\"\"\n\n            x2 = self.norm1(x)\n            x2 = x2.permute(0,2,1)\n            x2 = self.conv(x2)\n            x2 = x2.permute(0,2,1)\n            x = x + x2\n\n            x2 = self.norm2(x)\n            x2 = self.attn(x2[:,[-1]],x2,x2,key_padding_mask=mask)[0]\n            x = x + x2\n\n            x2 = self.norm3(x)\n            x2 = self.mlp(x2)\n            x = x + x2\n            return x\n\n\n    class EventAwareEncoder(nn.Module):\n        \"\"\"Event-aware encoder based on 1D-Conv.\"\"\"\n\n        def __init__(\n            self,\n            d_model: int = 32,\n            out_dim: int = 128,\n            readout: bool = True,\n            cat_feats: List[str] = [\"event_comb_code\", \"room_fqid_code\"]\n        ):\n            super(EventAwareEncoder, self).__init__()\n\n            # Network parameters\n            self.d_model = d_model\n            self.out_dim = out_dim\n            self.cat_feats = cat_feats\n            d_hidden = 128\n            nheads = 2\n            dropout = 0.2\n\n            # Model blocks\n            # Categorical embeddings\n            self.embs = nn.ModuleList()\n            for cat_feat in cat_feats:\n                self.embs.append(nn.Embedding(CAT_FEAT_SIZE[cat_feat] + 1, d_model, padding_idx=0))\n            self.emb_enc = nn.Sequential(\n                nn.Linear(d_model*len(cat_feats), 2*d_model*len(cat_feats)),\n                nn.GELU(),\n                nn.Linear(2*d_model*len(cat_feats), d_model*len(cat_feats)),\n                nn.GELU(),\n            )\n            self.dropout = nn.Dropout(0.2)\n            self.dense_layer = nn.Sequential(nn.Linear(1,2*d_model*len(cat_feats)), nn.GLU())\n\n            # Feature extractor        \n            self.combine_layer = nn.Linear(d_model*len(cat_feats), d_hidden)\n\n            self.encoder1 = Conformer(d_hidden, nheads, dropout, 10)\n            self.encoder2 = Conformer(d_hidden, nheads, dropout, 20)\n            self.encoder3 = Conformer(d_hidden, nheads, dropout, 10)\n\n            self.post_norm = nn.LayerNorm(d_hidden)\n\n            self.rnn1 = nn.LSTM(d_hidden,d_hidden,1,batch_first=True, bidirectional=True)\n            self.rnn2 = nn.LSTM(2*d_hidden,d_hidden,1,batch_first=True, bidirectional=False)\n\n            # Readout layer\n            self.readout = nn.Sequential(\n                nn.BatchNorm1d(4*d_hidden),\n                nn.Linear(4*d_hidden, out_dim),\n                nn.GELU(),\n                nn.Dropout(0.1)\n            )\n\n        def forward(self, x: Tensor, x_cat: Tensor) -> Tensor:\n            \"\"\"Forward pass.\n\n            Shape:\n                x: (B, P, C)\n                x_cat: (B, P, M)\n            \"\"\"\n            mask = torch.all(x_cat==-1,dim=-1)\n            # Categorical embeddings\n            x_cat = x_cat + 1\n            x_emb = []\n            for i in range(len(self.cat_feats)):\n                x_emb.append(self.embs[i](x_cat[..., i]))  # (B, P, emb_dim)\n            x_emb = torch.cat(x_emb, dim=-1)  # (B, P, C')\n            x_emb = self.emb_enc(x_emb) + x_emb  # (B, P, C')\n            x = self.dense_layer(x) * x_emb  # (B, P, C')\n\n            x = self.combine_layer(x)\n\n            x = self.encoder1(x,mask)\n            x = self.encoder2(x,mask)\n            x = self.encoder3(x,mask)\n\n            x = self.post_norm(x)\n\n            x = self.rnn1(x)[0]\n            x = self.rnn2(x)[0]\n\n            x_std = torch.std(x, dim=1)\n            x_sum = torch.sum(x, dim=1)\n            x_max = torch.max(x, dim=1).values\n            x_last = x[:,-1]\n\n            x = torch.cat([x_std, x_sum, x_max, x_last], dim=1)\n            x = self.readout(x)  # (B, out_dim)\n\n            return x\n\n\n    class EventConvSimple(nn.Module):\n\n        def __init__(self, out_dim: int, **model_cfg: Any):\n            self.name = self.__class__.__name__\n            super(EventConvSimple, self).__init__()\n\n            enc_out_dim = 512\n\n            # Network parameters\n            self.out_dim = out_dim\n            self.cat_feats = model_cfg[\"cat_feats\"]\n\n            self.encoder = EventAwareEncoder(d_model=32, out_dim=enc_out_dim, cat_feats=self.cat_feats)\n            self.clf = nn.Linear(enc_out_dim, out_dim)\n\n        def forward(self, x: Tensor, x_cat: Tensor) -> Tensor:\n            \"\"\"Forward pass.\n\n            Shape:\n                x: (B, P, C)\n                x_cat: (B, P, M)\n            \"\"\"\n            x = self.encoder(x, x_cat)\n            x = self.clf(x)\n\n            return x\n    \n    return EventConvSimple(out_dim, **model_cfg)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:03.319258Z","iopub.execute_input":"2023-06-28T10:52:03.319739Z","iopub.status.idle":"2023-06-28T10:52:03.361006Z","shell.execute_reply.started":"2023-06-28T10:52:03.319699Z","shell.execute_reply":"2023-06-28T10:52:03.359461Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def Model_v51(out_dim, **model_cfg):\n    class GEGLU(nn.Module):\n        \"\"\"\n        References:\n            Shazeer et al., \"GLU Variants Improve Transformer,\" 2020.\n            https://arxiv.org/abs/2002.05202\n        \"\"\"\n        def __init__(self, dim=1):\n            super(GEGLU, self).__init__()\n            self.dim=dim\n\n        def geglu(self, x: Tensor) -> Tensor:\n            assert x.shape[self.dim] % 2 == 0\n            a, b = x.chunk(2, dim=self.dim)\n            return a * F.gelu(b)\n\n        def forward(self, x: Tensor) -> Tensor:\n            return self.geglu(x)\n\n\n    class EventAwareEncoder(nn.Module):\n        \"\"\"Event-aware encoder based on 1D-Conv.\"\"\"\n\n        def __init__(\n            self,\n            d_model: int = 32,\n            out_dim: int = 128,\n            readout: bool = True,\n            cat_feats: List[str] = [\"event_comb_code\", \"room_fqid_code\"]\n        ):\n            super(EventAwareEncoder, self).__init__()\n\n            # Network parameters\n            self.d_model = d_model\n            self.out_dim = out_dim\n            self.cat_feats = cat_feats\n            d_hidden = 128\n            nheads = 2\n            dropout = 0.2\n\n            # Model blocks\n            # Categorical embeddings\n            self.embs = nn.ModuleList()\n            for cat_feat in cat_feats:\n                self.embs.append(nn.Embedding(CAT_FEAT_SIZE[cat_feat] + 1, d_model, padding_idx=0))\n            self.emb_enc = nn.Sequential(\n                nn.Linear(d_model*len(cat_feats), 2*d_model*len(cat_feats)),\n                nn.GELU(),\n                nn.Linear(2*d_model*len(cat_feats), d_model*len(cat_feats)),\n                nn.GELU(),\n            )\n            self.dropout = nn.Dropout(0.2)\n            self.dense_layer = nn.Sequential(nn.Linear(1,2*d_model*len(cat_feats)), nn.GLU())\n\n            # Feature extractor        \n            self.combine_layer = nn.Linear(d_model*len(cat_feats), d_hidden)\n\n            self.encoder = nn.TransformerEncoder(nn.TransformerEncoderLayer(d_hidden,\n                                                                             nheads,\n                                                                             4*d_hidden,\n                                                                             dropout,\n                                                                             'gelu',\n                                                                             batch_first=True,\n                                                                             norm_first=True\n                                                                            )\n                                                 ,3)\n\n            self.post_norm = nn.LayerNorm(d_hidden)\n\n            self.rnn1 = nn.LSTM(d_hidden,d_hidden,1,batch_first=True, bidirectional=True)\n            self.rnn2 = nn.LSTM(2*d_hidden,d_hidden,1,batch_first=True, bidirectional=False)\n\n            # Readout layer\n            self.readout = nn.Sequential(\n                nn.BatchNorm1d(4*d_hidden),\n                nn.Linear(4*d_hidden, out_dim),\n                nn.GELU(),\n                nn.Dropout(0.1)\n            )\n\n        def forward(self, x: Tensor, x_cat: Tensor) -> Tensor:\n            \"\"\"Forward pass.\n\n            Shape:\n                x: (B, P, C)\n                x_cat: (B, P, M)\n            \"\"\"\n            mask = torch.all(x_cat==-1,dim=-1)\n            # Categorical embeddings\n            x_cat = x_cat + 1\n            x_emb = []\n            for i in range(len(self.cat_feats)):\n                x_emb.append(self.embs[i](x_cat[..., i]))  # (B, P, emb_dim)\n            x_emb = torch.cat(x_emb, dim=-1)  # (B, P, C')\n            x_emb = self.emb_enc(x_emb) + x_emb  # (B, P, C')\n            x = self.dense_layer(x) * x_emb  # (B, P, C')\n\n            x = self.combine_layer(x)\n\n            x = self.encoder(x,src_key_padding_mask=mask)\n\n            x = self.post_norm(x)\n\n            x = self.rnn1(x)[0]\n            x = self.rnn2(x)[0]\n\n            x_std = torch.std(x, dim=1)\n            x_sum = torch.sum(x, dim=1)\n            x_max = torch.max(x, dim=1).values\n            x_last = x[:,-1]\n\n            x = torch.cat([x_std, x_sum, x_max, x_last], dim=1)\n            x = self.readout(x)  # (B, out_dim)\n\n            return x\n\n\n    class EventConvSimple(nn.Module):\n\n        def __init__(self, out_dim: int, **model_cfg: Any):\n            self.name = self.__class__.__name__\n            super(EventConvSimple, self).__init__()\n\n            enc_out_dim = 512\n\n            # Network parameters\n            self.out_dim = out_dim\n            self.cat_feats = model_cfg[\"cat_feats\"]\n\n            self.encoder = EventAwareEncoder(d_model=32, out_dim=enc_out_dim, cat_feats=self.cat_feats)\n            self.clf = nn.Linear(enc_out_dim, out_dim)\n\n        def forward(self, x: Tensor, x_cat: Tensor) -> Tensor:\n            \"\"\"Forward pass.\n\n            Shape:\n                x: (B, P, C)\n                x_cat: (B, P, M)\n            \"\"\"\n            x = self.encoder(x, x_cat)\n            x = self.clf(x)\n\n            return x\n    return EventConvSimple(out_dim, **model_cfg)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:03.362941Z","iopub.execute_input":"2023-06-28T10:52:03.363729Z","iopub.status.idle":"2023-06-28T10:52:03.396104Z","shell.execute_reply.started":"2023-06-28T10:52:03.363684Z","shell.execute_reply":"2023-06-28T10:52:03.394622Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"models1 = {\"0-4\": [], \"5-12\": [], \"13-22\": []}\nfor model_file in os.listdir(model_path1):\n    if \"model_\" in model_file:\n        if \"0-4\" in model_file:\n            model = Model_v32(3, **model_cfg)\n            model.eval()\n            model.load_state_dict(\n                torch.load(\n                    os.path.join(model_path1, model_file),\n                    map_location=torch.device('cpu')\n                )\n            )\n            models1[\"0-4\"].append(model)\n        elif \"5-12\" in model_file:\n            model = Model_v32(10, **model_cfg)\n            model.eval()\n            model.load_state_dict(\n                torch.load(\n                    os.path.join(model_path1, model_file),\n                    map_location=torch.device('cpu')\n                )\n            )\n            models1[\"5-12\"].append(model)\n        elif \"13-22\" in model_file:\n            model = Model_v32(5, **model_cfg)\n            model.eval()\n            model.load_state_dict(\n                torch.load(\n                    os.path.join(model_path1, model_file),\n                    map_location=torch.device('cpu')\n                )\n            )\n            models1[\"13-22\"].append(model)\n\nmodels2 = {\"0-4\": [], \"5-12\": [], \"13-22\": []}\nfor model_file in os.listdir(model_path2):\n    if \"model_\" in model_file:\n        if \"0-4\" in model_file:\n            model = Model_v47(3, **model_cfg)\n            model.eval()\n            model.load_state_dict(\n                torch.load(\n                    os.path.join(model_path2, model_file),\n                    map_location=torch.device('cpu')\n                )\n            )\n            models2[\"0-4\"].append(model)\n        elif \"5-12\" in model_file:\n            model = Model_v47(10, **model_cfg)\n            model.eval()\n            model.load_state_dict(\n                torch.load(\n                    os.path.join(model_path2, model_file),\n                    map_location=torch.device('cpu')\n                )\n            )\n            models2[\"5-12\"].append(model)\n        elif \"13-22\" in model_file:\n            model = Model_v47(5, **model_cfg)\n            model.eval()\n            model.load_state_dict(\n                torch.load(\n                    os.path.join(model_path2, model_file),\n                    map_location=torch.device('cpu')\n                )\n            )\n            models2[\"13-22\"].append(model)\n            \nmodels3 = []\nfor model_file in os.listdir(model_path3):\n    if \"model_\" in model_file:\n        model = Model_v51(18, **model_cfg)\n        model.eval()\n        model.load_state_dict(\n            torch.load(\n                os.path.join(model_path3, model_file),\n                map_location=torch.device('cpu')\n            )\n        )\n        models3.append(model)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:03.398259Z","iopub.execute_input":"2023-06-28T10:52:03.399041Z","iopub.status.idle":"2023-06-28T10:52:05.664414Z","shell.execute_reply.started":"2023-06-28T10:52:03.398992Z","shell.execute_reply":"2023-06-28T10:52:05.663489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cat2code_ = {}\nfor cat_feat in CAT_FEATS:\n    with open(os.path.join(cat2code_path, f\"{cat_feat[:-5]}2code.pkl\"), \"rb\") as f:\n        cat2code_[cat_feat] = pickle.load(f)\n        \net_diff_upper_bound = 3.6e6","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:05.665652Z","iopub.execute_input":"2023-06-28T10:52:05.666241Z","iopub.status.idle":"2023-06-28T10:52:05.680216Z","shell.execute_reply.started":"2023-06-28T10:52:05.666207Z","shell.execute_reply":"2023-06-28T10:52:05.678591Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@torch.no_grad()\ndef quick_infer(X: Tuple[Tensor, Optional[Tensor]], models: List[nn.Module]) -> Tensor:\n    \n    x, x_cat = X\n    n_models = len(models)\n    y_pred = []\n    for i, model in enumerate(models):\n        y_pred.append(F.sigmoid(model(x, x_cat).squeeze(0)).detach().numpy())   # (1, n_qns (out_dim))\n    y_pred = np.mean(y_pred,axis=0)\n    return y_pred\n\n@torch.no_grad()\ndef quick_infer2(X: Tuple[Tensor, Optional[Tensor]], models: List[nn.Module], grp: str) -> Tensor:\n    x, x_cat = X\n    n_models = len(models)\n    y_pred = []\n    for i, model in enumerate(models):\n        p = F.sigmoid(model(x, x_cat).squeeze(0)).detach().numpy()\n        if grp == '0-4':\n            p = p[:3]\n        elif grp == '5-12':\n            p = p[3:13]\n        else:\n            p = p[13:]\n        y_pred.append(p)   # (1, n_qns (out_dim))\n    y_pred = np.mean(y_pred,axis=0)\n    return y_pred","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:05.682078Z","iopub.execute_input":"2023-06-28T10:52:05.682559Z","iopub.status.idle":"2023-06-28T10:52:05.695158Z","shell.execute_reply.started":"2023-06-28T10:52:05.682519Z","shell.execute_reply":"2023-06-28T10:52:05.694034Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## END NN","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"std_catb_grp_base = \"/kaggle/input/psp-catboost-std-retrain\"\nstd_catb_18_in_1_base = \"/kaggle/input/psp-catb-1-model-18-qs/std-catb-26-06-2023/std-catb-26-06-2023\"\nstd_xgb_18_in_1_base = \"/kaggle/input/psp-catb-1-model-18-qs/std-xgb-26-06-2023/std-xgb-26-06-2023\"\nstd_xgb_18_in_1_2_base = \"/kaggle/input/psp-catb-1-model-18-qs/std-xgb-28-06-2023/std-xgb-28-06-2023\"","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:05.696888Z","iopub.execute_input":"2023-06-28T10:52:05.697739Z","iopub.status.idle":"2023-06-28T10:52:05.715807Z","shell.execute_reply.started":"2023-06-28T10:52:05.697695Z","shell.execute_reply":"2023-06-28T10:52:05.714271Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f_read = open('/kaggle/input/psp-catboost-sort-by-time-only/unused_data_values.pkl', 'rb')\nunused_data_values = pickle.load(f_read)\nf_read.close()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:05.720549Z","iopub.execute_input":"2023-06-28T10:52:05.721485Z","iopub.status.idle":"2023-06-28T10:52:05.731209Z","shell.execute_reply.started":"2023-06-28T10:52:05.721438Z","shell.execute_reply":"2023-06-28T10:52:05.730199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"trn_qs = [\n    (\"0-4\", 1, 4),\n    (\"5-12\", 4, 14),\n    (\"13-22\", 14, 19),\n]\n\nstd_catb_grp_models = {}\nstd_catb_grp_features = {}\n\nstd_catb_18in1_models = {}\nstd_catb_18in1_features = list(pd.read_csv(f\"{std_catb_18_in_1_base}/importance_0_final.csv\").feature.values)\n\nstd_xgb_18in1_models = {}\nstd_xgb_18in1_features = list(pd.read_csv(f\"{std_xgb_18_in_1_base}/importance_0_final.csv\").feature.values)\n\nstd_xgb_18in1_2_models = {}\nstd_xgb_18in1_2_features = list(pd.read_csv(f\"{std_xgb_18_in_1_2_base}/importance_0_final.csv\").feature.values)\n\nfor grp_idx, (grp, a, b) in enumerate(trn_qs):\n    std_catb_grp_models[grp] = []\n    for fold in range(5):\n        std_catb_grp_models[grp].append(\n            CatBoostClassifier().load_model(\n                f\"{std_catb_grp_base}/fold{fold}_grp_{grp}_final.cbm\"\n            )\n        )\n    std_catb_grp_features[grp] = list(pd.read_csv(f\"{std_catb_grp_base}/importance_0_grp_{grp}_final.csv\").feature.values)\n    \nfor fold in range(5):\n    std_catb_18in1_models[fold] = CatBoostClassifier().load_model(\n        f\"{std_catb_18_in_1_base}/fold{fold}_final.cbm\"\n    )    \n    std_xgb_18in1_models[fold] = xgb.Booster(\n        model_file=f\"{std_xgb_18_in_1_base}/fold{fold}_final.cbm\"\n    )   \n    std_xgb_18in1_2_models[fold] = xgb.Booster(\n        model_file=f\"{std_xgb_18_in_1_2_base}/fold{fold}_final.cbm\"\n    )","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:05.732631Z","iopub.execute_input":"2023-06-28T10:52:05.733214Z","iopub.status.idle":"2023-06-28T10:52:07.472188Z","shell.execute_reply.started":"2023-06-28T10:52:05.733181Z","shell.execute_reply":"2023-06-28T10:52:07.470937Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"union_features = {}\nfor grp, _, _ in trn_qs:\n    union_features[grp] = set(std_catb_grp_features[grp]) | set(std_catb_18in1_features)\n    print(len(union_features[grp]))","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:07.473703Z","iopub.execute_input":"2023-06-28T10:52:07.474395Z","iopub.status.idle":"2023-06-28T10:52:07.482995Z","shell.execute_reply.started":"2023-06-28T10:52:07.474356Z","shell.execute_reply":"2023-06-28T10:52:07.481720Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(\"/kaggle/input/psp-catboost-std-retrain/constants.pkl\", \"rb\") as f:\n    constants = pickle.load(f)\n    \nfor key in constants.keys():\n    print(key)\n    \nevent_name_feature = constants[\"event_name_feature\"]\nfqid_feature = constants[\"fqid_feature\"]\nroom_fqid_feature = constants[\"room_fqid_feature\"]\ntext_feature = constants[\"text_feature\"]\ntext_fqid_feature = constants[\"text_fqid_feature\"]\nCATS = constants[\"CATS\"]\nNUMS = constants[\"NUMS\"]\nname_feature = constants[\"name_feature\"]\npage_feature = constants[\"page_feature\"]","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:07.484894Z","iopub.execute_input":"2023-06-28T10:52:07.485857Z","iopub.status.idle":"2023-06-28T10:52:07.502014Z","shell.execute_reply.started":"2023-06-28T10:52:07.485794Z","shell.execute_reply":"2023-06-28T10:52:07.500547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def time_feature(train):\n    train[\"year\"] = train[\"session_id\"].apply(lambda x: int(str(x)[:2])).astype(np.uint8)\n    train[\"month\"] = train[\"session_id\"].apply(lambda x: int(str(x)[2:4])+1).astype(np.uint8)\n    train[\"day\"] = train[\"session_id\"].apply(lambda x: int(str(x)[4:6])).astype(np.uint8)\n    train[\"hour\"] = train[\"session_id\"].apply(lambda x: int(str(x)[6:8])).astype(np.uint8)\n    train[\"minute\"] = train[\"session_id\"].apply(lambda x: int(str(x)[8:10])).astype(np.uint8)\n    train[\"second\"] = train[\"session_id\"].apply(lambda x: int(str(x)[10:12])).astype(np.uint8)\n    return train\n\ncolumns = [\n    pl.col(\"page\").cast(pl.Float32),\n    (\n        (pl.col(\"elapsed_time\") - pl.col(\"elapsed_time\").shift(1)) # time used for each action\n         .fill_null(0)\n         .clip(0, 1e9)\n         .over([\"session_id\", \"level_group\"])\n         .alias(\"elapsed_time_diff\")\n    ),\n    (\n        (pl.col(\"elapsed_time\").shift(-1) - pl.col(\"elapsed_time\")) # time used for each action\n         .fill_null(0)\n         .clip(0, 1e9)\n         .over([\"session_id\", \"level_group\"])\n         .alias(\"elapsed_time_diff_forward\")\n    ),\n    (\n        (pl.col(\"screen_coor_x\") - pl.col(\"screen_coor_x\").shift(1)) # location x changed for click \n         .abs()\n         .over([\"session_id\", \"level_group\"])\n    ),\n    (\n        (pl.col(\"screen_coor_y\") - pl.col(\"screen_coor_y\").shift(1)) # location y changed for click \n         .abs()\n         .over([\"session_id\", \"level_group\"])\n    ),\n    pl.col(\"fqid\").fill_null(\"fqid_None\"),\n    pl.col(\"text_fqid\").fill_null(\"text_fqid_None\"),\n    pl.col(\"text\").fill_null(\"text_None\"),\n    pl.col(\"room_coor_x\").forward_fill().alias(\"room_coor_x\"),\n    pl.col(\"room_coor_y\").forward_fill().alias(\"room_coor_y\"),\n    pl.col(\"screen_coor_x\").forward_fill().alias(\"screen_coor_x\"),\n    pl.col(\"screen_coor_y\").forward_fill().alias(\"screen_coor_y\"),\n]\n\ncolumns_1 = [\n    (pl.col(\"room_coor_x\") - pl.col(\"room_coor_x\").shift(1)).over([\"session_id\"]).pow(2).alias(\"room_coor_x_dis\"),\n    (pl.col(\"room_coor_y\") - pl.col(\"room_coor_y\").shift(1)).over([\"session_id\"]).pow(2).alias(\"room_coor_y_dis\"),    \n    (pl.col(\"screen_coor_x\") - pl.col(\"screen_coor_x\").shift(1)).over([\"session_id\"]).pow(2).alias(\"screen_coor_x_dis\"),\n    (pl.col(\"screen_coor_y\") - pl.col(\"screen_coor_y\").shift(1)).over([\"session_id\"]).pow(2).alias(\"screen_coor_y_dis\"),    \n    pl.col(\"fqid\").cast(pl.Utf8),\n    pl.col(\"room_fqid\").cast(pl.Utf8),\n    pl.col(\"text_fqid\").cast(pl.Utf8),\n]\n\ncolumns_2 = [\n    (pl.col(\"room_coor_x_dis\") + pl.col(\"room_coor_y_dis\")).pow(1/2).alias(\"room_dis\"),\n    (pl.col(\"screen_coor_x_dis\") + pl.col(\"screen_coor_y_dis\")).pow(1/2).alias(\"screen_dis\"),\n]","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:07.504192Z","iopub.execute_input":"2023-06-28T10:52:07.504675Z","iopub.status.idle":"2023-06-28T10:52:07.529136Z","shell.execute_reply.started":"2023-06-28T10:52:07.504623Z","shell.execute_reply":"2023-06-28T10:52:07.528147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def feature_engineer(x, grp, use_extra, feature_suffix):\n    pl.toggle_string_cache(True)\n    if (grp=='0-4'):\n        level_feature = range(5)\n    elif (grp=='5-12'):\n#         level_feature = [0,1,2,3,4,5,6,7,8,9,10,11,12]\n        level_feature = range(5, 13)\n    else:\n#         level_feature = [0,1,2,3,4,5,6,7,8,9,10,11,12,13,14,15,16,17,18,19,20,21,22]\n        level_feature = range(13, 23)\n        \n    aggs = [\n        pl.col(\"index\").count().alias(f\"session_number_{feature_suffix}\"),\n        pl.col(\"elapsed_time\").apply(lambda s: s.max()-s.min()).alias(f\"session_time_{feature_suffix}\"),\n        \n        *[pl.col(c).drop_nulls().n_unique().alias(f\"{c}_unique_{feature_suffix}\") for c in CATS],\n        \n        *[pl.col(c).sum().alias(f\"{c}_sum_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).mean().alias(f\"{c}_mean_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).min().alias(f\"{c}_min_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).max().alias(f\"{c}_max_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).std().alias(f\"{c}_std_{feature_suffix}\") for c in NUMS],\n        *[pl.col(c).count().alias(f\"{c}_count_{feature_suffix}\") for c in NUMS],\n        \n    ]\n    \n    for cat in ['fqid', 'room_fqid', 'name', 'text', 'text_fqid', 'level', 'page']:\n        for col in ['elapsed_time_diff']:\n            aggs.extend([\n                *[pl.col(col).filter(pl.col(cat)==c).sum().alias(f\"{cat}_{c}_{col}_sum_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n                *[pl.col(col).filter(pl.col(cat)==c).mean().alias(f\"{cat}_{c}_{col}_mean_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n                *[pl.col(col).filter(pl.col(cat)==c).max().alias(f\"{cat}_{c}_{col}_max_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n                *[pl.col(col).filter(pl.col(cat)==c).min().alias(f\"{cat}_{c}_{col}_min_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n                *[pl.col(col).filter(pl.col(cat)==c).std().alias(f\"{cat}_{c}_{col}_std_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n                *[pl.col(col).filter((pl.col(cat)==c) & (~pl.col(col).is_null())).count().alias(f\"{cat}_{c}_{col}_count_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n            ])\n        for col in ['elapsed_time_diff_forward', 'hover_duration', 'room_dis', 'screen_dis']:\n            aggs.extend([\n                *[pl.col(col).filter(pl.col(cat)==c).sum().alias(f\"{cat}_{c}_{col}_sum_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n                *[pl.col(col).filter(pl.col(cat)==c).mean().alias(f\"{cat}_{c}_{col}_mean_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n                *[pl.col(col).filter(pl.col(cat)==c).max().alias(f\"{cat}_{c}_{col}_max_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n                *[pl.col(col).filter(pl.col(cat)==c).min().alias(f\"{cat}_{c}_{col}_min_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n                *[pl.col(col).filter(pl.col(cat)==c).std().alias(f\"{cat}_{c}_{col}_std_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n            ])\n        \n        aggs.extend([\n                *[pl.col('elapsed_time').filter(pl.col(cat)==c).diff().fill_null(0).clip(0, 1e9).sum().alias(f\"{cat}_{c}_ET2_sum_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n                *[pl.col('elapsed_time').filter(pl.col(cat)==c).diff().fill_null(0).clip(0, 1e9).mean().alias(f\"{cat}_{c}_ET2_mean_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n                *[pl.col('elapsed_time').filter(pl.col(cat)==c).diff().fill_null(0).clip(0, 1e9).max().alias(f\"{cat}_{c}_ET2_max_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n                *[pl.col('elapsed_time').filter(pl.col(cat)==c).diff().fill_null(0).clip(0, 1e9).min().alias(f\"{cat}_{c}_ET2_min_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n                *[pl.col('elapsed_time').filter(pl.col(cat)==c).diff().fill_null(0).clip(0, 1e9).std().alias(f\"{cat}_{c}_ET2_std_{feature_suffix}\") for c in eval(f'{cat}_feature')],\n            ])\n    \n    aggs.extend([*[pl.col('elapsed_time').filter(pl.col('level')==c).\\\n                   apply(lambda s: s.max()-s.min())\\\n                   .alias(f\"level_{c}_duration_{feature_suffix}\") for c in level_feature]])\n\n    for col in ['elapsed_time_diff']:\n\n        aggs.extend([\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"text\").is_in(unused_data_values[level][\"text\"]))).count().alias(f\"level_{level}_unused_text_{col}_counts\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"fqid\").is_in(unused_data_values[level][\"fqid\"]))).count().alias(f\"level_{level}_unused_fqid_{col}_counts\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"room_fqid\").is_in(unused_data_values[level][\"room_fqid\"]))).count().alias(f\"level_{level}_unused_room_fqid_{col}_counts\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"text_fqid\").is_in(unused_data_values[level][\"text_fqid\"]))).count().alias(f\"level_{level}_unused_text_fqid_{col}_counts\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"text\").is_in(unused_data_values[level][\"text\"]))).sum().alias(f\"level_{level}_unused_text_{col}_sum\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"fqid\").is_in(unused_data_values[level][\"fqid\"]))).sum().alias(f\"level_{level}_unused_fqid_{col}_sum\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"room_fqid\").is_in(unused_data_values[level][\"room_fqid\"]))).sum().alias(f\"level_{level}_unused_room_fqid_{col}_sum\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"text_fqid\").is_in(unused_data_values[level][\"text_fqid\"]))).sum().alias(f\"level_{level}_unused_text_fqid_{col}_sum\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"text\").is_in(unused_data_values[level][\"text\"]))).max().alias(f\"level_{level}_unused_text_{col}_max\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"fqid\").is_in(unused_data_values[level][\"fqid\"]))).max().alias(f\"level_{level}_unused_fqid_{col}_max\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"room_fqid\").is_in(unused_data_values[level][\"room_fqid\"]))).max().alias(f\"level_{level}_unused_room_fqid_{col}_max\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"text_fqid\").is_in(unused_data_values[level][\"text_fqid\"]))).max().alias(f\"level_{level}_unused_text_fqid_{col}_max\") for level in level_feature],\n        ])\n    \n    for col in ['elapsed_time_diff_forward', 'hover_duration', 'room_dis', 'screen_dis']:\n        \n        aggs.extend([\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"text\").is_in(unused_data_values[level][\"text\"]))).sum().alias(f\"level_{level}_unused_text_{col}_sum\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"fqid\").is_in(unused_data_values[level][\"fqid\"]))).sum().alias(f\"level_{level}_unused_fqid_{col}_sum\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"room_fqid\").is_in(unused_data_values[level][\"room_fqid\"]))).sum().alias(f\"level_{level}_unused_room_fqid_{col}_sum\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"text_fqid\").is_in(unused_data_values[level][\"text_fqid\"]))).sum().alias(f\"level_{level}_unused_text_fqid_{col}_sum\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"text\").is_in(unused_data_values[level][\"text\"]))).max().alias(f\"level_{level}_unused_text_{col}_max\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"fqid\").is_in(unused_data_values[level][\"fqid\"]))).max().alias(f\"level_{level}_unused_fqid_{col}_max\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"room_fqid\").is_in(unused_data_values[level][\"room_fqid\"]))).max().alias(f\"level_{level}_unused_room_fqid_{col}_max\") for level in level_feature],\n             *[pl.col(col).filter((pl.col(\"level\") == level) & (pl.col(\"text_fqid\").is_in(unused_data_values[level][\"text_fqid\"]))).max().alias(f\"level_{level}_unused_text_fqid_{col}_max\") for level in level_feature],\n        ])\n    \n    # Only keep the feature transformation that are in the feature list\n    aggs = [\n        agg\n        for agg in aggs\n        if agg.meta.output_name() in union_features[grp]\n    ]\n    df = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n    \n    consecutive_unique_room_fqid_counts = x.filter((pl.col(\"room_fqid\") != pl.col(\"room_fqid\").shift(1)).over([\"session_id\"])).groupby(\"session_id\").agg([\n        pl.col(\"room_fqid\").count().alias(\"consecutive_unique_room_fqid_counts\")\n    ])\n    \n    consecutive_unique_fqid_counts = x.filter((pl.col(\"fqid\") != pl.col(\"fqid\").shift(1)).over([\"session_id\"])).groupby(\"session_id\").agg([\n        pl.col(\"fqid\").count().alias(\"consecutive_unique_fqid_counts\")\n    ])\n    \n    consecutive_unique_text_fqid_counts = x.filter((pl.col(\"text_fqid\") != pl.col(\"text_fqid\").shift(1)).over([\"session_id\"])).groupby(\"session_id\").agg([\n        pl.col(\"text_fqid\").count().alias(\"consecutive_unique_text_fqid_counts\")\n    ])\n    \n    consecutive_data = consecutive_unique_room_fqid_counts.join(consecutive_unique_fqid_counts, on=\"session_id\", how='left').join(consecutive_unique_text_fqid_counts, on=\"session_id\", how='left')\n    \n    for level in level_feature:\n        for feature_type in ['fqid', 'room_fqid', 'text_fqid']:\n            tmp_consecutive_unique_data = x.filter(pl.col(\"level\") == level).filter((pl.col(feature_type) != pl.col(feature_type).shift(1)).over([\"session_id\"])).groupby(\"session_id\").agg([\n                pl.col(\"text_fqid\").count().alias(f\"consecutive_unique_{feature_type}_level{level}_counts\")\n            ])\n            consecutive_data = consecutive_data.join(tmp_consecutive_unique_data, on=\"session_id\", how='left')\n    \n    # Only keep the feature transformation that are in the feature list\n    aggs = [\n        agg\n        for agg in aggs\n        if agg.meta.output_name() in union_features[grp]\n    ]\n    df = df.join(consecutive_data, on=\"session_id\", how='left')\n    \n    if use_extra:\n        if grp == '0-4':\n            aggs = [\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Now where did I put my notebook?\") | (pl.col(\"text\") == \"Found it!\")).apply(lambda s: s.max() - s.min()).alias(\"find_notebook_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Now where did I put my notebook?\") | (pl.col(\"text\") == \"Found it!\")).apply(lambda s: s.max() - s.min()).alias(\"find_notebook_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Found it!\") | (pl.col(\"text\") == \"Let's get started. The Wisconsin Wonders exhibit opens tomorrow!\")).apply(lambda s: s.max() - s.min()).alias(\"go_upstairs_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Found it!\") | (pl.col(\"text\") == \"Let's get started. The Wisconsin Wonders exhibit opens tomorrow!\")).apply(lambda s: s.max() - s.min()).alias(\"go_upstairs_events\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Hey Jo, let's take a look at the shirt!\") | (pl.col(\"text\") == \"This looks like a clue!\")).apply(lambda s: s.max() - s.min()).alias(\"find_jersey_clue_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Hey Jo, let's take a look at the shirt!\") | (pl.col(\"text\") == \"This looks like a clue!\")).apply(lambda s: s.max() - s.min()).alias(\"find_jersey_clue_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"I'll be at the Capitol. Let me know if you find anything!\") | (pl.col(\"text\") == \"Our shirt is too old to be a basketball jersey!\")).apply(lambda s: s.max() - s.min()).alias(\"find_jersey_old_label_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"I'll be at the Capitol. Let me know if you find anything!\") | (pl.col(\"text\") == \"Our shirt is too old to be a basketball jersey!\")).apply(lambda s: s.max() - s.min()).alias(\"find_jersey_old_label_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"I need to get to the Capitol and tell Gramps!\") | (pl.col(\"fqid\") ==\"chap1_finale\")).apply(lambda s: s.max() - s.min()).alias(\"to_chap1_final_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"I need to get to the Capitol and tell Gramps!\") | (pl.col(\"fqid\") == \"chap1_finale\")).apply(lambda s: s.max() - s.min()).alias(\"to_chap1_final_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"fqid\") ==\"chap1_finale\") | (pl.col(\"fqid\") ==\"chap1_finale_c\")).apply(lambda s: s.max() - s.min()).alias(\"chap1_answer_duration\"),\n                pl.col(\"index\").filter((pl.col(\"fqid\") ==\"chap1_finale\") | (pl.col(\"fqid\") == \"chap1_finale_c\")).apply(lambda s: s.max() - s.min()).alias(\"chap1_answer_indexCount\"),\n            ]\n           # Only keep the feature transformation that are in the feature list\n            aggs = [\n                agg\n                for agg in aggs\n                if agg.meta.output_name() in union_features[grp]\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\") \n            df = df.join(tmp, on=\"session_id\", how='left')\n        if grp=='5-12':\n            aggs = [\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Here's the log book.\") | (pl.col(\"fqid\") == 'logbook.page.bingo')).apply(lambda s: s.max() - s.min()).alias(\"logbook_bingo_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Here's the log book.\") | (pl.col(\"fqid\") == 'logbook.page.bingo')).apply(lambda s: s.max() - s.min()).alias(\"logbook_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader')) | (pl.col(\"fqid\") == \"reader.paper2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\"reader_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader')) | (pl.col(\"fqid\") == \"reader.paper2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\"reader_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals')) | (pl.col(\"fqid\") == \"journals.pic_2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\"journals_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals')) | (pl.col(\"fqid\") == \"journals.pic_2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\"journals_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"I got here and the whole place was a mess!\") | (pl.col(\"text\") == \"He's our expert record keeper.\")).apply(lambda s: s.max() - s.min()).alias(\"tidy_up_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"I got here and the whole place was a mess!\") | (pl.col(\"text\") == \"He's our expert record keeper.\")).apply(lambda s: s.max() - s.min()).alias(\"tidy_up_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"He's our expert record keeper.\") | (pl.col(\"text\") == \"I need your help!\")).apply(lambda s: s.max() - s.min()).alias(\"find_archivist_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"He's our expert record keeper.\") | (pl.col(\"text\") == \"I need your help!\")).apply(lambda s: s.max() - s.min()).alias(\"find_archivist_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Now if only I could read this thing.\") | (pl.col(\"text\") == \"I bet the archivist could use this!\")).apply(lambda s: s.max() - s.min()).alias(\"find_glasses_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Now if only I could read this thing.\") | (pl.col(\"text\") == \"I bet the archivist could use this!\")).apply(lambda s: s.max() - s.min()).alias(\"find_glasses_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Great! Thanks for the help!\") | (pl.col(\"text\") == \"Hello there!\")).apply(lambda s: s.max() - s.min()).alias(\"find_textile_expert_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Great! Thanks for the help!\") | (pl.col(\"text\") == \"Hello there!\")).apply(lambda s: s.max() - s.min()).alias(\"find_textile_expert_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"This place was around in 1916! I can start there!\") | (pl.col(\"text\") == \"Hi! How can I help you?\")).apply(lambda s: s.max() - s.min()).alias(\"find_worker_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"This place was around in 1916! I can start there!\") | (pl.col(\"text\") == \"Hi! How can I help you?\")).apply(lambda s: s.max() - s.min()).alias(\"find_worker_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"He re's the log book.\") | (pl.col(\"text\") == \"It's a match!\")).apply(lambda s: s.max() - s.min()).alias(\"search_logbook_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"He re's the log book.\") | (pl.col(\"text\") == \"It's a match!\")).apply(lambda s: s.max() - s.min()).alias(\"search_logbook_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Thanks for the help!\") | (pl.col(\"text\") == \"Oh, hello there!\")).apply(lambda s: s.max() - s.min()).alias(\"go_to_library_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Thanks for the help!\") | (pl.col(\"text\") == \"Oh, hello there!\")).apply(lambda s: s.max() - s.min()).alias(\"go_to_library_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Check out our microfiche. It's right through that door.\") | (pl.col(\"text\") == \"Youmans was a suffragist!\")).apply(lambda s: s.max() - s.min()).alias(\"check_microfiche_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Check out our microfiche. It's right through that door.\") | (pl.col(\"text\") == \"Youmans was a suffragist!\")).apply(lambda s: s.max() - s.min()).alias(\"check_microfiche_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"She helped get votes for women!\") | (pl.col(\"text\") == \"What was Wells doing here?\")).apply(lambda s: s.max() - s.min()).alias(\"pick_up_wells_card_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"She helped get votes for women!\") | (pl.col(\"text\") == \"What was Wells doing here?\")).apply(lambda s: s.max() - s.min()).alias(\"pick_up_wells_card_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"You could ask the archivist. He knows everybody!\") | (pl.col(\"text\") == \"Can you help me? I need to find Wells!\")).apply(lambda s: s.max() - s.min()).alias(\"find_archivist_2_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"You could ask the archivist. He knows everybody!\") | (pl.col(\"text\") == \"Can you help me? I need to find Wells!\")).apply(lambda s: s.max() - s.min()).alias(\"find_archivist_2_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Right outside the door.\") | (pl.col(\"fqid\") == \"journals.pic_2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\"find_journals_pic_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Right outside the door.\") | (pl.col(\"fqid\") == \"journals.pic_2.bingo\")).apply(lambda s: s.max() - s.min()).alias(\"find_journals_pic_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"I should go to the Capitol and tell everyone!\") | (pl.col(\"fqid\") == \"chap2_finale_c\")).apply(lambda s: s.max() - s.min()).alias(\"go_to_capital_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"I should go to the Capitol and tell everyone!\") | (pl.col(\"fqid\") == \"chap2_finale_c\")).apply(lambda s: s.max() - s.min()).alias(\"go_to_capital_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"fqid\") == \"chap2_finale_c\") | (pl.col(\"event_name\") == \"checkpoint\")).apply(lambda s: s.max() - s.min()).alias(\"chap2_answer_duration\"),\n                pl.col(\"index\").filter((pl.col(\"fqid\") == \"chap2_finale_c\") | (pl.col(\"event_name\") == \"checkpoint\")).apply(lambda s: s.max() - s.min()).alias(\"chap2_answer_indexCount\"),\n                (pl.col(\"elapsed_time\").filter(pl.col(\"level_group\") == \"5-12\").min() - pl.col(\"elapsed_time\").filter(pl.col(\"level_group\") == \"0-4\").max()).alias(\"chap1_answer_time\")\n            ]\n            # Only keep the feature transformation that are in the feature list\n            aggs = [\n                agg\n                for agg in aggs\n                if agg.meta.output_name() in union_features[grp]\n            ]\n            \n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n\n        if grp=='13-22':\n            aggs = [\n                pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader_flag')) | (pl.col(\"fqid\") == \"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"reader_flag_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'reader_flag')) | (pl.col(\"fqid\") == \"tunic.library.microfiche.reader_flag.paper2.bingo\")).apply(lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"reader_flag_indexCount\"),\n                pl.col(\"elapsed_time\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals_flag')) | (pl.col(\"fqid\") == \"journals_flag.pic_0.bingo\")).apply(lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"journalsFlag_bingo_duration\"),\n                pl.col(\"index\").filter(((pl.col(\"event_name\") == 'navigate_click') & (pl.col(\"fqid\") == 'journals_flag')) | (pl.col(\"fqid\") == \"journals_flag.pic_0.bingo\")).apply(lambda s: s.max() - s.min() if s.len() > 0 else 0).alias(\"journalsFlag_bingo_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"I'll go look at everyone's pictures!\") | (pl.col(\"text\") == \"Those are the same glasses!\")).apply(lambda s: s.max() - s.min()).alias(\"find_glasses_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"I'll go look at everyone's pictures!\") | (pl.col(\"text\") == \"Those are the same glasses!\")).apply(lambda s: s.max() - s.min()).alias(\"find_glasses_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"The archivist must've taken Teddy!\") | (pl.col(\"text\") == \"Yes! It's the key for Teddy's cage!\")).apply(lambda s: s.max() - s.min()).alias(\"find_key_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"The archivist must've taken Teddy!\") | (pl.col(\"text\") == \"Yes! It's the key for Teddy's cage!\")).apply(lambda s: s.max() - s.min()).alias(\"find_key_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Yes! It's the key for Teddy's cage!\") | (pl.col(\"text\") == \"I found the key!\")).apply(lambda s: s.max() - s.min()).alias(\"unlock_teddy_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Yes! It's the key for Teddy's cage!\") | (pl.col(\"text\") == \"I found the key!\")).apply(lambda s: s.max() - s.min()).alias(\"unlock_teddy_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Come on, Teddy. Let's go help Gramps!\") | (pl.col(\"text\") == \"Teddy! I'm glad to see you.\")).apply(lambda s: s.max() - s.min()).alias(\"go_to_collection_room_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Come on, Teddy. Let's go help Gramps!\") | (pl.col(\"text\") == \"Teddy! I'm glad to see you.\")).apply(lambda s: s.max() - s.min()).alias(\"go_to_collection_room_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"I'll ride with you!\") | (pl.col(\"text\") == \"Oh no! What happened to that crane?\")).apply(lambda s: s.max() - s.min()).alias(\"go_to_wildlife_center_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"I'll ride with you!\") | (pl.col(\"text\") == \"Oh no! What happened to that crane?\")).apply(lambda s: s.max() - s.min()).alias(\"go_to_wildlife_center_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"That hoofprint doesn't match the flag!\") | (pl.col(\"text\") == \"Hey, nice dog! What breed is he?\")).apply(lambda s: s.max() - s.min()).alias(\"meet_flag_expert_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"That hoofprint doesn't match the flag!\") | (pl.col(\"text\") == \"Hey, nice dog! What breed is he?\")).apply(lambda s: s.max() - s.min()).alias(\"meet_flag_expert_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Hey, nice dog! What breed is he?\") | (pl.col(\"text\") == \"Welcome back, Dear! How can I help you?\")).apply(lambda s: s.max() - s.min()).alias(\"go_to_library_2_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Hey, nice dog! What breed is he?\") | (pl.col(\"text\") == \"Welcome back, Dear! How can I help you?\")).apply(lambda s: s.max() - s.min()).alias(\"go_to_library_2_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Go check the microfiche. Maybe you'll find something!\") | (pl.col(\"text\") == \"Hey! That's Governor Nelson in front of our flag!\")).apply(lambda s: s.max() - s.min()).alias(\"check_microfiche_2_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Go check the microfiche. Maybe you'll find something!\") | (pl.col(\"text\") == \"Hey! That's Governor Nelson in front of our flag!\")).apply(lambda s: s.max() - s.min()).alias(\"check_microfiche_2_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Okay. Thanks!\") | (pl.col(\"text\") == \"It's for the flag display!\")).apply(lambda s: s.max() - s.min()).alias(\"find_archivist_3_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Okay. Thanks!\") | (pl.col(\"text\") == \"It's for the flag display!\")).apply(lambda s: s.max() - s.min()).alias(\"find_archivist_3_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"Here's a call number for the Stacks. Go find some photos.\") | (pl.col(\"text\") == \"Look at all those activists!\")).apply(lambda s: s.max() - s.min()).alias(\"find_photos_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"Here's a call number for the Stacks. Go find some photos.\") | (pl.col(\"text\") == \"Look at all those activists!\")).apply(lambda s: s.max() - s.min()).alias(\"find_photos_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"text\") == \"I should go to the Capitol and tell Mrs. M!\") | (pl.col(\"fqid\") == \"chap4_finale_c\")).apply(lambda s: s.max() - s.min()).alias(\"go_to_capital_final_duration\"),\n                pl.col(\"index\").filter((pl.col(\"text\") == \"I should go to the Capitol and tell Mrs. M!\") | (pl.col(\"fqid\") == \"chap4_finale_c\")).apply(lambda s: s.max() - s.min()).alias(\"go_to_capital_final_indexCount\"),\n                pl.col(\"elapsed_time\").filter((pl.col(\"fqid\") == \"chap4_finale_c\") | (pl.col(\"event_name\") == \"checkpoint\")).apply(lambda s: s.max() - s.min()).alias(\"chap4_answer_duration\"),\n                pl.col(\"index\").filter((pl.col(\"fqid\") == \"chap4_finale_c\") | (pl.col(\"event_name\") == \"checkpoint\")).apply(lambda s: s.max() - s.min()).alias(\"chap4_answer_indexCount\"),\n                (pl.col(\"elapsed_time\").filter(pl.col(\"level_group\") == \"13-22\").min() - pl.col(\"elapsed_time\").filter(pl.col(\"level_group\") == \"5-12\").max()).alias(\"chap2_answer_time\")\n            ]\n            \n            # Only keep the feature transformation that are in the feature list\n            aggs = [\n                agg\n                for agg in aggs\n                if agg.meta.output_name() in union_features[grp]\n            ]\n            tmp = x.groupby([\"session_id\"], maintain_order=True).agg(aggs).sort(\"session_id\")\n            df = df.join(tmp, on=\"session_id\", how='left')\n        \n    return df.to_pandas()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:07.530997Z","iopub.execute_input":"2023-06-28T10:52:07.531428Z","iopub.status.idle":"2023-06-28T10:52:07.700920Z","shell.execute_reply.started":"2023-06-28T10:52:07.531396Z","shell.execute_reply":"2023-06-28T10:52:07.699537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import jo_wilder_310\nenv = jo_wilder_310.make_env()\niter_test = env.iter_test()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:07.702722Z","iopub.execute_input":"2023-06-28T10:52:07.703152Z","iopub.status.idle":"2023-06-28T10:52:07.714212Z","shell.execute_reply.started":"2023-06-28T10:52:07.703119Z","shell.execute_reply":"2023-06-28T10:52:07.713285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pickle\n    \nfilename = \"/kaggle/input/blend-stdgrp-std18in1/\"\nloaded_model_GBT = pickle.load(open(filename+\"blend_GBT.bin\", 'rb'))\nloaded_model_NN = pickle.load(open(filename+\"blend_NN.bin\", 'rb'))\n\ncached_dfs = {}\nlast_et = {}\n\nlimits = {'0-4':(1,4), '5-12':(4,14), '13-22':(14,19)}\nfor (test, sample_submission) in iter_test:\n    # Debug column type\n#     for f in test.columns:\n#         print(f, test[f].dtype)\n    \n    ######## Pandas ##########\n    sub_cols = sample_submission.columns\n    sample_submission['q'] = sample_submission['session_id'].str.split('_').str[1].str[1:].astype(int)\n    sample_submission = sample_submission.sort_values('q')\n    sample_submission = sample_submission[sub_cols]\n    \n    \n    ###### NM inference ######\n    test_NN = test.copy().sort_values(['elapsed_time','index']).reset_index(drop=True)\n    \n    sid = test_NN['session_id'].iloc[0]\n    cur_lv_gp = test_NN[\"level_group\"].iloc[0]\n    first_et = test_NN[\"elapsed_time\"].iloc[0]\n    test_NN[\"et_diff\"] = test_NN[\"elapsed_time\"].diff()\n    if sid in last_et:\n        test_NN[\"et_diff\"].iloc[0] = first_et - last_et[sid]\n    last_et[sid] = test_NN[\"elapsed_time\"].iloc[-1]\n    test_NN[\"et_diff\"] = test_NN[\"et_diff\"].fillna(0).clip(0, et_diff_upper_bound)\n    test_NN[\"et_diff\"] = np.log1p(test_NN[\"et_diff\"])\n    test_NN[\"event_comb\"] = test_NN[\"event_name\"] + \"_\" + test_NN[\"name\"]\n    for cat_feat in CAT_FEATS:\n        orig_col = cat_feat[:-5]\n        test_NN[orig_col] = test_NN[orig_col].astype(str).fillna('None')\n        test_NN[cat_feat] = test_NN[orig_col].map(cat2code_[cat_feat]).fillna(-1)\n    \n    x = test_NN[\"et_diff\"].values[-t_window:]\n    x_cat = test_NN[CAT_FEATS].values[-t_window:]\n    if len(x) < t_window:\n        pad_len = t_window - len(x)\n        x = np.pad(x, (pad_len, 0), \"constant\")\n        x_cat = np.pad(x_cat, ((pad_len, 0), (0, 0)), \"constant\", constant_values=-1)  # (P, #cat_feats)\n    x = torch.tensor(x, dtype=torch.float32).unsqueeze(dim=0).unsqueeze(dim=-1)   # Add B, C dim, (1, P, 1)\n    x_cat = torch.tensor(x_cat, dtype=torch.int64).unsqueeze(dim=0)   # (1, P, C)\n        \n    # Run quick inference\n    y_pred_NN = []\n    y_pred_NN.append(quick_infer((x, x_cat), models1[cur_lv_gp]))\n    y_pred_NN.append(quick_infer((x, x_cat), models2[cur_lv_gp]))\n    y_pred_NN.append(quick_infer2((x, x_cat), models3, cur_lv_gp))\n    y_pred_NN = np.stack(y_pred_NN,axis=-1)\n    ###### END NN ######\n    \n    \n    grp = test.level_group.values[0]\n    session_id = test.session_id.values[0]\n    \n    cached_dfs[grp] = test.copy()\n    \n    if grp == '0-4':\n        pd_df = cached_dfs['0-4']\n        assert '5-12' not in cached_dfs\n        assert '13-22' not in cached_dfs\n    elif grp == '5-12':\n        assert '13-22' not in cached_dfs\n        pd_df = pd.concat([cached_dfs['0-4'], cached_dfs['5-12']])\n    else:\n        pd_df = pd.concat([cached_dfs['0-4'], cached_dfs['5-12'], cached_dfs['13-22']])\n        # Reset\n        cached_dfs = {}\n    \n    ######## Polars ##########\n    df = (pl.from_pandas(pd_df)\n          .drop([\"fullscreen\", \"hq\", \"music\"])\n          .sort(\"elapsed_time\")\n          .with_columns(columns).with_columns(columns_1).with_columns(columns_2))\n        \n    # Make sure only 1 session in the df\n    assert df.select(pl.col(\"session_id\").n_unique()).item() == 1\n    \n    df = feature_engineer(df, grp, use_extra=True, feature_suffix='')\n    df = time_feature(df)\n    \n    \n    # Make sure only 1 session in the df\n    assert df.shape[0] == 1\n    \n    for f in std_catb_18in1_features:\n        if f not in df.columns:\n            df[f] = np.nan\n    \n    a, b = limits[grp]\n    # Repeat the row b-a times\n    df = df.loc[df.index.repeat(b-a)]\n    df[\"q\"] = list(range(a, b))\n    \n    df[\"std_catb_grp\"] = 0\n    df[\"std_catb_18_in_1\"] = 0\n    df[\"std_xgb_18_in_1\"] = 0\n    df[\"std_xgb_18_in_1_2\"] = 0\n    \n    for fold in range(5):        \n        model = std_catb_grp_models[grp][fold]\n        _preds = model.predict_proba(df[std_catb_grp_features[grp]])[:, 1]\n        df[\"std_catb_grp\"] += _preds / 5.0\n        \n        model = std_catb_18in1_models[fold]\n        _preds = model.predict_proba(df[std_catb_18in1_features])[:, 1]\n        df[\"std_catb_18_in_1\"] += _preds / 5.0\n        \n        model = std_xgb_18in1_models[fold]\n        _preds = model.predict(xgb.DMatrix(df[std_xgb_18in1_features]))\n        df[\"std_xgb_18_in_1\"] += _preds / 5.0\n        \n        model = std_xgb_18in1_2_models[fold]\n        _preds = model.predict(xgb.DMatrix(df[std_xgb_18in1_2_features]))\n        df[\"std_xgb_18_in_1_2\"] += _preds / 5.0\n    \n#     preds = loaded_model.predict_proba(df[[\"std_catb_grp\", \"std_catb_18_in_1\", \"std_xgb_18_in_1\", \"std_xgb_18_in_1_2\"]].values)[:,1]\n    \n    preds_GBT = loaded_model_GBT.predict_proba(df[[\"std_catb_grp\",\"std_catb_18_in_1\",\"std_xgb_18_in_1\",\"std_xgb_18_in_1_2\"]].values)[:,1]\n    preds_NN = loaded_model_NN.predict_proba(y_pred_NN)[:,1]\n    preds = 0.6*preds_GBT+0.4*preds_NN\n    \n    ## ALL months\n    # 0.654668797257206  0.7045700773750883\n    # 0.6511487416987165 0.7044170716465112\n    \n    ## After 2021 Dec\n    # 0.654668797257206  0.7074751186994148\n    # 0.6511487416987165 0.7071420621435605\n    \n    preds = (preds > 0.654668797257206).astype(int)\n    assert len(preds) == sample_submission.shape[0] \n    sample_submission[\"correct\"] = preds\n    env.predict(sample_submission)","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:07.716551Z","iopub.execute_input":"2023-06-28T10:52:07.717648Z","iopub.status.idle":"2023-06-28T10:52:34.946525Z","shell.execute_reply.started":"2023-06-28T10:52:07.717599Z","shell.execute_reply":"2023-06-28T10:52:34.945224Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv('submission.csv')\nprint(sub.shape, sub.correct.mean())\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-06-28T10:52:34.948165Z","iopub.execute_input":"2023-06-28T10:52:34.948739Z","iopub.status.idle":"2023-06-28T10:52:34.977814Z","shell.execute_reply.started":"2023-06-28T10:52:34.948707Z","shell.execute_reply":"2023-06-28T10:52:34.976927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}