{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":45533,"databundleVersionId":5748852,"sourceType":"competition"},{"sourceId":7070999,"sourceType":"datasetVersion","datasetId":4072039},{"sourceId":7088177,"sourceType":"datasetVersion","datasetId":4076151},{"sourceId":7139387,"sourceType":"datasetVersion","datasetId":4120338},{"sourceId":7173986,"sourceType":"datasetVersion","datasetId":4145344},{"sourceId":7173970,"sourceType":"datasetVersion","datasetId":4145329}],"dockerImageVersionId":30588,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nimport json\nimport pandas as pd\nimport torch.nn as nn\nfrom torch.optim import AdamW\nimport torch\nfrom tqdm.notebook import tqdm\nfrom sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, classification_report, ConfusionMatrixDisplay, confusion_matrix, roc_curve, auc\nimport numpy as np\nfrom sklearn.preprocessing import LabelEncoder\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-11T08:34:32.937948Z","iopub.execute_input":"2023-12-11T08:34:32.938696Z","iopub.status.idle":"2023-12-11T08:34:32.945232Z","shell.execute_reply.started":"2023-12-11T08:34:32.938662Z","shell.execute_reply":"2023-12-11T08:34:32.944204Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta = pd.read_csv(\"/kaggle/input/predict-student-performance-from-game-play/train.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:34:33.663159Z","iopub.execute_input":"2023-12-11T08:34:33.663537Z","iopub.status.idle":"2023-12-11T08:36:27.013501Z","shell.execute_reply.started":"2023-12-11T08:34:33.663504Z","shell.execute_reply":"2023-12-11T08:36:27.012397Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta['event_name_name'] = meta['event_name'] + \" \" + meta['name']\nmeta.text   = meta.text.fillna(\" \")\nmeta.fqid   = meta.fqid.fillna(\" \")\nmeta.text_fqid   = meta.text_fqid.fillna(\" \")","metadata":{"execution":{"iopub.status.busy":"2023-12-11T07:19:14.308543Z","iopub.execute_input":"2023-12-11T07:19:14.308853Z","iopub.status.idle":"2023-12-11T07:19:28.451766Z","shell.execute_reply.started":"2023-12-11T07:19:14.308828Z","shell.execute_reply":"2023-12-11T07:19:28.450877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta.room_fqid   = meta.room_fqid.fillna(\" \")","metadata":{"execution":{"iopub.status.busy":"2023-12-11T07:19:28.452891Z","iopub.execute_input":"2023-12-11T07:19:28.453206Z","iopub.status.idle":"2023-12-11T07:19:31.459015Z","shell.execute_reply.started":"2023-12-11T07:19:28.453179Z","shell.execute_reply":"2023-12-11T07:19:31.458003Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encode_json = {}\ndef encode_data(data, name):\n    label_encoder = LabelEncoder()\n    encoded_labels = label_encoder.fit_transform(data)\n    encoded_labels = [str(s) for s in encoded_labels]\n    label_dict = dict(zip(data, encoded_labels))\n    encode_json[name] = label_dict","metadata":{"execution":{"iopub.status.busy":"2023-12-11T07:19:31.461080Z","iopub.execute_input":"2023-12-11T07:19:31.461395Z","iopub.status.idle":"2023-12-11T07:19:31.466805Z","shell.execute_reply.started":"2023-12-11T07:19:31.461352Z","shell.execute_reply":"2023-12-11T07:19:31.465795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encode_data(meta['event_name_name'] , \"event_name_name\")\nencode_data(meta.text , \"text\")\nencode_data(meta.fqid, \"fqid\" )\nencode_data(meta.room_fqid , \"room_fqid\")","metadata":{"execution":{"iopub.status.busy":"2023-12-11T07:19:31.467876Z","iopub.execute_input":"2023-12-11T07:19:31.468187Z","iopub.status.idle":"2023-12-11T07:22:06.730929Z","shell.execute_reply.started":"2023-12-11T07:19:31.468163Z","shell.execute_reply":"2023-12-11T07:22:06.730102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nwith open(\"encode_data.json\", \"w\", encoding=\"utf-8\") as json_file:\n    json.dump(encode_json, json_file, ensure_ascii=False, indent=4)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T07:22:11.470962Z","iopub.execute_input":"2023-12-11T07:22:11.471712Z","iopub.status.idle":"2023-12-11T07:22:11.479133Z","shell.execute_reply.started":"2023-12-11T07:22:11.471676Z","shell.execute_reply":"2023-12-11T07:22:11.478321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"encode_json","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"meta[meta.session_id == 20090312431273200].to_csv('test.csv')","metadata":{"execution":{"iopub.status.busy":"2023-12-11T07:55:09.692590Z","iopub.execute_input":"2023-12-11T07:55:09.692888Z","iopub.status.idle":"2023-12-11T07:55:09.739695Z","shell.execute_reply.started":"2023-12-11T07:55:09.692862Z","shell.execute_reply":"2023-12-11T07:55:09.738766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ConvBlock(nn.Module):\n    def __init__(self, d_model, dropout_rate):\n        super(ConvBlock, self).__init__()\n        self.conv1d = nn.Conv1d(in_channels=d_model, out_channels=d_model, kernel_size=5, padding=2)\n        self.gelu = nn.GELU(\"tanh\")\n        self.layer_norm = nn.LayerNorm(d_model)\n        self.dropout = nn.Dropout(p=dropout_rate)\n        \n    def forward(self, inputs):\n        x = self.conv1d(inputs)\n        x = self.gelu(x)\n        x = x + inputs\n        x = self.layer_norm(x.transpose(1, 2)).transpose(1, 2)\n        \n        outputs = self.dropout(x)\n        return outputs\n\nclass TimeEmbedding(nn.Module):\n    def __init__(self, n_blocks, d_model, dropout_rate):\n        super(TimeEmbedding, self).__init__()\n        self.conv_blocks = nn.ModuleList([ConvBlock(d_model, dropout_rate=dropout_rate) for _ in range(n_blocks)])\n        self.d_model = d_model\n    \n    def forward(self, inputs):\n        x = inputs.view(-1, 1).unsqueeze(-1)\n        b, r, c = x.shape\n        x = x.expand(b, self.d_model , c)\n        for conv_block in self.conv_blocks:\n\n            x = conv_block(x)\n\n        return x","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:16:33.023156Z","iopub.execute_input":"2023-12-11T08:16:33.023547Z","iopub.status.idle":"2023-12-11T08:16:33.033842Z","shell.execute_reply.started":"2023-12-11T08:16:33.023516Z","shell.execute_reply":"2023-12-11T08:16:33.032867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ConvNet(nn.Module):\n    def __init__(self, input_dims, d_model, n_blocks=4):\n        super(ConvNet, self).__init__()\n        self.input_dims = input_dims\n        self.d_model = d_model\n        self.n_blocks = n_blocks\n        self.event_embedding = nn.Embedding(input_dims['event_name_name'], d_model, padding_idx=0)\n        self.room_embedding = nn.Embedding(input_dims['room_fqid'], d_model, padding_idx=0)\n        self.text_embedding = nn.Embedding(input_dims['text'], d_model, padding_idx=0)\n        self.fqid_embedding = nn.Embedding(input_dims['fqid'], d_model, padding_idx=0)\n        self.duration_embedding = TimeEmbedding(n_blocks=n_blocks, d_model=d_model, dropout_rate=0.2)\n        self.gap = nn.AdaptiveAvgPool1d(1)\n        encoder_layer = nn.TransformerEncoderLayer(\n            d_model=24,\n            nhead=8,\n            dim_feedforward=24,\n            dropout=0.1,\n            batch_first=True,\n            activation=\"relu\",\n        )\n        self.encoder = nn.TransformerEncoder(encoder_layer, num_layers=1)\n    def forward(self, inputs):\n        inputs['event_name_name'] = inputs['event_name_name'].long()\n        inputs['room_fqid'] = inputs['room_fqid'].long()\n        inputs['text'] = inputs['text'].long()\n        inputs['fqid'] = inputs['fqid'].long()\n        inputs['duration'] = inputs['duration'].to(torch.float32)\n        event = self.event_embedding(inputs['event_name_name'])\n        room = self.room_embedding(inputs['room_fqid'])\n        text = self.text_embedding(inputs['text'])\n        fqid = self.fqid_embedding(inputs['fqid'])\n        \n        inputs['duration'] = inputs['duration'].unsqueeze(0).unsqueeze(0)\n        duration = self.duration_embedding(inputs['duration']).squeeze(-1)\n        sum_cag = (event + room + text + fqid)\n\n        x = sum_cag*duration.unsqueeze(1).expand(sum_cag.shape)\n        x = self.encoder(x)\n        outputs = self.gap(x)\n        return outputs.transpose(2, 0)\n    \nclass SimpleHead(nn.Module):\n    def __init__(self, n_units, n_outputs, dropout_rate=0.2):\n        super(SimpleHead, self).__init__()\n        self.ffs = nn.ModuleList([nn.Linear(n_units[i-1], n_units[i]) for i in range(1, len(n_units))])\n        self.out = nn.Linear(n_units[-1], n_outputs)\n        self.dropout = nn.Dropout(p=dropout_rate)\n\n    def forward(self, inputs):\n        x = inputs.transpose(2, 1)\n        # print(x.shape)\n        for ff in self.ffs:\n            x = F.leaky_relu(ff(x))\n            x = self.dropout(x)\n        outputs = torch.sigmoid(self.out(x))\n        return outputs\n    ","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:16:33.585459Z","iopub.execute_input":"2023-12-11T08:16:33.585803Z","iopub.status.idle":"2023-12-11T08:16:33.603106Z","shell.execute_reply.started":"2023-12-11T08:16:33.585776Z","shell.execute_reply":"2023-12-11T08:16:33.601682Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class PreTrainingModel(nn.Module):\n    def __init__(self, convnet, head):\n        super(PreTrainingModel, self).__init__()\n        self.convnet = convnet\n        self.head = head\n        \n    def forward(self, inputs: dict):\n        x = self.convnet(inputs)\n#         print(x.shape)\n        outputs = self.head(x)\n        return outputs","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:16:33.763590Z","iopub.execute_input":"2023-12-11T08:16:33.764030Z","iopub.status.idle":"2023-12-11T08:16:33.771819Z","shell.execute_reply.started":"2023-12-11T08:16:33.763983Z","shell.execute_reply":"2023-12-11T08:16:33.770380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_questions(level_group):\n    return (['q1', 'q2', 'q3'] if level_group == '0-4' \n        else ['q4', 'q5', 'q6', 'q7', 'q8', 'q9', 'q10', 'q11', 'q12', 'q13'] if level_group == '5-12' \n        else ['q14', 'q15', 'q16', 'q17', 'q18'])\nclass Data(Dataset):\n    def __init__(self, level, path_data,path_label,  max_length = 600):\n        self.level = level\n        self.data = pd.read_csv(path_data).reset_index()\n        self.data_label = pd.read_csv(path_label).reset_index()\n        self.question = get_questions(level)\n        self.max_length = max_length\n    \n    def __len__(self):\n        return len(self.data)\n    \n    def get_label(self, session):\n        \n        sample = self.data_label[self.data_label.session_id == session]\n        label_question = []\n        for i in self.question:\n            label_question.append(sample[sample.question == i].correct.values[0])\n        return label_question\n    \n    def __getitem__(self, index):\n\n        row = self.data.iloc[index]\n        \n        session = row.session_id\n    \n        file_path = f\"/kaggle/input/student-performance-from-game-play/data/processed_data/{self.level}/{session}.json\"\n\n        with open(file_path, 'r') as json_file:\n            data = json.load(json_file)\n              \n        label  = self.get_label(session)\n        \n        if len(data['event_name_name']) > self.max_length:\n            for i in ['event_name_name', 'room_fqid', 'text', 'fqid']:\n                data[i] = data[i][-self.max_length: ]\n                \n        elif len(data['event_name_name']) < self.max_length:\n            # missing data\n            nb_miss = self.max_length -  len(data['event_name_name']) \n            ls_miss = [-1 for i in range(nb_miss)]\n\n            for i in ['event_name_name', 'room_fqid', 'text', 'fqid']:\n                data[i] = ls_miss + data[i]\n    \n        for i in range(len(data['event_name_name'])):\n            data['event_name_name'][i] +=1\n            data['room_fqid'][i] +=1\n            data['text'][i] +=1\n            data['fqid'][i] +=1\n        return {\n            \n                'event_name_name': torch.tensor(data['event_name_name']) ,\n                'room_fqid': torch.tensor(data['room_fqid']) , \n                'text': torch.tensor(data['text']) , \n                'fqid': torch.tensor(data['fqid']) , \n                'duration': torch.tensor(data['duration']), \n                'label': torch.tensor(label), \n                \"session_id\": torch.tensor(session)\n        }","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:16:33.934614Z","iopub.execute_input":"2023-12-11T08:16:33.934945Z","iopub.status.idle":"2023-12-11T08:16:33.949661Z","shell.execute_reply.started":"2023-12-11T08:16:33.934913Z","shell.execute_reply":"2023-12-11T08:16:33.948735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Config\npath_label_train = \"/kaggle/input/student-performance-from-game-play/data/processed_data/train.csv\"\npath_label_test = \"/kaggle/input/student-performance-from-game-play/data/processed_data/test.csv\"\n\npath_data_train = \"/kaggle/working/train.csv\"\npath_data_test = \"/kaggle/working/test.csv\"\n\nD_MODEL = 24\ninput_dims = {\n    \"event_name_name\": 20, \n    \"room_fqid\": 20, \n    \"text\": 599, \n    \"fqid\": 130\n}\n\ntrain = pd.read_csv(path_label_train)\ntrain.drop_duplicates(\"session_id\", inplace = True)\ntrain[['session_id']].to_csv(\"train.csv\")\n\ntest = pd.read_csv(path_label_test)\ntest.drop_duplicates(\"session_id\", inplace = True)\ntest[['session_id']].to_csv(\"test.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:16:34.211846Z","iopub.execute_input":"2023-12-11T08:16:34.212490Z","iopub.status.idle":"2023-12-11T08:16:34.666059Z","shell.execute_reply.started":"2023-12-11T08:16:34.212454Z","shell.execute_reply":"2023-12-11T08:16:34.665275Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef compute_scores(y_true, y_pred, thres, ps = 0.5):\n    y_pred = [int(i >= thres) for i in y_pred]\n    \n    cnt_pred = {0: 0, 1:0}\n    cnt_true = {0: 0, 1:0}\n    \n    cm_test = confusion_matrix(y_true, y_pred)\n    f1 = f1_score(y_true, y_pred, average='weighted')\n    acc = accuracy_score(y_true, y_pred)\n    \n    for i in y_true: cnt_true[i] +=1 \n    \n#     print(cnt_true)\n    p = (cm_test[0][0]/cnt_true[0]) * ps + (cm_test[1][1]/cnt_true[1])* (1 - ps)\n    \n    return f1, acc, p, y_pred, cm_test[0][0], cm_test[1][1]\n    ","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:16:34.885038Z","iopub.execute_input":"2023-12-11T08:16:34.885889Z","iopub.status.idle":"2023-12-11T08:16:34.892753Z","shell.execute_reply.started":"2023-12-11T08:16:34.885857Z","shell.execute_reply":"2023-12-11T08:16:34.891736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:16:35.755825Z","iopub.execute_input":"2023-12-11T08:16:35.756220Z","iopub.status.idle":"2023-12-11T08:16:35.760646Z","shell.execute_reply.started":"2023-12-11T08:16:35.756187Z","shell.execute_reply":"2023-12-11T08:16:35.759568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"2000/10","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:16:35.946543Z","iopub.execute_input":"2023-12-11T08:16:35.947193Z","iopub.status.idle":"2023-12-11T08:16:35.952956Z","shell.execute_reply.started":"2023-12-11T08:16:35.947163Z","shell.execute_reply":"2023-12-11T08:16:35.951948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef setting_train(level_group, LENGTH):\n    batch_size = 10\n\n    questions = get_questions(level_group)\n    train_dataset =  Data(path_label = path_label_train, path_data = path_data_train, level = level_group, max_length = LENGTH)\n    test_dataset =  Data(path_label = path_label_test, path_data = path_data_test, level = level_group, max_length = LENGTH)\n    \n    epochs = 100\n    test_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=True, num_workers=2)\n    train_loader = DataLoader(train_dataset, batch_size=batch_size, shuffle=True, num_workers=2)\n    device = \"cuda:0\"\n    N_UNIT = [LENGTH, 512, 512, 512]\n    ### Eval\n    convnet = ConvNet( input_dims = input_dims, d_model=D_MODEL, n_blocks = 11).to(device)\n    head = SimpleHead(n_units = N_UNIT, n_outputs = len(questions)).to(device)\n    model = PreTrainingModel(convnet, head).to(device)\n    criterion = nn.BCELoss()\n    optimizer = AdamW(model.parameters(), lr=3e-5)\n    scheduler = torch.optim.lr_scheduler.CyclicLR(optimizer, base_lr=3e-5, max_lr=0.01, cycle_momentum = False)\n    \n#     model.load_state_dict(torch.load(f'/kaggle/input/best-weight-model/best_acc_{level_group}.pth'))\n    model.train()\n    \n    logs = []\n    accumulation_steps = 200\n    \n    best_score = 1000\n    for epoch in range(epochs):\n        total_loss_train = []\n\n        for i, batch in tqdm(enumerate(train_loader), total = len(train_loader)):\n                list_true_label_test = []\n                list_pred_label_test = []\n                for key in batch.keys():\n                    batch[key] = batch[key].to(dtype=torch.float32, device = device)\n                output = model(batch)\n\n\n                output = model(batch)\n\n                loss = criterion(output[0].view(1, -1)[0] ,batch['label'].view(1, -1).float()[0])\n\n                # Save ouput\n                total_loss_train.append(loss.item())\n\n                # backward\n                loss /= accumulation_steps\n                loss.backward()\n#                 total_loss_train.append(loss)\n                if (i+1) % accumulation_steps == 0:\n                    optimizer.step()\n                    optimizer.zero_grad()\n                    scheduler.step()\n\n        total_loss_test = []\n        list_true_label_test = []\n        list_pred_label_test = []\n\n        model.eval()\n        cnt = 0\n        with torch.no_grad():\n            for i, batch in tqdm(enumerate(test_loader), total = len(test_loader)):\n                for key in batch.keys():\n                    batch[key] = batch[key].to(dtype=torch.float32, device = device)\n                output = model(batch)\n                list_true_label_test += batch['label'].view(1, -1).tolist()[0]\n                list_pred_label_test += output[0].view(1, -1).tolist()[0]\n                loss = criterion(output[0].view(1, -1)[0] ,batch['label'].view(1, -1).float()[0])\n                total_loss_test.append(loss.item())\n#         print(total_loss_train)\n        logs.append(\n                {\n                    \"epoch\": epoch, \n                    \"loss_train\": np.mean(total_loss_train),\n                    \"loss_test\": np.mean(total_loss_test),\n\n                }\n                )\n        if np.mean(total_loss_test) < best_score:\n            print(\"Update weight model\")\n            torch.save(model.state_dict(), f'best_acc_{level_group}.pth')\n            best_score = np.mean(total_loss_test)\n            desc = 0\n        else:\n            desc  += 1\n        \n        if desc > 7:\n            print(\"Stop early model\")\n            break\n        print(logs[-1])\n        with open(f'./logs_{level_group}.json', 'w') as json_file:\n            json.dump(logs, json_file, indent=5)\n#     del train_dataset\n    del test_dataset\n#     del train_loader\n    del test_loader\n    del model\n    \n    gc.collect()\n    torch.cuda.empty_cache()\n#     return  {'f1': f1_test, 'acc': acc_test, 'p': p}","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:16:36.622051Z","iopub.execute_input":"2023-12-11T08:16:36.622716Z","iopub.status.idle":"2023-12-11T08:16:36.642025Z","shell.execute_reply.started":"2023-12-11T08:16:36.622683Z","shell.execute_reply":"2023-12-11T08:16:36.641102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# setting_train('0-4', 300)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:16:37.547334Z","iopub.execute_input":"2023-12-11T08:16:37.547708Z","iopub.status.idle":"2023-12-11T08:16:37.551844Z","shell.execute_reply.started":"2023-12-11T08:16:37.547677Z","shell.execute_reply":"2023-12-11T08:16:37.550830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# setting_train('5-12', 1000)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:16:37.724506Z","iopub.execute_input":"2023-12-11T08:16:37.724862Z","iopub.status.idle":"2023-12-11T08:16:37.729197Z","shell.execute_reply.started":"2023-12-11T08:16:37.724833Z","shell.execute_reply":"2023-12-11T08:16:37.728207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# setting_train('13-22', 1500)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:16:38.133861Z","iopub.execute_input":"2023-12-11T08:16:38.134728Z","iopub.status.idle":"2023-12-11T08:16:38.138430Z","shell.execute_reply.started":"2023-12-11T08:16:38.134692Z","shell.execute_reply":"2023-12-11T08:16:38.137493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import gc","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:16:38.466705Z","iopub.execute_input":"2023-12-11T08:16:38.467436Z","iopub.status.idle":"2023-12-11T08:16:38.471374Z","shell.execute_reply.started":"2023-12-11T08:16:38.467403Z","shell.execute_reply":"2023-12-11T08:16:38.470394Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef testing_all(level_group, LENGTH, thres):\n    batch_size = 20\n\n    questions = get_questions(level_group)\n    test_dataset =  Data(path_label = path_label_test, path_data = path_data_test, level = level_group, max_length = LENGTH)\n    \n    epochs = 100\n    test_loader = DataLoader(test_dataset, batch_size=batch_size, shuffle=True, num_workers=2)\n    \n    device = \"cuda:0\"\n    N_UNIT = [LENGTH, 512, 512, 512]\n    ### Eval\n    convnet = ConvNet( input_dims = input_dims, d_model=D_MODEL, n_blocks = 11).to(device)\n    head = SimpleHead(n_units = N_UNIT, n_outputs = len(questions)).to(device)\n    model = PreTrainingModel(convnet, head).to(device)\n    model.load_state_dict(torch.load(f'/kaggle/input/best-weight-model-tranformer/best_acc_{level_group}.pth'))\n    \n    total_loss_test = []\n    list_true_label_test = []\n    list_pred_label_test = []\n\n    model.eval()\n    cnt = 0\n    with torch.no_grad():\n        for i, batch in tqdm(enumerate(test_loader), total = len(test_loader)):\n            for key in batch.keys():\n                batch[key] = batch[key].to(dtype=torch.float32, device = device)\n            output = model(batch)\n            list_true_label_test += batch['label'].view(1, -1).tolist()[0]\n            list_pred_label_test += output[0].view(1, -1).tolist()[0]\n\n     # Compute measure wwith thres opt\n    f1_test, acc_test,p,  y_pred, c0, c1 = compute_scores(y_true = list_true_label_test, y_pred = list_pred_label_test, thres = thres)\n    \n    print(f\"Level group: {level_group} Acc: {acc_test} F1: {f1_test} P: {p} Thres: {thres}\")\n#     del train_dataset\n    del test_dataset\n#     del train_loader\n    del test_loader\n    del model\n    \n    gc.collect()\n    torch.cuda.empty_cache()\n    return  {'f1': f1_test, 'acc': acc_test, 'p': p}","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:18:15.911695Z","iopub.execute_input":"2023-12-11T08:18:15.912098Z","iopub.status.idle":"2023-12-11T08:18:15.923472Z","shell.execute_reply.started":"2023-12-11T08:18:15.912067Z","shell.execute_reply":"2023-12-11T08:18:15.922527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"f1_overall = []\nacc_overall = []\np_overall = []\n\nfor i in np.arange(0.1, 1, 0.1):\n    score04 = testing_all('0-4', 300, i)\n    score512 = testing_all('5-12', 1000, i)\n    score1322 = testing_all('13-22',1500,  i)\n#     s = (score04 + score512 + score1322)/3\n    f1_overall.append((score04['f1'] + score512['f1'] + score1322['f1'])/3)\n    acc_overall.append((score04['acc'] + score512['acc'] + score1322['acc'])/3)\n    p_overall.append((score04['p'] + score512['p'] + score1322['p'])/3)\n    print(f\"Overall score: {f1_overall[-1]} Thres: {i}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:18:16.335721Z","iopub.execute_input":"2023-12-11T08:18:16.336115Z","iopub.status.idle":"2023-12-11T08:27:05.395680Z","shell.execute_reply.started":"2023-12-11T08:18:16.336081Z","shell.execute_reply":"2023-12-11T08:27:05.394596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"en","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"max(f1_overall), max(acc_overall), max(p_overall)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:36:27.015423Z","iopub.execute_input":"2023-12-11T08:36:27.015738Z","iopub.status.idle":"2023-12-11T08:36:27.021942Z","shell.execute_reply.started":"2023-12-11T08:36:27.015712Z","shell.execute_reply":"2023-12-11T08:36:27.021031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Specify the file path\nfile_path = 'f1_overall.json'\n\n# Save the list to a JSON file\nwith open(file_path, 'w') as json_file:\n    json.dump(f1_overall, json_file)\n\nprint(f'The list has been saved to {file_path}')","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:36:27.022887Z","iopub.execute_input":"2023-12-11T08:36:27.023180Z","iopub.status.idle":"2023-12-11T08:36:27.034528Z","shell.execute_reply.started":"2023-12-11T08:36:27.023156Z","shell.execute_reply":"2023-12-11T08:36:27.033692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Specify the file path\nfile_path = 'acc_overall.json'\n\n# Save the list to a JSON file\nwith open(file_path, 'w') as json_file:\n    json.dump(acc_overall, json_file)\n\nprint(f'The list has been saved to {file_path}')","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:36:27.037115Z","iopub.execute_input":"2023-12-11T08:36:27.037381Z","iopub.status.idle":"2023-12-11T08:36:27.044829Z","shell.execute_reply.started":"2023-12-11T08:36:27.037359Z","shell.execute_reply":"2023-12-11T08:36:27.043948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Specify the file path\nfile_path = 'p_overall.json'\n\n# Save the list to a JSON file\nwith open(file_path, 'w') as json_file:\n    json.dump(p_overall, json_file)\n\nprint(f'The list has been saved to {file_path}')","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:36:27.045988Z","iopub.execute_input":"2023-12-11T08:36:27.046343Z","iopub.status.idle":"2023-12-11T08:36:27.054259Z","shell.execute_reply.started":"2023-12-11T08:36:27.046310Z","shell.execute_reply":"2023-12-11T08:36:27.053408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\n\n# Mở tệp tin JSON để đọc\nwith open('/kaggle/input/eval-sol-tranformer-last/logs_0-4.json', 'r') as json_file:\n    # Đọc nội dung từ tệp tin\n    logs04 = json.load(json_file)\nwith open('/kaggle/input/eval-sol-tranformer-last/logs_5-12.json', 'r') as json_file:\n    # Đọc nội dung từ tệp tin\n    logs512 = json.load(json_file)\nwith open('/kaggle/input/eval-sol-tranformer-last/logs_13-22.json', 'r') as json_file:\n    # Đọc nội dung từ tệp tin\n    logs1322 = json.load(json_file)","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:36:27.055413Z","iopub.execute_input":"2023-12-11T08:36:27.055757Z","iopub.status.idle":"2023-12-11T08:36:27.091559Z","shell.execute_reply.started":"2023-12-11T08:36:27.055726Z","shell.execute_reply":"2023-12-11T08:36:27.090871Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logs04[0]['loss_train']","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:36:27.092487Z","iopub.execute_input":"2023-12-11T08:36:27.092720Z","iopub.status.idle":"2023-12-11T08:36:27.098311Z","shell.execute_reply.started":"2023-12-11T08:36:27.092699Z","shell.execute_reply":"2023-12-11T08:36:27.097374Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:36:27.099390Z","iopub.execute_input":"2023-12-11T08:36:27.099648Z","iopub.status.idle":"2023-12-11T08:36:27.106766Z","shell.execute_reply.started":"2023-12-11T08:36:27.099625Z","shell.execute_reply":"2023-12-11T08:36:27.105907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tạo mảng x từ 0.1 đến 1 với bước 0.1\nx = np.arange(len(logs04))\n\n# Vẽ biểu đồ\nplt.plot(x, [sample['loss_train'] for sample in logs04], marker='o', linestyle='-', color='b', label='Loss Train')\nplt.plot(x, [sample['loss_test'] for sample in logs04], marker='o', linestyle='-', color='r', label='Loss Test')\n\nplt.title('Qúa trình huấn luyện với Level group 0-4')\nplt.xlabel('epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid(True)\nplt.savefig('plot_04.png')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:36:27.107873Z","iopub.execute_input":"2023-12-11T08:36:27.108472Z","iopub.status.idle":"2023-12-11T08:36:27.541721Z","shell.execute_reply.started":"2023-12-11T08:36:27.108439Z","shell.execute_reply":"2023-12-11T08:36:27.540778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tạo mảng x từ 0.1 đến 1 với bước 0.1\nx = np.arange(len(logs512))\n\n# Vẽ biểu đồ\nplt.plot(x, [sample['loss_train'] for sample in logs512], marker='o', linestyle='-', color='b', label='Loss Train')\nplt.plot(x, [sample['loss_test'] for sample in logs512], marker='o', linestyle='-', color='r', label='Loss Test')\n\nplt.title('Qúa trình huấn luyện với Level group 5-12')\nplt.xlabel('epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid(True)\nplt.savefig('plot_512.png')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:36:27.544993Z","iopub.execute_input":"2023-12-11T08:36:27.545664Z","iopub.status.idle":"2023-12-11T08:36:27.971179Z","shell.execute_reply.started":"2023-12-11T08:36:27.545626Z","shell.execute_reply":"2023-12-11T08:36:27.970284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Tạo mảng x từ 0.1 đến 1 với bước 0.1\nx = np.arange(len(logs1322))\n\n# Vẽ biểu đồ\nplt.plot(x, [sample['loss_train'] for sample in logs1322], marker='o', linestyle='-', color='b', label='Loss Train')\nplt.plot(x, [sample['loss_test'] for sample in logs1322], marker='o', linestyle='-', color='r', label='Loss Test')\n\nplt.title('Qúa trình huấn luyện với Level group 13-22')\nplt.xlabel('epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid(True)\nplt.savefig('plot_1322.png')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-12-11T08:36:27.972488Z","iopub.execute_input":"2023-12-11T08:36:27.972861Z","iopub.status.idle":"2023-12-11T08:36:28.349829Z","shell.execute_reply.started":"2023-12-11T08:36:27.972826Z","shell.execute_reply":"2023-12-11T08:36:28.348744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}