{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":19991,"databundleVersionId":1117522,"sourceType":"competition"}],"dockerImageVersionId":30918,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd\nimport torch\nimport os\nimport torch.nn.functional as F\nimport matplotlib.pyplot as plt\nfrom tqdm import tqdm","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-03-04T04:18:41.940139Z","iopub.execute_input":"2025-03-04T04:18:41.940474Z","iopub.status.idle":"2025-03-04T04:18:41.944540Z","shell.execute_reply.started":"2025-03-04T04:18:41.940445Z","shell.execute_reply":"2025-03-04T04:18:41.943564Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_path = '../input/alaska2-image-steganalysis'\n\ndef read_images_path(dir_name, label):\n    folder_path = os.path.join(base_path, dir_name)  # Đường dẫn đến thư mục\n    return [[os.path.join(folder_path, filename), label] for filename in os.listdir(folder_path)]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T04:18:43.795010Z","iopub.execute_input":"2025-03-04T04:18:43.795274Z","iopub.status.idle":"2025-03-04T04:18:43.799402Z","shell.execute_reply.started":"2025-03-04T04:18:43.795254Z","shell.execute_reply":"2025-03-04T04:18:43.798697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_size = 15000\n\ncover_img = read_images_path('Cover', 0)[:sample_size]\njmipod_img = read_images_path('JMiPOD', 1)[:(sample_size//3)]\njuniward_img = read_images_path('JUNIWARD', 1)[:(sample_size//3)]\nuerd_img = read_images_path('UERD', 1)[:(sample_size//3)]\n\ndata = cover_img + jmipod_img + juniward_img + uerd_img\ndf = pd.DataFrame(data=data, columns=['path', 'label'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T04:18:45.385272Z","iopub.execute_input":"2025-03-04T04:18:45.385660Z","iopub.status.idle":"2025-03-04T04:18:48.708461Z","shell.execute_reply.started":"2025-03-04T04:18:45.385576Z","shell.execute_reply":"2025-03-04T04:18:48.707805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def split_dataframe(df, ratio):\n    # Xáo trộn dữ liệu để đảm bảo tính ngẫu nhiên\n    df = df.sample(frac=1).reset_index(drop=True)\n    \n    # Chia dữ liệu thành hai phần\n    split_point = int(len(df) * ratio)\n    return df.iloc[:split_point], df.iloc[split_point:]\n\ntrain, val = split_dataframe(df, 0.8)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T04:18:52.174940Z","iopub.execute_input":"2025-03-04T04:18:52.175220Z","iopub.status.idle":"2025-03-04T04:18:52.199569Z","shell.execute_reply.started":"2025-03-04T04:18:52.175200Z","shell.execute_reply":"2025-03-04T04:18:52.198441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T04:18:54.459123Z","iopub.execute_input":"2025-03-04T04:18:54.459425Z","iopub.status.idle":"2025-03-04T04:18:54.488054Z","shell.execute_reply.started":"2025-03-04T04:18:54.459405Z","shell.execute_reply":"2025-03-04T04:18:54.487174Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\ndef preprocess_image(image_path, size=(384, 384), mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]):\n    \n    # Bước 1: Tải hình ảnh bằng PIL\n    img = Image.open(image_path)\n    \n    # Bước 2: Chuyển đổi không gian màu sang YCbCr\n    img = img.convert(\"YCbCr\")\n    \n    # Bước 3: Resize hình ảnh về kích thước cố định\n    img = img.resize(size)\n    \n    # Bước 4: Chuyển đổi thành numpy array và chuẩn hóa giá trị pixel về [0, 1]\n    img_array = np.array(img, dtype=np.float32) / 255.0\n    \n    # Đổi thứ tự kênh từ (H, W, C) sang (C, H, W)\n    img_array = np.transpose(img_array, (2, 0, 1))\n    \n    # Chuyển thành tensor PyTorch\n    img_tensor = torch.tensor(img_array)\n    \n    # Chuẩn hóa từng kênh bằng mean và std\n    for t, m, s in zip(img_tensor, mean, std):\n        t.sub_(m).div_(s)\n    \n    return img_tensor\n\ndef load_image(path):\n    img = preprocess_image(path)\n    return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T04:18:56.677437Z","iopub.execute_input":"2025-03-04T04:18:56.677790Z","iopub.status.idle":"2025-03-04T04:18:56.683349Z","shell.execute_reply.started":"2025-03-04T04:18:56.677762Z","shell.execute_reply":"2025-03-04T04:18:56.682655Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision.models.efficientnet import efficientnet_b2\nif torch.cuda.is_available():\n    device = torch.device('cuda')\n    print(\"CUDA is available. Using GPU.\")\nelse:\n    device = torch.device('cpu')\n    print(\"CUDA is not available. Using CPU.\")\n\n# Khởi tạo mô hình EfficientNet-B2 với trọng số pre-trained\nefficientnet_model = efficientnet_b2(pretrained=True)\n\n# Tùy chỉnh lớp cuối cùng của mô hình\nnum_classes = 2  # Số lượng lớp trong bài toán (phân loại nhị phân)\nefficientnet_model.classifier[1] = torch.nn.Linear(efficientnet_model.classifier[1].in_features, num_classes)\n\n# Đóng băng các tham số trong phần backbone\nfor name, param in efficientnet_model.named_parameters():\n    if \"classifier\" not in name:  # Chỉ giữ nguyên các lớp không phải classifier\n        param.requires_grad = False\n\n# Đưa mô hình lên thiết bị\nefficientnet_model.to(device)\n\n# Chuyển sang chế độ đánh giá\nefficientnet_model.eval()\n\nprint(\"EfficientNet-B2 model has been customized and is ready!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T04:18:58.947165Z","iopub.execute_input":"2025-03-04T04:18:58.947468Z","iopub.status.idle":"2025-03-04T04:19:02.986417Z","shell.execute_reply.started":"2025-03-04T04:18:58.947444Z","shell.execute_reply":"2025-03-04T04:19:02.985706Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.utils.data import Dataset, DataLoader\n\nclass SimpleImageDataset(Dataset):\n    def __init__(self, image_paths, labels):\n        self.image_paths = image_paths\n        self.labels = labels\n\n    def __len__(self):\n        return len(self.image_paths)\n\n    def __getitem__(self, index):\n        # Lấy đường dẫn ảnh và nhãn\n        img_path = self.image_paths[index]\n        label = self.labels[index]\n\n        # Sử dụng hàm preprocess_image\n        image = preprocess_image(img_path)\n\n        return image, label\n\nBATCHSIZE = 256\nEPOCHS = 10\nLR = 1e-4\nWEIGHT_DECAY = 0\nWEIGHTS = torch.tensor([4, 1], dtype=torch.float32).to(device)\n\n# Sử dụng resnet_model thay vì model\noptimizer = torch.optim.Adam(efficientnet_model.parameters(), lr=LR, weight_decay=WEIGHT_DECAY)\n\n# Tạo dataset cho train và validation\ntrain_dataset = SimpleImageDataset(train['path'].tolist(), train['label'].tolist())\nval_dataset = SimpleImageDataset(val['path'].tolist(), val['label'].tolist())\n\n# Tạo DataLoadera\ntrain_dataloader = DataLoader(train_dataset, batch_size=BATCHSIZE, shuffle=True, num_workers=2, pin_memory=True)\nval_dataloader = DataLoader(val_dataset, batch_size=BATCHSIZE, shuffle=False, num_workers=2, pin_memory=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T04:19:04.862395Z","iopub.execute_input":"2025-03-04T04:19:04.862883Z","iopub.status.idle":"2025-03-04T04:19:04.873725Z","shell.execute_reply.started":"2025-03-04T04:19:04.862855Z","shell.execute_reply":"2025-03-04T04:19:04.872918Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, roc_curve\ndef weighted_auc(y_true, y_score, tpr_thresholds=[0.0, 0.4, 1.0], weights=[2, 1]):\n    fpr, tpr, _ = roc_curve(y_true, y_score)\n    \n    # Tính AUC cho từng vùng\n    auc_scores = []\n    for i in range(len(tpr_thresholds) - 1):\n        start = tpr_thresholds[i]\n        end = tpr_thresholds[i+1]\n        \n        # Lọc các điểm trong vùng hiện tại\n        mask = (tpr >= start) & (tpr < end)\n        auc_scores.append(np.trapz(tpr[mask], fpr[mask]))\n    \n    # Tính AUC có trọng số\n    weighted_auc = np.sum(np.multiply(auc_scores, weights)) / np.sum(weights)\n    \n    return weighted_auc\n\ndef calculate_metrics(model, dataloader, device='cpu'):\n    model.to(device).eval()\n    labels, probs, preds = [], [], []\n\n    with torch.no_grad():\n        for images, batch_labels in dataloader:\n            outputs = model(images.to(device))\n            batch_probs = F.softmax(outputs, dim=1)[:, 1]\n            batch_preds = torch.argmax(outputs, dim=1)\n\n            labels.extend(batch_labels.cpu().numpy())\n            probs.extend(batch_probs.cpu().numpy())\n            preds.extend(batch_preds.cpu().numpy())\n\n    labels = np.array(labels)\n    probs = np.array(probs)\n    preds = np.array(preds)\n\n    return {\n        'Weighted AUC': weighted_auc(labels, probs),\n        'Accuracy': accuracy_score(labels, preds)\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T04:19:39.463050Z","iopub.execute_input":"2025-03-04T04:19:39.463337Z","iopub.status.idle":"2025-03-04T04:19:40.225894Z","shell.execute_reply.started":"2025-03-04T04:19:39.463316Z","shell.execute_reply":"2025-03-04T04:19:40.225001Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm\ntrain_losses = []  # Store training loss per epoch\naccuracies = []  # Store accuracy per epoch\nweighted_aucs = []  # Store weighted AUC per epoch\nbest_auc = 0\n\nfor epoch in range(EPOCHS):\n    print(f'EPOCH: {epoch+1}')\n    \n    # Training phase\n    efficientnet_model.train()\n    train_loss = 0\n    for images, labels in tqdm(train_dataloader, desc=\"Training\"):\n        images, labels = images.to(device), labels.to(device)\n        \n        optimizer.zero_grad()\n        loss = F.cross_entropy(efficientnet_model(images), labels, weight=WEIGHTS)\n        loss.backward()\n        optimizer.step()\n        \n        train_loss += loss.item()\n    avg_train_loss = train_loss / len(train_dataloader)\n    train_losses.append(avg_train_loss)\n    print(f'Average Training Loss: {avg_train_loss:.4f}')\n    \n    # Validation phase\n    metrics = calculate_metrics(efficientnet_model, val_dataloader, device)\n    acc = metrics['Accuracy']\n    current_auc = metrics['Weighted AUC']\n    \n    accuracies.append(acc)\n    weighted_aucs.append(current_auc)\n                        \n    print(f'Accuracy: {acc:.4f}')\n    print(f'Weighted AUC: {current_auc:.4f}')\n    \n    # Save best model\n    if current_auc > best_auc:\n        best_auc = current_auc\n        torch.save(efficientnet_model.state_dict(), 'best_model.pth')\n        print(f\"New best model saved with Weighted AUC: {best_auc:.4f}\")\n    \n    print()\n\n# Load best model\nefficientnet_model.load_state_dict(torch.load('best_model.pth'))\nprint(\"Best model loaded!\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T04:19:42.933114Z","iopub.execute_input":"2025-03-04T04:19:42.933638Z","iopub.status.idle":"2025-03-04T05:10:08.557807Z","shell.execute_reply.started":"2025-03-04T04:19:42.933582Z","shell.execute_reply":"2025-03-04T05:10:08.556761Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot Training Metrics\nplt.figure(figsize=(12, 4))\n\n# Plot Loss\nplt.subplot(1, 3, 1)\nplt.plot(range(1, EPOCHS+1), train_losses, marker='o', label='Training Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.title('Loss per Epoch')\nplt.legend()\n\n# Plot Accuracy\nplt.subplot(1, 3, 2)\nplt.plot(range(1, EPOCHS+1), accuracies, marker='o', label='Accuracy', color='green')\nplt.xlabel('Epoch')\nplt.ylabel('Accuracy')\nplt.title('Accuracy per Epoch')\nplt.legend()\n\n# Plot Weighted AUC\nplt.subplot(1, 3, 3)\nplt.plot(range(1, EPOCHS+1), weighted_aucs, marker='o', label='Weighted AUC', color='red')\nplt.xlabel('Epoch')\nplt.ylabel('AUC')\nplt.title('Weighted AUC per Epoch')\nplt.legend()\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T05:10:29.196325Z","iopub.execute_input":"2025-03-04T05:10:29.196691Z","iopub.status.idle":"2025-03-04T05:10:30.002144Z","shell.execute_reply.started":"2025-03-04T05:10:29.196658Z","shell.execute_reply":"2025-03-04T05:10:30.001092Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_img = read_images_path('Test', 0.7)\ntest = pd.DataFrame(data=test_img, columns=['path', 'label'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T05:11:20.296873Z","iopub.execute_input":"2025-03-04T05:11:20.297210Z","iopub.status.idle":"2025-03-04T05:11:20.371115Z","shell.execute_reply.started":"2025-03-04T05:11:20.297173Z","shell.execute_reply":"2025-03-04T05:11:20.370383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(model, test_dataloader, device):\n    model.eval()\n    predictions = []\n\n    with torch.no_grad():\n        for images, _ in test_dataloader:\n            images = images.to(device)\n            outputs = model(images)\n            scores = F.softmax(outputs, dim=1)[:, 1]  # Probability of being a stego image\n            predictions.extend(scores.cpu().numpy())\n\n    return predictions\n\n\ntest_dataset = SimpleImageDataset(test['path'].tolist(), test['label'].tolist())\n\ntest_dataloader = DataLoader(test_dataset, batch_size=BATCHSIZE, shuffle=False, num_workers=2, pin_memory=True)\n\nprint(\"Predicting...\")\ntest_predictions = predict(efficientnet_model, test_dataloader, device)\nprint(\"Finish.\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T05:11:21.964915Z","iopub.execute_input":"2025-03-04T05:11:21.965248Z","iopub.status.idle":"2025-03-04T05:12:31.987762Z","shell.execute_reply.started":"2025-03-04T05:11:21.965224Z","shell.execute_reply":"2025-03-04T05:12:31.986695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_image_ids = [os.path.basename(path) for path in test_dataloader.dataset.image_paths]\nsubmission_df = pd.DataFrame({\n    'Id': test_image_ids,\n    'Label': test_predictions\n})\n\nsubmission_df['Id'] = submission_df['Id'].str.replace('.jpg', '').astype(int)\nsubmission_df = submission_df.sort_values(by='Id')\nsubmission_df['Id'] = submission_df['Id'].astype(str).str.zfill(4) + '.jpg'\n\n# Save submission file\nsubmission_df.to_csv('submission.csv', index=False)\nprint(\"Submission file created: submission.csv\")\nsubmission_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-03-04T05:13:01.028484Z","iopub.execute_input":"2025-03-04T05:13:01.028798Z","iopub.status.idle":"2025-03-04T05:13:01.089571Z","shell.execute_reply.started":"2025-03-04T05:13:01.028774Z","shell.execute_reply":"2025-03-04T05:13:01.088686Z"}},"outputs":[],"execution_count":null}]}