{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"isSourceIdPinned":false,"sourceType":"competition"},{"sourceId":13261609,"sourceType":"datasetVersion","datasetId":8403706}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Histopathologic Cancer Detection**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\nbase_dir = '../input/histopathologic-cancer-detection/'\nprint(os.listdir(base_dir))\n\n# Matplotlib and Seaborn for visualization\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nplt.style.use(\"ggplot\")\n\n# OpenCV Image Library\nimport cv2\n\n# Import PyTorch\nimport torchvision.transforms as transforms\nfrom torch.utils.data.sampler import SubsetRandomSampler\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import TensorDataset, DataLoader, Dataset, random_split\nimport torchvision\nimport torch.optim as optim\n\n# Import useful sklearn functions\nimport sklearn\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import (\n    accuracy_score, roc_auc_score, classification_report, confusion_matrix, roc_curve\n)\nfrom PIL import Image\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T07:53:05.758614Z","iopub.execute_input":"2025-10-03T07:53:05.758973Z","iopub.status.idle":"2025-10-03T07:53:06.222979Z","shell.execute_reply.started":"2025-10-03T07:53:05.758946Z","shell.execute_reply":"2025-10-03T07:53:06.222140Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Data Preprocessing**","metadata":{}},{"cell_type":"code","source":"# Data paths\ntrain_path = '../input/histopathologic-cancer-detection/train/'\ntest_path = '../input/histopathologic-cancer-detection/test/'","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read CSV file\ntrain_df = pd.read_csv(\"../input/histopathologic-cancer-train-test-index/train_split.csv\")\ntest_df = pd.read_csv(\"../input/histopathologic-cancer-train-test-index/test_split.csv\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Customize class for datasets\nclass CreateDataset(Dataset):\n    def __init__(self, df_data, data_dir='./', transform=None):\n        super().__init__()\n        self.df = df_data.reset_index(drop=True)\n        self.data_dir = data_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        img_name, label = self.df.iloc[index]\n        img_path = os.path.join(self.data_dir, img_name + '.tif')\n        image = cv2.imread(img_path)\n        if image is None:\n            raise FileNotFoundError(f\"Image {img_path} not found.\")\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        if self.transform:\n            image = self.transform(image)\n        return image, torch.tensor(label, dtype=torch.long)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T17:11:23.548668Z","iopub.execute_input":"2025-10-02T17:11:23.549060Z","iopub.status.idle":"2025-10-02T17:11:23.555585Z","shell.execute_reply.started":"2025-10-02T17:11:23.549034Z","shell.execute_reply":"2025-10-02T17:11:23.554360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transforms_train = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.ToTensor(),\n    transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5))\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T17:11:27.869878Z","iopub.execute_input":"2025-10-02T17:11:27.870224Z","iopub.status.idle":"2025-10-02T17:11:27.874963Z","shell.execute_reply.started":"2025-10-02T17:11:27.870189Z","shell.execute_reply":"2025-10-02T17:11:27.874101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = CreateDataset(df_data=train_df, data_dir=train_path, transform=transforms_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T17:11:42.478443Z","iopub.execute_input":"2025-10-02T17:11:42.478779Z","iopub.status.idle":"2025-10-02T17:11:42.498880Z","shell.execute_reply.started":"2025-10-02T17:11:42.478753Z","shell.execute_reply":"2025-10-02T17:11:42.498070Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size = 128    # Set Batch Size\nvalid_size = 0.2    # Percentage of training set to use as validation\n\nnum_train = len(train_data)  \ntrain_size = int((1 - valid_size) * num_train)  \nvalid_size = num_train - train_size  \n\ntrain_data, valid_data = random_split(train_data, [train_size, valid_size])\n\n# Prepare data loaders (combine dataset and sampler)\ntrain_loader = DataLoader(train_data, batch_size=batch_size, shuffle=True)\nvalid_loader = DataLoader(valid_data, batch_size=batch_size, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T17:19:02.120225Z","iopub.execute_input":"2025-10-02T17:19:02.121319Z","iopub.status.idle":"2025-10-02T17:19:02.134102Z","shell.execute_reply.started":"2025-10-02T17:19:02.121275Z","shell.execute_reply":"2025-10-02T17:19:02.133178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transforms_test = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.ToTensor(),\n    transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5))\n])\n\n# Creating test data\ntest_data = CreateDataset(df_data=test_df, data_dir=train_path, transform=transforms_test)\n\n# Prepare the test loader\ntest_loader = DataLoader(test_data, batch_size=batch_size, shuffle=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Defining Model Architecture**","metadata":{}},{"cell_type":"code","source":"class CNN(nn.Module):\n    def __init__(self):\n        super(CNN,self).__init__()\n        # Convolutional and Pooling Layers\n        self.conv1=nn.Sequential(\n                nn.Conv2d(in_channels=3,out_channels=32,kernel_size=3,stride=1,padding=0),\n                nn.BatchNorm2d(32),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        self.conv2=nn.Sequential(\n                nn.Conv2d(in_channels=32,out_channels=64,kernel_size=2,stride=1,padding=1),\n                nn.BatchNorm2d(64),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        self.conv3=nn.Sequential(\n                nn.Conv2d(in_channels=64,out_channels=128,kernel_size=3,stride=1,padding=1),\n                nn.BatchNorm2d(128),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        self.conv4=nn.Sequential(\n                nn.Conv2d(in_channels=128,out_channels=256,kernel_size=3,stride=1,padding=1),\n                nn.BatchNorm2d(256),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        self.conv5=nn.Sequential(\n                nn.Conv2d(in_channels=256, out_channels=512, kernel_size=3, stride=1, padding=1),\n                nn.BatchNorm2d(512),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        \n        self.dropout2d = nn.Dropout2d()\n        \n        self.fc=nn.Sequential(\n                nn.Linear(512*3*3,1024),\n                nn.ReLU(inplace=True),\n                nn.Dropout(0.4),\n                nn.Linear(1024,512),\n                nn.Dropout(0.4),\n                nn.Linear(512, 1),\n                nn.Sigmoid())\n        \n    def forward(self,x):\n        \"\"\"Method for Forward Prop\"\"\"\n        x=self.conv1(x)\n        x=self.conv2(x)\n        x=self.conv3(x)\n        x=self.conv4(x)\n        x=self.conv5(x)\n        x=x.view(x.shape[0],-1)\n        x=self.fc(x)\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T17:30:05.597709Z","iopub.execute_input":"2025-10-02T17:30:05.598072Z","iopub.status.idle":"2025-10-02T17:30:05.611559Z","shell.execute_reply.started":"2025-10-02T17:30:05.598046Z","shell.execute_reply":"2025-10-02T17:30:05.610532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a complete CNN\nmodel = CNN()\nprint(model)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Training on {device}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Trainable Parameters\npytorch_total_params = sum(p.numel() for p in model.parameters() if p.requires_grad)\nprint(\"Number of trainable parameters: \\n{}\".format(pytorch_total_params))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Training Model**","metadata":{}},{"cell_type":"code","source":"def train_model(model, train_loader, val_loader, criterion, optimizer, device, num_epochs=50):\n    \"\"\" \n    Train & validate a PyTorch model \n        Args: \n            model : CNN model \n            train_loader: DataLoader cho tập train \n            val_loader : DataLoader cho tập validation \n            criterion : Loss function (nn.BCELoss hoặc nn.CrossEntropyLoss) \n            optimizer : Optimizer (Adam, SGD,...) \n            device : 'cuda' hoặc 'cpu' \n            num_epochs : số epoch huấn luyện \n        Returns: \n            model : mô hình sau khi train \n            history : dict chứa loss/acc theo epoch \n    \"\"\"\n    \n    history = {\n        \"train_loss\": [],\n        \"val_loss\": [],\n        \"train_auc\": [],\n        \"val_auc\": []\n    }\n    \n    best_val_loss = float(\"inf\")\n    best_model_wts = model.state_dict()\n    \n    model.to(device)\n\n    for epoch in range(num_epochs):\n        print(f\"\\nEpoch {epoch+1}/{num_epochs}\")\n        print(\"-\" * 30)\n        \n        # ------------------------\n        # TRAINING PHASE\n        # ------------------------\n        model.train()\n        running_loss, total = 0.0, 0\n        all_labels, all_outputs = [], []\n\n        for images, labels in tqdm(train_loader, desc=\"Training\"):\n            images, labels = images.to(device), labels.to(device)\n\n            optimizer.zero_grad()\n            \n            outputs = model(images).squeeze()   # đã sigmoid rồi\n            loss = criterion(outputs, labels.float())\n\n            all_outputs.extend(outputs.detach().cpu().numpy())\n            all_labels.extend(labels.detach().cpu().numpy())\n            \n            loss.backward()\n            optimizer.step()\n            \n            running_loss += loss.item() * images.size(0)\n            total += labels.size(0)\n        \n        epoch_train_loss = running_loss / total\n        try:\n            epoch_train_auc = roc_auc_score(all_labels, all_outputs)\n        except ValueError:\n            epoch_train_auc = 0.0\n\n        # ------------------------\n        # VALIDATION PHASE\n        # ------------------------\n        model.eval()\n        val_loss, val_total = 0.0, 0\n        val_labels, val_outputs = [], []\n\n        with torch.no_grad():\n            for images, labels in tqdm(val_loader, desc=\"Validation\"):\n                images, labels = images.to(device), labels.to(device)\n                \n                outputs = model(images).squeeze() \n                loss = criterion(outputs, labels.float())\n\n                val_outputs.extend(outputs.detach().cpu().numpy())\n                val_labels.extend(labels.detach().cpu().numpy())\n                \n                val_loss += loss.item() * images.size(0)\n                val_total += labels.size(0)\n        \n        epoch_val_loss = val_loss / val_total\n        try:\n            epoch_val_auc = roc_auc_score(val_labels, val_outputs)\n        except ValueError:\n            epoch_val_auc = 0.0\n\n        # ------------------------\n        # LOGGING\n        # ------------------------\n        history[\"train_loss\"].append(epoch_train_loss)\n        history[\"val_loss\"].append(epoch_val_loss)\n        history[\"train_auc\"].append(epoch_train_auc)\n        history[\"val_auc\"].append(epoch_val_auc)\n\n        print(f\"Train Loss: {epoch_train_loss:.4f} AUC: {epoch_train_auc:.4f}\")\n        print(f\"Valid Loss: {epoch_val_loss:.4f} AUC: {epoch_val_auc:.4f}\")\n        \n        # ------------------------\n        # SAVE BEST MODEL (theo val_loss nhỏ nhất)\n        # ------------------------\n        if epoch_val_loss < best_val_loss:\n            best_val_loss = epoch_val_loss\n            best_model_wts = model.state_dict().copy()\n            torch.save(best_model_wts, \"best_model.pth\")\n            print(\">>> Saved best model with Val Loss:\", best_val_loss)\n    \n    # load best weights\n    model.load_state_dict(best_model_wts)\n    return model, history","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"criterion = nn.BCELoss()\noptimizer = optim.AdamW(model.parameters(), lr=0.00015, weight_decay=1e-4)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model, history = train_model(model, train_loader, valid_loader, criterion, optimizer, device, num_epochs=30)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\n\nwith open(\"train_history.pkl\", \"wb\") as f:\n    pickle.dump(history, f)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_history(history):\n    # Plot loss\n    plt.figure(figsize=(10,5))\n    plt.plot(history[\"train_loss\"], label=\"Train Loss\")\n    plt.plot(history[\"val_loss\"], label=\"Validation Loss\")\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(\"Loss\")\n    plt.title(\"Train vs Validation Loss\")\n    plt.legend()\n    plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate_model(model, data_loader, device, threshold=None, min_tpr=0.95, tune_threshold=True):\n    \"\"\"\n    Evaluate model trên tập dữ liệu và (tuỳ chọn) tự động tìm threshold tốt nhất\n    theo tiêu chí Recall (TPR) ưu tiên.\n    \n    Args:\n        model : mô hình đã huấn luyện\n        data_loader : DataLoader cho tập validation/test\n        device : 'cuda' hoặc 'cpu'\n        threshold : threshold cố định (nếu không muốn auto-tune)\n        min_tpr : ngưỡng tối thiểu của Recall khi tune threshold\n        tune_threshold : nếu True -> tự động tìm threshold tối ưu theo Recall\n\n    Returns:\n        metrics : dict chứa kết quả đánh giá\n    \"\"\"\n    model.eval()\n    y_true, y_probs = [], []\n\n    with torch.no_grad():\n        for xb, yb in data_loader:\n            xb, yb = xb.to(device), yb.to(device)\n            logits = model(xb)\n            \n            # Nếu model cuối có Sigmoid rồi thì bỏ sigmoid ở đây\n            if logits.shape[-1] == 1:\n                probs = logits.squeeze().detach().cpu().numpy()\n            else:\n                probs = torch.sigmoid(logits).squeeze().detach().cpu().numpy()\n\n            y_probs.extend(probs)\n            y_true.extend(yb.cpu().numpy())\n\n    y_true = np.array(y_true)\n    y_probs = np.array(y_probs)\n\n    # =====================\n    # 1️⃣ Tune threshold nếu bật\n    # =====================\n    if tune_threshold:\n        fpr, tpr, thresholds = roc_curve(y_true, y_probs)\n        valid_idx = np.where(tpr >= min_tpr)[0]\n\n        if len(valid_idx) == 0:\n            print(f\"⚠️ Không có threshold nào đạt TPR >= {min_tpr}. Sẽ chọn threshold với TPR cao nhất.\")\n            best_idx = np.argmax(tpr)\n        else:\n            best_idx = valid_idx[np.argmin(fpr[valid_idx])]\n\n        threshold = thresholds[best_idx]\n        print(f\"\\n🔍 Tuning threshold (TPR priority): {threshold:.4f}\")\n        print(f\"TPR={tpr[best_idx]:.4f}, FPR={fpr[best_idx]:.4f}\")\n\n    elif threshold is None:\n        threshold = 0.5  # Mặc định\n\n    # =====================\n    # 2️⃣ Áp dụng threshold để tính metric\n    # =====================\n    y_pred = (y_probs >= threshold).astype(int)\n\n    acc = accuracy_score(y_true, y_pred)\n    try:\n        auc = roc_auc_score(y_true, y_probs)\n    except ValueError:\n        auc = None\n\n    report = classification_report(y_true, y_pred, digits=4)\n    cm = confusion_matrix(y_true, y_pred)\n\n    print(\"\\n=== Evaluation Report ===\")\n    print(f\"Accuracy : {acc:.4f}\")\n    if auc is not None:\n        print(f\"ROC AUC  : {auc:.4f}\")\n    print(f\"Threshold: {threshold:.4f}\")\n    print(\"\\nClassification Report:\\n\", report)\n\n    # =====================\n    # 3️⃣ Confusion Matrix\n    # =====================\n    plt.figure(figsize=(5, 4))\n    sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=[0, 1], yticklabels=[0, 1])\n    plt.xlabel(\"Predicted\")\n    plt.ylabel(\"True\")\n    plt.title(f\"Confusion Matrix (Threshold={threshold:.3f})\")\n    plt.show()\n\n    # =====================\n    # 4️⃣ ROC Curve\n    # =====================\n    if auc is not None:\n        fpr, tpr, _ = roc_curve(y_true, y_probs)\n        plt.figure(figsize=(5, 4))\n        plt.plot(fpr, tpr, label=f\"AUC = {auc:.4f}\")\n        plt.plot([0, 1], [0, 1], linestyle=\"--\", color=\"gray\")\n        plt.xlabel(\"False Positive Rate\")\n        plt.ylabel(\"True Positive Rate\")\n        plt.title(\"ROC Curve\")\n        plt.legend()\n        plt.show()\n\n    return {\n        \"accuracy\": acc,\n        \"auc\": auc,\n        \"report\": report,\n        \"confusion_matrix\": cm,\n        \"threshold\": threshold\n    }","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%matplotlib inline\nplot_history(history)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_results = evaluate_model(\n    model,\n    test_loader,\n    device,\n    tune_threshold=True,\n    min_tpr=0.95\n)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}