{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# **Histopathologic Cancer Detection**","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\n\nimport os\nbase_dir = '../input/histopathologic-cancer-detection/'\nprint(os.listdir(base_dir))\n\n# Matplotlib for visualization\nimport matplotlib.pyplot as plt\nplt.style.use(\"ggplot\")\n\n# OpenCV Image Library\nimport cv2\n\n# Import PyTorch\nimport torchvision.transforms as transforms\nfrom torch.utils.data.sampler import SubsetRandomSampler\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import TensorDataset, DataLoader, Dataset\nimport torchvision\nimport torch.optim as optim\n\n# Import useful sklearn functions\nimport sklearn\nimport seaborn as sns\nfrom sklearn.metrics import (\n    accuracy_score, roc_auc_score, classification_report, confusion_matrix, roc_curve\n)\nfrom PIL import Image\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-03T07:53:05.758614Z","iopub.execute_input":"2025-10-03T07:53:05.758973Z","iopub.status.idle":"2025-10-03T07:53:06.222979Z","shell.execute_reply.started":"2025-10-03T07:53:05.758946Z","shell.execute_reply":"2025-10-03T07:53:06.222140Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Read data**","metadata":{}},{"cell_type":"code","source":"full_data = pd.read_csv(\"../input/histopathologic-cancer-detection/train_labels.csv\")\nfull_data.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T17:08:55.260585Z","iopub.execute_input":"2025-10-02T17:08:55.261142Z","iopub.status.idle":"2025-10-02T17:08:55.714993Z","shell.execute_reply.started":"2025-10-02T17:08:55.261095Z","shell.execute_reply":"2025-10-02T17:08:55.714078Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels_count = full_data.label.value_counts()\n\nplt.pie(labels_count, labels=['No Cancer', 'Cancer'], startangle=180, \n        autopct='%1.1f', colors=['#06d6a0','#ef476f'])\nplt.figure(figsize=(16,16))\nplt.show()\n\nlabels_count","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T17:08:55.716670Z","iopub.execute_input":"2025-10-02T17:08:55.716980Z","iopub.status.idle":"2025-10-02T17:08:55.900686Z","shell.execute_reply.started":"2025-10-02T17:08:55.716954Z","shell.execute_reply":"2025-10-02T17:08:55.899886Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Sampling data**","metadata":{}},{"cell_type":"code","source":"# Number of samples in each class\nSAMPLE_SIZE = 80000\n\n# Data paths\ntrain_path = '../input/histopathologic-cancer-detection/train/'\ntest_path = '../input/histopathologic-cancer-detection/test/'\n\n# Use 80000 positive and negative examples\ndf_negatives = full_data[full_data['label'] == 0].sample(SAMPLE_SIZE, random_state=42)\ndf_positives = full_data[full_data['label'] == 1].sample(SAMPLE_SIZE, random_state=42)\n\n# Concatenate the two dfs and shuffle them up\ntrain_df = sklearn.utils.shuffle(pd.concat([df_positives, df_negatives], axis=0).reset_index(drop=True))\n\ntrain_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T17:08:55.901613Z","iopub.execute_input":"2025-10-02T17:08:55.901936Z","iopub.status.idle":"2025-10-02T17:08:55.990128Z","shell.execute_reply.started":"2025-10-02T17:08:55.901908Z","shell.execute_reply":"2025-10-02T17:08:55.989185Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Data Preprocessing**","metadata":{}},{"cell_type":"code","source":"# Customize class for datasets\nclass CreateDataset(Dataset):\n    def __init__(self, df_data, data_dir = './', transform=None):\n        super().__init__()\n        self.df = df_data.values\n        self.data_dir = data_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        img_name,label = self.df[index]\n        img_path = os.path.join(self.data_dir, img_name+'.tif')\n        image = cv2.imread(img_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        if self.transform is not None:\n            image = self.transform(image)\n        return image, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T17:11:23.548668Z","iopub.execute_input":"2025-10-02T17:11:23.549060Z","iopub.status.idle":"2025-10-02T17:11:23.555585Z","shell.execute_reply.started":"2025-10-02T17:11:23.549034Z","shell.execute_reply":"2025-10-02T17:11:23.554360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transforms_train = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.RandomHorizontalFlip(p=0.4),\n    transforms.RandomVerticalFlip(p=0.4),\n    transforms.RandomRotation(20),\n    transforms.ToTensor(),\n    transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5))\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T17:11:27.869878Z","iopub.execute_input":"2025-10-02T17:11:27.870224Z","iopub.status.idle":"2025-10-02T17:11:27.874963Z","shell.execute_reply.started":"2025-10-02T17:11:27.870189Z","shell.execute_reply":"2025-10-02T17:11:27.874101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data = CreateDataset(df_data=train_df, data_dir=train_path, transform=transforms_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T17:11:42.478443Z","iopub.execute_input":"2025-10-02T17:11:42.478779Z","iopub.status.idle":"2025-10-02T17:11:42.498880Z","shell.execute_reply.started":"2025-10-02T17:11:42.478753Z","shell.execute_reply":"2025-10-02T17:11:42.498070Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size = 128    # Set Batch Size\nvalid_size = 0.2    # Percentage of training set to use as validation\n\n# obtain training indices that will be used for validation\nnum_train = len(train_data)\nindices = list(range(num_train))\n\n# np.random.shuffle(indices)\nsplit = int(np.floor(valid_size * num_train))\ntrain_idx, valid_idx = indices[split:], indices[:split]\n\n# Create Samplers\ntrain_sampler = SubsetRandomSampler(train_idx)\nvalid_sampler = SubsetRandomSampler(valid_idx)\n\n# prepare data loaders (combine dataset and sampler)\ntrain_loader = DataLoader(train_data, batch_size=batch_size, sampler=train_sampler)\nvalid_loader = DataLoader(train_data, batch_size=batch_size, sampler=valid_sampler)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T17:19:02.120225Z","iopub.execute_input":"2025-10-02T17:19:02.121319Z","iopub.status.idle":"2025-10-02T17:19:02.134102Z","shell.execute_reply.started":"2025-10-02T17:19:02.121275Z","shell.execute_reply":"2025-10-02T17:19:02.133178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\n\n# Save train and validation index\nwith open('train_idx.pkl', 'wb') as f:\n    pickle.dump(train_idx, f)\nwith open('valid_idx.pkl', 'wb') as f:\n    pickle.dump(valid_idx, f)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T17:26:22.170204Z","iopub.execute_input":"2025-10-02T17:26:22.170518Z","iopub.status.idle":"2025-10-02T17:26:22.180082Z","shell.execute_reply.started":"2025-10-02T17:26:22.170496Z","shell.execute_reply":"2025-10-02T17:26:22.179284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transforms_test = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.ToTensor(),\n    transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5))\n])\n\n# Creating test data\nsample_sub = pd.read_csv(\"../input/histopathologic-cancer-detection/sample_submission.csv\")\ntest_data = CreateDataset(df_data=sample_sub, data_dir=test_path, transform=transforms_test)\n\n# Prepare the test loader\ntest_loader = DataLoader(test_data, batch_size=batch_size, shuffle=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Defining Model Architecture**","metadata":{}},{"cell_type":"code","source":"class CNN(nn.Module):\n    def __init__(self):\n        super(CNN,self).__init__()\n        # Convolutional and Pooling Layers\n        self.conv1=nn.Sequential(\n                nn.Conv2d(in_channels=3,out_channels=32,kernel_size=3,stride=1,padding=0),\n                nn.BatchNorm2d(32),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        self.conv2=nn.Sequential(\n                nn.Conv2d(in_channels=32,out_channels=64,kernel_size=2,stride=1,padding=1),\n                nn.BatchNorm2d(64),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        self.conv3=nn.Sequential(\n                nn.Conv2d(in_channels=64,out_channels=128,kernel_size=3,stride=1,padding=1),\n                nn.BatchNorm2d(128),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        self.conv4=nn.Sequential(\n                nn.Conv2d(in_channels=128,out_channels=256,kernel_size=3,stride=1,padding=1),\n                nn.BatchNorm2d(256),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        self.conv5=nn.Sequential(\n                nn.Conv2d(in_channels=256, out_channels=512, kernel_size=3, stride=1, padding=1),\n                nn.BatchNorm2d(512),\n                nn.ReLU(inplace=True),\n                nn.MaxPool2d(2,2))\n        \n        self.dropout2d = nn.Dropout2d()\n        \n        self.fc=nn.Sequential(\n                nn.Linear(512*3*3,1024),\n                nn.ReLU(inplace=True),\n                nn.Dropout(0.4),\n                nn.Linear(1024,512),\n                nn.Dropout(0.4),\n                nn.Linear(512, 1),\n                nn.Sigmoid())\n        \n    def forward(self,x):\n        \"\"\"Method for Forward Prop\"\"\"\n        x=self.conv1(x)\n        x=self.conv2(x)\n        x=self.conv3(x)\n        x=self.conv4(x)\n        x=self.conv5(x)\n        x=x.view(x.shape[0],-1)\n        x=self.fc(x)\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-02T17:30:05.597709Z","iopub.execute_input":"2025-10-02T17:30:05.598072Z","iopub.status.idle":"2025-10-02T17:30:05.611559Z","shell.execute_reply.started":"2025-10-02T17:30:05.598046Z","shell.execute_reply":"2025-10-02T17:30:05.610532Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a complete CNN\nmodel = CNN()\nprint(model)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(f\"Training on {device}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Trainable Parameters\npytorch_total_params = sum(p.numel() for p in model.parameters() if p.requires_grad)\nprint(\"Number of trainable parameters: \\n{}\".format(pytorch_total_params))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Training Model**","metadata":{}},{"cell_type":"code","source":"def train_model(model, train_loader, val_loader, criterion, optimizer, device, num_epochs=50):\n    \"\"\" \n    Train & validate a PyTorch model \n        Args: \n            model : CNN model \n            train_loader: DataLoader cho tập train \n            val_loader : DataLoader cho tập validation \n            criterion : Loss function (nn.BCELoss hoặc nn.CrossEntropyLoss) \n            optimizer : Optimizer (Adam, SGD,...) \n            device : 'cuda' hoặc 'cpu' \n            num_epochs : số epoch huấn luyện \n        Returns: \n            model : mô hình sau khi train \n            history : dict chứa loss/acc theo epoch \n    \"\"\"\n    \n    history = {\n        \"train_loss\": [],\n        \"val_loss\": [],\n        \"train_auc\": [],\n        \"val_auc\": []\n    }\n    \n    best_val_loss = float(\"inf\")\n    best_model_wts = model.state_dict()\n    \n    model.to(device)\n\n    for epoch in range(num_epochs):\n        print(f\"\\nEpoch {epoch+1}/{num_epochs}\")\n        print(\"-\" * 30)\n        \n        # ------------------------\n        # TRAINING PHASE\n        # ------------------------\n        model.train()\n        running_loss, total = 0.0, 0\n        all_labels, all_outputs = [], []\n\n        for images, labels in tqdm(train_loader, desc=\"Training\"):\n            images, labels = images.to(device), labels.to(device)\n\n            optimizer.zero_grad()\n            \n            outputs = model(images).squeeze()   # đã sigmoid rồi\n            loss = criterion(outputs, labels.float())\n\n            all_outputs.extend(outputs.detach().cpu().numpy())\n            all_labels.extend(labels.detach().cpu().numpy())\n            \n            loss.backward()\n            optimizer.step()\n            \n            running_loss += loss.item() * images.size(0)\n            total += labels.size(0)\n        \n        epoch_train_loss = running_loss / total\n        try:\n            epoch_train_auc = roc_auc_score(all_labels, all_outputs)\n        except ValueError:\n            epoch_train_auc = 0.0\n\n        # ------------------------\n        # VALIDATION PHASE\n        # ------------------------\n        model.eval()\n        val_loss, val_total = 0.0, 0\n        val_labels, val_outputs = [], []\n\n        with torch.no_grad():\n            for images, labels in tqdm(val_loader, desc=\"Validation\"):\n                images, labels = images.to(device), labels.to(device)\n                \n                outputs = model(images).squeeze() \n                loss = criterion(outputs, labels.float())\n\n                val_outputs.extend(outputs.detach().cpu().numpy())\n                val_labels.extend(labels.detach().cpu().numpy())\n                \n                val_loss += loss.item() * images.size(0)\n                val_total += labels.size(0)\n        \n        epoch_val_loss = val_loss / val_total\n        try:\n            epoch_val_auc = roc_auc_score(val_labels, val_outputs)\n        except ValueError:\n            epoch_val_auc = 0.0\n\n        # ------------------------\n        # LOGGING\n        # ------------------------\n        history[\"train_loss\"].append(epoch_train_loss)\n        history[\"val_loss\"].append(epoch_val_loss)\n        history[\"train_auc\"].append(epoch_train_auc)\n        history[\"val_auc\"].append(epoch_val_auc)\n\n        print(f\"Train Loss: {epoch_train_loss:.4f} AUC: {epoch_train_auc:.4f}\")\n        print(f\"Valid Loss: {epoch_val_loss:.4f} AUC: {epoch_val_auc:.4f}\")\n        \n        # ------------------------\n        # SAVE BEST MODEL (theo val_loss nhỏ nhất)\n        # ------------------------\n        if epoch_val_loss < best_val_loss:\n            best_val_loss = epoch_val_loss\n            best_model_wts = model.state_dict().copy()\n            torch.save(best_model_wts, \"best_model.pth\")\n            print(\">>> Saved best model with Val Loss:\", best_val_loss)\n    \n    # load best weights\n    model.load_state_dict(best_model_wts)\n    return model, history","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"criterion = nn.BCELoss()\n# optimizer = optim.Adam(model.parameters(), lr=0.00015)\n# optimizer = torch.optim.SGD(model.parameters(), lr=0.00015, momentum=0.9, weight_decay=1e-4)\noptimizer = optim.AdamW(model.parameters(), lr=0.00015, weight_decay=1e-4)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model, history = train_model(model, train_loader, valid_loader, criterion, optimizer, device, num_epochs=20)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pickle\n\nwith open(\"train_history.pkl\", \"wb\") as f:\n    pickle.dump(history, f)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_history(history):\n    # Plot loss\n    plt.figure(figsize=(10,5))\n    plt.plot(history[\"train_loss\"], label=\"Train Loss\")\n    plt.plot(history[\"val_loss\"], label=\"Validation Loss\")\n    plt.xlabel(\"Epoch\")\n    plt.ylabel(\"Loss\")\n    plt.title(\"Train vs Validation Loss\")\n    plt.legend()\n    plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate_model(model, data_loader, device, threshold=0.5):\n    model.eval()\n    y_true = []\n    y_probs = []\n\n    with torch.no_grad():\n        for xb, yb in data_loader:\n            xb, yb = xb.to(device), yb.to(device)\n            logits = model(xb)\n            probs = torch.sigmoid(logits).squeeze().cpu().numpy()\n            y_probs.extend(probs)\n            y_true.extend(yb.cpu().numpy())\n\n    y_true = np.array(y_true)\n    y_probs = np.array(y_probs)\n    y_pred = (y_probs >= threshold).astype(int)\n\n    # Metrics\n    acc = accuracy_score(y_true, y_pred)\n    try:\n        auc = roc_auc_score(y_true, y_probs)\n    except ValueError:\n        auc = None\n    report = classification_report(y_true, y_pred, digits=4)\n    cm = confusion_matrix(y_true, y_pred)\n\n    print(\"=== Evaluation Report ===\")\n    print(f\"Accuracy : {acc:.4f}\")\n    if auc is not None:\n        print(f\"ROC AUC  : {auc:.4f}\")\n    print(\"\\nClassification Report:\\n\", report)\n\n    # Confusion Matrix\n    plt.figure(figsize=(5,4))\n    sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=[0,1], yticklabels=[0,1])\n    plt.xlabel(\"Predicted\")\n    plt.ylabel(\"True\")\n    plt.title(\"Confusion Matrix\")\n    plt.show()\n\n    # ROC Curve\n    if auc is not None:\n        fpr, tpr, _ = roc_curve(y_true, y_probs)\n        plt.figure(figsize=(5,4))\n        plt.plot(fpr, tpr, label=f\"AUC = {auc:.4f}\")\n        plt.plot([0,1],[0,1], linestyle=\"--\", color=\"gray\")\n        plt.xlabel(\"False Positive Rate\")\n        plt.ylabel(\"True Positive Rate\")\n        plt.title(\"ROC Curve\")\n        plt.legend()\n        plt.show()\n\n    return {\"accuracy\": acc, \"auc\": auc, \"report\": report, \"confusion_matrix\": cm}","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%matplotlib inline\nplot_history(history)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_results = evaluate_model(model, test_loader, device)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}