{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":30177,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport pickle\nimport os\nimport torch\nimport torch.nn as nn\nfrom tqdm import tqdm\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms, models\nfrom PIL import Image\nimport pandas as pd\nimport os\nfrom sklearn.model_selection import train_test_split\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nfrom PIL import Image\nimport pandas as pd\nimport os\n\ndef getDevice(which=\"cuda:0\", yellAtCpu=True):\n    if torch.cuda.is_available():\n        device = torch.device(which)\n    else:\n        if yellAtCpu:\n             raise Exception(\"I won't run on CPU!\")\n            \n        device = torch.device(\"cpu\")\n        \n    return device\ndevice=getDevice()","metadata":{"execution":{"iopub.status.busy":"2024-07-18T19:42:13.888981Z","iopub.execute_input":"2024-07-18T19:42:13.889702Z","iopub.status.idle":"2024-07-18T19:42:13.895533Z","shell.execute_reply.started":"2024-07-18T19:42:13.889662Z","shell.execute_reply":"2024-07-18T19:42:13.894733Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def merge_history(hlist):\n    history = {}\n    for k in hlist[0].history.keys():\n        history[k] = sum([h.history[k] for h in hlist], [])\n    return history\n\ndef vis_training(h, start=1):\n    epoch_range = range(start, len(h['loss'])+1)\n    s = slice(start-1, None)\n\n    plt.figure(figsize=[14,4])\n\n    n = int(len(h.keys()) / 2)\n\n    for i in range(n):\n        k = list(h.keys())[i]\n        plt.subplot(1,n,i+1)\n        plt.plot(epoch_range, h[k][s], label='Training')\n        plt.plot(epoch_range, h['val_' + k][s], label='Validation')\n        plt.xlabel('Epoch'); plt.ylabel(k); plt.title(k)\n        plt.grid()\n        plt.legend()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-18T19:30:32.470622Z","iopub.execute_input":"2024-07-18T19:30:32.470924Z","iopub.status.idle":"2024-07-18T19:30:32.481468Z","shell.execute_reply.started":"2024-07-18T19:30:32.470889Z","shell.execute_reply":"2024-07-18T19:30:32.480577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/histopathologic-cancer-detection/train_labels.csv', dtype=str)\nprint(train.shape)\ntrain.id = train.id + '.tif'\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2024-07-18T20:18:44.852690Z","iopub.execute_input":"2024-07-18T20:18:44.852990Z","iopub.status.idle":"2024-07-18T20:18:45.185257Z","shell.execute_reply.started":"2024-07-18T20:18:44.852957Z","shell.execute_reply":"2024-07-18T20:18:45.184576Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train[\"label\"] = train[\"label\"].astype(int)\n# train = train.iloc[0:140000]\ntrain_df, valid_df = train_test_split(train, test_size=0.2, random_state=42, stratify=train.label)\ntrain.shape","metadata":{"execution":{"iopub.status.busy":"2024-07-18T20:18:45.493016Z","iopub.execute_input":"2024-07-18T20:18:45.493810Z","iopub.status.idle":"2024-07-18T20:18:45.644485Z","shell.execute_reply.started":"2024-07-18T20:18:45.493769Z","shell.execute_reply":"2024-07-18T20:18:45.643738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = \"../input/histopathologic-cancer-detection/train\"\n\nsample = train.sample(n=16).reset_index()\n\nplt.figure(figsize=(6,6))\n\nfor i, row in sample.iterrows():\n\n    img = mpimg.imread(f'../input/histopathologic-cancer-detection/train/{row.id}')    \n    label = row.label\n\n    plt.subplot(4,4,i+1)\n    plt.imshow(img)\n    plt.text(0, -5, f'Class {label}', color='k')\n        \n    plt.axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-07-18T20:18:49.158850Z","iopub.execute_input":"2024-07-18T20:18:49.159112Z","iopub.status.idle":"2024-07-18T20:19:10.103342Z","shell.execute_reply.started":"2024-07-18T20:18:49.159085Z","shell.execute_reply":"2024-07-18T20:19:10.102563Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# Custom dataset class for Patch Camelyon\nclass PatchCamelyonDataset(Dataset):\n    def __init__(self, dataframe, directory, transform=None):\n        self.dataframe = dataframe\n        self.directory = directory\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.dataframe)\n\n    def __getitem__(self, idx):\n        img_name = os.path.join(self.directory, self.dataframe.iloc[idx, 0])  # Assuming images are PNG\n        image = Image.open(img_name).convert('RGB')\n        label = self.dataframe.iloc[idx, 1]\n        \n        if self.transform:\n            image = self.transform(image)\n\n        return image, label\n\n    # Constants\nBATCH_SIZE = 128\nIMAGE_SIZE = (96, 96)\n\n# Define transformations\ntransform = transforms.Compose([\n    transforms.Resize(IMAGE_SIZE),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406], std=[0.229, 0.224, 0.225]),\n])\n","metadata":{"execution":{"iopub.status.busy":"2024-07-18T20:39:13.938490Z","iopub.execute_input":"2024-07-18T20:39:13.938776Z","iopub.status.idle":"2024-07-18T20:39:13.948435Z","shell.execute_reply.started":"2024-07-18T20:39:13.938749Z","shell.execute_reply":"2024-07-18T20:39:13.947603Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Paths to the image directories\ntrain_path = '/kaggle/input/histopathologic-cancer-detection/train'\nvalid_path = '/kaggle/input/histopathologic-cancer-detection/train'\n\n# Create dataset instances\ntrain_dataset = PatchCamelyonDataset(dataframe=train_df, directory=train_path, transform=transform)\nvalid_dataset = PatchCamelyonDataset(dataframe=valid_df, directory=valid_path, transform=transform)\n\n# Create data loaders\ntrain_loader = DataLoader(train_dataset, batch_size=BATCH_SIZE, shuffle=True, num_workers=4, pin_memory=True)\nvalid_loader = DataLoader(valid_dataset, batch_size=BATCH_SIZE, shuffle=False, num_workers=4, pin_memory=True)","metadata":{"execution":{"iopub.status.busy":"2024-07-18T20:39:14.470614Z","iopub.execute_input":"2024-07-18T20:39:14.470900Z","iopub.status.idle":"2024-07-18T20:39:14.478952Z","shell.execute_reply.started":"2024-07-18T20:39:14.470868Z","shell.execute_reply":"2024-07-18T20:39:14.478267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Swish(nn.Module):\n    def __init__(self, beta=1.0):\n        super().__init__()\n        self.beta = beta\n\n    def forward(self, x):\n        return x * torch.sigmoid(self.beta * x)\n\nclass LinearProjection(nn.Module):\n    def __init__(self, embedding_dim, projection_dim, dropout=0.1):\n        super(LinearProjection, self).__init__()\n        self.projection = nn.Linear(embedding_dim, projection_dim)\n        self.swish = Swish(beta=1.0)\n        self.batch_norm = nn.BatchNorm1d(projection_dim)\n        self.fc = nn.Linear(projection_dim, projection_dim)\n        self.dropout = nn.Dropout(dropout)\n        self.layer_norm = nn.LayerNorm(projection_dim)\n\n    def forward(self, x):\n        projected = self.projection(x)\n        projected = self.batch_norm(projected)\n        x = self.swish(projected)\n        x = self.fc(x)\n        x = self.dropout(x)\n        x = x + projected\n        x = self.layer_norm(x)\n        return x\n    \nmodel = models.resnext50_32x4d(pretrained=True)\nnum_ftrs = model.fc.in_features\n# model.fc = nn.Linear(num_ftrs, 2)  # Assuming binary classification (negative/positive)\n\nlinear_proj = LinearProjection(model.fc.in_features, 2)\nmodel.fc=linear_proj\nnum_ftrs","metadata":{"execution":{"iopub.status.busy":"2024-07-18T20:39:16.857936Z","iopub.execute_input":"2024-07-18T20:39:16.858235Z","iopub.status.idle":"2024-07-18T20:39:17.367247Z","shell.execute_reply.started":"2024-07-18T20:39:16.858197Z","shell.execute_reply":"2024-07-18T20:39:17.366456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"criterion = nn.CrossEntropyLoss()\noptimizer = optim.AdamW(model.parameters(), lr=0.0001)\nscheduler = optim.lr_scheduler.StepLR(optimizer, step_size=7, gamma=0.1)\nmodel=model.to(device)\n\n# Training loop with accuracy tracking\nnum_epochs = 5\n\ntrain_losses = []\ntrain_accuracies = []\nvalid_losses = []\nvalid_accuracies = []\n\nfor epoch in range(num_epochs):\n    print(f'Epoch {epoch}/{num_epochs - 1}')\n    print('-' * 10)\n    \n    # Initialize dictionaries to store losses and accuracies\n    epoch_loss = {'train': 0.0, 'valid': 0.0}\n    epoch_acc = {'train': 0.0, 'valid': 0.0}\n    \n    for phase in ['train', 'valid']:\n        if phase == 'train':\n            model.train()  # Set model to training mode\n            dataloader = train_loader\n        else:\n            model.eval()  # Set model to evaluate mode\n            dataloader = valid_loader\n        \n        running_loss = 0.0\n        running_corrects = 0\n        \n        # Iterate over data\n        for inputs, labels in tqdm(dataloader):\n            \n            inputs = inputs.to(device)\n            labels = torch.tensor(labels).to(device)\n            \n            optimizer.zero_grad()\n            \n            # Forward\n            with torch.set_grad_enabled(phase == 'train'):\n                outputs = model(inputs)\n                _, preds = torch.max(outputs, 1)\n                loss = criterion(outputs, labels)\n                \n                # Backward + optimize only if in training phase\n                if phase == 'train':\n                    loss.backward()\n                    optimizer.step()\n            \n            # Statistics\n            running_loss += loss.item() * inputs.size(0)\n            running_corrects += torch.sum(preds == labels.data)\n        \n        if phase == 'train':\n            scheduler.step()\n        \n        epoch_loss[phase] = running_loss / len(dataloader.dataset)\n        epoch_acc[phase] = running_corrects.double() / len(dataloader.dataset)\n    \n    # Append epoch statistics to lists\n    train_losses.append(epoch_loss['train'])\n    train_accuracies.append(epoch_acc['train'].item())\n    valid_losses.append(epoch_loss['valid'])\n    valid_accuracies.append(epoch_acc['valid'].item())\n    \n    # Print epoch statistics\n    print(f'Train Loss: {epoch_loss[\"train\"]:.4f} Acc: {epoch_acc[\"train\"]:.4f}')\n    print(f'Valid Loss: {epoch_loss[\"valid\"]:.4f} Acc: {epoch_acc[\"valid\"]:.4f}')\n    print()\n","metadata":{"execution":{"iopub.status.busy":"2024-07-18T20:39:17.725375Z","iopub.execute_input":"2024-07-18T20:39:17.725877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"valid_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"de403e6cdaf463393843e70470d8c8bf178fd14b.tif\" in _df[\"id\"].tolist()","metadata":{"execution":{"iopub.status.busy":"2024-07-18T20:29:27.088926Z","iopub.execute_input":"2024-07-18T20:29:27.089183Z","iopub.status.idle":"2024-07-18T20:29:27.096967Z","shell.execute_reply.started":"2024-07-18T20:29:27.089156Z","shell.execute_reply":"2024-07-18T20:29:27.096157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"filenames","metadata":{"execution":{"iopub.status.busy":"2024-07-18T20:37:09.312453Z","iopub.execute_input":"2024-07-18T20:37:09.313201Z","iopub.status.idle":"2024-07-18T20:37:09.318313Z","shell.execute_reply.started":"2024-07-18T20:37:09.313162Z","shell.execute_reply":"2024-07-18T20:37:09.317545Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# Folder path\nfolder_path = '/kaggle/input/histopathologic-cancer-detection/train'\n\n# Read filenames from the folder\nfilenames = os.listdir(folder_path)\n","metadata":{"execution":{"iopub.status.busy":"2024-07-18T20:37:44.754858Z","iopub.execute_input":"2024-07-18T20:37:44.755149Z","iopub.status.idle":"2024-07-18T20:37:47.466164Z","shell.execute_reply.started":"2024-07-18T20:37:44.755113Z","shell.execute_reply":"2024-07-18T20:37:47.465460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"de403e6cdaf463393843e70470d8c8bf178fd14b.tif\" in filenames","metadata":{"execution":{"iopub.status.busy":"2024-07-18T20:37:51.841176Z","iopub.execute_input":"2024-07-18T20:37:51.841497Z","iopub.status.idle":"2024-07-18T20:37:51.847892Z","shell.execute_reply.started":"2024-07-18T20:37:51.841461Z","shell.execute_reply":"2024-07-18T20:37:51.847100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nseries = valid_df[\"id\"]\n\n# Folder path\nfolder_path = '/kaggle/input/histopathologic-cancer-detection/test'\n\n# Read filenames from the folder\nfilenames = os.listdir(folder_path)\n\n# Check if filenames are in the Pandas Series\nresults = {filename: filename in series.values for filename in filenames}\n\n# Print results\nfor filename, exists in tqdm(results.items()):\n    print(f\"{filename}: {'Exists' if exists else 'Does not exist'}\")\n","metadata":{"_kg_hide-input":true,"trusted":true},"execution_count":null,"outputs":[]}]}