{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31236,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom tqdm import tqdm\nimport numpy as np\nimport pandas as pd\nfrom itertools import accumulate\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom PIL import Image\n\n%matplotlib inline\n\nimport torch\nfrom torch.utils.data import TensorDataset, DataLoader,Dataset, random_split\nimport torch.nn as nn\nimport torchvision\nimport torchvision.transforms as transforms\nimport torch.optim as optim\n\ndef warn(*args, **kwargs):\n    pass\nimport warnings\nwarnings.warn = warn\nwarnings.filterwarnings('ignore')\n\nsns.set_context('notebook')\nsns.set_style('white')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:25:40.659966Z","iopub.execute_input":"2025-12-20T14:25:40.660555Z","iopub.status.idle":"2025-12-20T14:25:50.235579Z","shell.execute_reply.started":"2025-12-20T14:25:40.660527Z","shell.execute_reply":"2025-12-20T14:25:50.234815Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path_data = '/kaggle/input/histopathologic-cancer-detection'\nprint(os.listdir(path_data))\n\nprint(os.listdir(os.path.join(path_data, 'train'))[:5])\nprint(os.listdir(os.path.join(path_data, 'test'))[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:25:50.236811Z","iopub.execute_input":"2025-12-20T14:25:50.237148Z","iopub.status.idle":"2025-12-20T14:25:53.786925Z","shell.execute_reply.started":"2025-12-20T14:25:50.237125Z","shell.execute_reply":"2025-12-20T14:25:53.786314Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.cuda import is_available, get_device_name\n\nif is_available():\n    print(f\"The environment has a compatible GPU ({get_device_name()}) available.\")\nelse:\n    print(f\"The environment does NOT have a compatible GPU model available.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:25:53.787810Z","iopub.execute_input":"2025-12-20T14:25:53.788025Z","iopub.status.idle":"2025-12-20T14:25:53.898936Z","shell.execute_reply.started":"2025-12-20T14:25:53.788007Z","shell.execute_reply":"2025-12-20T14:25:53.898352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from numpy import clip , array\nfrom matplotlib import pyplot as plt\nfrom torch import Tensor\n\ndef imshow(inp: Tensor) -> None:\n    \"\"\"Imshow for Tensor.\"\"\"\n    inp = inp.cpu().numpy()\n    inp = inp.transpose((1, 2, 0))\n    mean = array([0.485, 0.456, 0.406])\n    std = array([0.229, 0.224, 0.225])\n    inp = std * inp + mean\n    inp = clip(inp, 0, 1)\n    plt.imshow(inp)\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:25:53.899711Z","iopub.execute_input":"2025-12-20T14:25:53.899965Z","iopub.status.idle":"2025-12-20T14:25:53.904981Z","shell.execute_reply.started":"2025-12-20T14:25:53.899943Z","shell.execute_reply":"2025-12-20T14:25:53.904311Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Dataset","metadata":{}},{"cell_type":"code","source":"## Load the label of data\nlabels_df = pd.read_csv(\"/kaggle/input/histopathologic-cancer-detection/train_labels.csv\")\nlabels_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:25:53.906677Z","iopub.execute_input":"2025-12-20T14:25:53.907157Z","iopub.status.idle":"2025-12-20T14:25:54.167177Z","shell.execute_reply.started":"2025-12-20T14:25:53.907136Z","shell.execute_reply":"2025-12-20T14:25:54.166353Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig = plt.figure(figsize=(25, 8))\n\ntrain_imgs = os.listdir(os.path.join(path_data, 'train'))\n\nfor idx, img in enumerate(np.random.choice(train_imgs, 40)):\n\n    ax = fig.add_subplot(4, 40//4, idx+1)\n\n    im = Image.open(os.path.join(path_data, 'train', img))\n\n    plt.imshow(im)\n\n    lab = labels_df.loc[labels_df[\"id\"] == img.split('.')[0], 'label'].values[0]\n\n    ax.set_title(f\"Label: {lab}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:25:54.168175Z","iopub.execute_input":"2025-12-20T14:25:54.168869Z","iopub.status.idle":"2025-12-20T14:26:01.168775Z","shell.execute_reply.started":"2025-12-20T14:25:54.168843Z","shell.execute_reply":"2025-12-20T14:26:01.168072Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Data Preparation","metadata":{}},{"cell_type":"code","source":"labels_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:01.169771Z","iopub.execute_input":"2025-12-20T14:26:01.170121Z","iopub.status.idle":"2025-12-20T14:26:01.175415Z","shell.execute_reply.started":"2025-12-20T14:26:01.170085Z","shell.execute_reply":"2025-12-20T14:26:01.174712Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels_df = labels_df.set_index('id')\nlabels_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:01.176343Z","iopub.execute_input":"2025-12-20T14:26:01.176573Z","iopub.status.idle":"2025-12-20T14:26:01.197368Z","shell.execute_reply.started":"2025-12-20T14:26:01.176544Z","shell.execute_reply":"2025-12-20T14:26:01.196830Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import random\n\nrandom.seed(666)\nidx = list(labels_df.index)\nidx[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:01.198144Z","iopub.execute_input":"2025-12-20T14:26:01.198319Z","iopub.status.idle":"2025-12-20T14:26:01.218423Z","shell.execute_reply.started":"2025-12-20T14:26:01.198303Z","shell.execute_reply":"2025-12-20T14:26:01.217842Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"idx_p = list(range(len(idx)))\nprint(idx_p[:5])\nrandom.shuffle(idx_p)\nprint(idx_p[:5])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:01.219213Z","iopub.execute_input":"2025-12-20T14:26:01.219676Z","iopub.status.idle":"2025-12-20T14:26:01.318588Z","shell.execute_reply.started":"2025-12-20T14:26:01.219656Z","shell.execute_reply":"2025-12-20T14:26:01.317836Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"idx_random = [idx[x] for x in idx_p]\nidx_random[:5]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:01.319397Z","iopub.execute_input":"2025-12-20T14:26:01.319647Z","iopub.status.idle":"2025-12-20T14:26:01.371453Z","shell.execute_reply.started":"2025-12-20T14:26:01.319626Z","shell.execute_reply":"2025-12-20T14:26:01.370917Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"total = len(idx_random)\nidx_70 = int(total * 0.70)\nidx_90 = int(total * 0.90)\n\nidx_frac_70 = idx_random[:idx_70] \nidx_frac_20 = idx_random[idx_70:idx_90]\nidx_frac_10 = idx_random[idx_90:] \n\nprint(f\"(70%): {len(idx_frac_70)} itens\")\nprint(f\"(20%): {len(idx_frac_20)} itens\")\nprint(f\"(10%): {len(idx_frac_10)} itens\")\nprint(f\"Total: {len(idx_frac_70) + len(idx_frac_20) + len(idx_frac_10)} itens\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:01.372316Z","iopub.execute_input":"2025-12-20T14:26:01.372569Z","iopub.status.idle":"2025-12-20T14:26:01.382761Z","shell.execute_reply.started":"2025-12-20T14:26:01.372548Z","shell.execute_reply":"2025-12-20T14:26:01.382215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from pathlib import Path\n\nclass CancerDataset(Dataset):\n    def __init__(\n        self,\n        path_to_dataset: str,\n        transform,\n        dataset_type=None):\n\n        path_dataset = Path(path_to_dataset)\n        if not path_dataset.is_dir():\n            raise OSError('This is not directory')\n\n        check_data_split =  ['train', 'test']\n        sub_dirs = [os.path.basename(str(x)) for x in path_dataset.iterdir()]\n        \n        if not check_data_split[0] in sub_dirs and not check_data_split[1] in sub_dirs:\n            raise Exception('Does not exists train dir or test dir')\n\n        self.path_dataset_train = path_dataset.joinpath(check_data_split[0])\n        self.path_dataset_test = path_dataset.joinpath(check_data_split[1])\n\n        if not 'train_labels.csv' in sub_dirs:\n            raise Exception('File labels does not found')\n        \n        self.labels_file = path_dataset / \"train_labels.csv\"\n        self.df_labels = pd.read_csv(self.labels_file)\n        self.df_labels.set_index(\"id\", inplace=True)\n\n        random.seed(666)\n        idx = list(labels_df.index)\n        idx_p = list(range(len(idx)))\n        random.shuffle(idx_p)\n        idx_random = [idx[x] for x in idx_p]\n\n        total = len(idx_random)\n        idx_70 = int(total * 0.70)\n        idx_90 = int(total * 0.90)\n        \n        idx_frac_70 = idx_random[:idx_70] \n        idx_frac_20 = idx_random[idx_70:idx_90]\n        idx_frac_10 = idx_random[idx_90:] \n        \n        #print(f\"Total: {len(idx_frac_70) + len(idx_frac_20) + len(idx_frac_10)} itens\")\n    \n        if dataset_type == \"train\":\n            self.labels = list(self.df_labels.loc[idx_frac_70, 'label'])\n            self.full_filenames = [self.path_dataset_train / f\"{f}.tif\" for f in idx_frac_70]\n            print(f\"(70%): {len(idx_frac_70)} itens\")\n            print(\"training dataset\")\n            \n        elif dataset_type == \"val\":\n            self.labels = list(self.df_labels.loc[idx_frac_20, 'label'])\n            self.full_filenames = [self.path_dataset_train / f\"{f}.tif\" for f in idx_frac_20]\n            print(f\"(20%): {len(idx_frac_20)} itens\")\n            print(\"validation dataset\")\n            \n        elif dataset_type == \"test\":\n            self.labels = list(self.df_labels.loc[idx_frac_10, 'label'])\n            self.full_filenames = [self.path_dataset_train / f\"{f}.tif\" for f in idx_frac_10]\n            print(\"testing dataset\")\n            print(f\"(10%): {len(idx_frac_10)} itens\")\n            \n        else:\n            raise Exception('Fail')\n\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.full_filenames)\n\n    def __getitem__(self, idx):\n        img = Image.open(self.full_filenames[idx]) # PIL image\n        img = self.transform(img)\n        \n        return img, self.labels[idx]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:01.383553Z","iopub.execute_input":"2025-12-20T14:26:01.383805Z","iopub.status.idle":"2025-12-20T14:26:01.409313Z","shell.execute_reply.started":"2025-12-20T14:26:01.383754Z","shell.execute_reply":"2025-12-20T14:26:01.408723Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision import transforms\n\nmean = [0.485, 0.456, 0.406]\nstd = [0.229, 0.224, 0.225]\ntransform_train = transforms.Compose([\n                               transforms.Resize((224, 224)),\n                               transforms.RandomHorizontalFlip(),\n                               transforms.RandomRotation(degrees=5),\n                               transforms.ToTensor(),\n                               transforms.Normalize(mean, std)\n                               ])\n\ntransform_pos_processing = transforms.Compose([\n                               transforms.Resize((224, 224)),\n                               transforms.ToTensor(),\n                               transforms.Normalize(mean, std)\n                               ])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:01.411672Z","iopub.execute_input":"2025-12-20T14:26:01.411981Z","iopub.status.idle":"2025-12-20T14:26:01.426574Z","shell.execute_reply.started":"2025-12-20T14:26:01.411960Z","shell.execute_reply":"2025-12-20T14:26:01.425936Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"path_data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:01.427266Z","iopub.execute_input":"2025-12-20T14:26:01.428093Z","iopub.status.idle":"2025-12-20T14:26:01.440301Z","shell.execute_reply.started":"2025-12-20T14:26:01.428072Z","shell.execute_reply":"2025-12-20T14:26:01.439606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"str(next(Path(path_data).iterdir()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:01.441138Z","iopub.execute_input":"2025-12-20T14:26:01.441378Z","iopub.status.idle":"2025-12-20T14:26:01.451439Z","shell.execute_reply.started":"2025-12-20T14:26:01.441352Z","shell.execute_reply":"2025-12-20T14:26:01.450883Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"training_set = CancerDataset(\n    path_to_dataset=path_data,\n    transform=transform_train,\n    dataset_type=\"train\"\n)\n \nvalidation_set = CancerDataset(\n    path_to_dataset=path_data,\n    transform=transform_pos_processing,\n    dataset_type=\"val\"\n)\n\ntest_set = CancerDataset(\n    path_to_dataset=path_data,\n    transform=transform_pos_processing,\n    dataset_type=\"test\"\n)\n\nprint(f'training dataset length: {len(training_set)}')\nprint(f'validation dataset length: {len(validation_set)}')\nprint(f'test dataset length: {len(test_set)}')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:01.452259Z","iopub.execute_input":"2025-12-20T14:26:01.452525Z","iopub.status.idle":"2025-12-20T14:26:03.934172Z","shell.execute_reply.started":"2025-12-20T14:26:01.452497Z","shell.execute_reply":"2025-12-20T14:26:03.933422Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_loader = torch.utils.data.DataLoader(\n    training_set,\n    batch_size=10,\n    shuffle=True,\n    num_workers=2)\n\ntest_loader = torch.utils.data.DataLoader(\n    test_set,\n    batch_size=10,\n    shuffle=False,\n    num_workers=2)\n\nN_CLASSES: int = 2\nBATCH_SIZE: int = 30\nLEARNING_RATE: float = 3e-4\nN_EPOCHS: int = 2\nprint(\"done\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:03.935052Z","iopub.execute_input":"2025-12-20T14:26:03.935320Z","iopub.status.idle":"2025-12-20T14:26:03.940431Z","shell.execute_reply.started":"2025-12-20T14:26:03.935296Z","shell.execute_reply":"2025-12-20T14:26:03.939832Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision import models\n\nmodel = models.resnet34(\n    pretrained=True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:03.941217Z","iopub.execute_input":"2025-12-20T14:26:03.941452Z","iopub.status.idle":"2025-12-20T14:26:04.884529Z","shell.execute_reply.started":"2025-12-20T14:26:03.941433Z","shell.execute_reply":"2025-12-20T14:26:04.883707Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.nn import CrossEntropyLoss\n\ncriterion = CrossEntropyLoss()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:04.885452Z","iopub.execute_input":"2025-12-20T14:26:04.885697Z","iopub.status.idle":"2025-12-20T14:26:04.889580Z","shell.execute_reply.started":"2025-12-20T14:26:04.885676Z","shell.execute_reply":"2025-12-20T14:26:04.889019Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torch.optim import Adam\n\noptimizer: Adam = Adam(\n    model.parameters(),\n    lr=LEARNING_RATE\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:04.890116Z","iopub.execute_input":"2025-12-20T14:26:04.890306Z","iopub.status.idle":"2025-12-20T14:26:04.904876Z","shell.execute_reply.started":"2025-12-20T14:26:04.890288Z","shell.execute_reply":"2025-12-20T14:26:04.904198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"for param in model.parameters():\n    param.requires_grad = False\n\nnum_ftrs = model.fc.in_features\nmodel.fc = nn.Linear(num_ftrs, N_CLASSES)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:04.905739Z","iopub.execute_input":"2025-12-20T14:26:04.906042Z","iopub.status.idle":"2025-12-20T14:26:04.919708Z","shell.execute_reply.started":"2025-12-20T14:26:04.906016Z","shell.execute_reply":"2025-12-20T14:26:04.918897Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Train and Test the Model","metadata":{}},{"cell_type":"code","source":"for epoch in range(N_EPOCHS):\n    running_loss = 0.0\n    for i, data in enumerate(tqdm(train_loader)):\n        inputs, labels = data\n        optimizer.zero_grad()\n        outputs = model(inputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optimizer.step()\n        running_loss += loss.item()\n\n        if i % 100 == 99:\n            tqdm.write('[%d, %5d] loss: %.3f' % (epoch + 1, i + 1, running_loss / 100))\n            running_loss = 0.0\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:26:04.920642Z","iopub.execute_input":"2025-12-20T14:26:04.921060Z","iopub.status.idle":"2025-12-20T14:32:09.644118Z","shell.execute_reply.started":"2025-12-20T14:26:04.921025Z","shell.execute_reply":"2025-12-20T14:32:09.643013Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"try:\n    correct = 0\n    total = 0\n    \n    with torch.no_grad():\n        for data in tqdm(test_loader):\n            images, labels = data\n            outputs = model(images)\n            _, predicted = torch.max(outputs.data, 1)\n            total += labels.size(0)\n            correct += (predicted == labels).sum().item()\n    \n    print('Accuracy of the network on the test images: %d %%' % (100 * correct / total))\nexcept Exception as e:\n    print(e, flush=True)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-20T14:32:09.645515Z","iopub.status.idle":"2025-12-20T14:32:09.645825Z","shell.execute_reply.started":"2025-12-20T14:32:09.645698Z","shell.execute_reply":"2025-12-20T14:32:09.645714Z"}},"outputs":[],"execution_count":null}]}