{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpuV5e8","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31234,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics lSibraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-12-24T17:40:43.664081Z","iopub.execute_input":"2025-12-24T17:40:43.664876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"full_train_df = pd.read_csv(\"/kaggle/input/histopathologic-cancer-detection/train_labels.csv\")\nfull_train_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T17:49:56.968290Z","iopub.execute_input":"2025-12-24T17:49:56.968774Z","iopub.status.idle":"2025-12-24T17:49:57.558251Z","shell.execute_reply.started":"2025-12-24T17:49:56.968745Z","shell.execute_reply":"2025-12-24T17:49:57.557443Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Train Size: {}\".format(len(os.listdir('/kaggle/input/histopathologic-cancer-detection/train'))))\nprint(\"Test Size: {}\".format(len(os.listdir('/kaggle/input/histopathologic-cancer-detection/test'))))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T18:54:55.829895Z","iopub.execute_input":"2025-12-24T18:54:55.830239Z","iopub.status.idle":"2025-12-24T18:54:57.865188Z","shell.execute_reply.started":"2025-12-24T18:54:55.830211Z","shell.execute_reply":"2025-12-24T18:54:57.864315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import sklearn.utils\n#SAMPLING\nSAMPLE_SIZE = 80000\n\n# DATA_PATHS\ntrain_path = '/kaggle/input/histopathologic-cancer-detection/train'\ntest_path = '/kaggle/input/histopathologic-cancer-detection/test'\n\n# use 80000 P & N \ndf_negatives = full_train_df[full_train_df['label'] == 0].sample(SAMPLE_SIZE ,random_state = 42)\n\ndf_positives = full_train_df[full_train_df['label'] == 1].sample(SAMPLE_SIZE ,random_state = 42)\n\n# Concatenate the two dfs and shuffle them up\ntrain_df = sklearn.utils.shuffle(pd.concat([df_positives, df_negatives], axis=0).reset_index(drop=True))\n\ntrain_df.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T18:57:53.370639Z","iopub.execute_input":"2025-12-24T18:57:53.370976Z","iopub.status.idle":"2025-12-24T18:57:54.152370Z","shell.execute_reply.started":"2025-12-24T18:57:53.370949Z","shell.execute_reply":"2025-12-24T18:57:54.151421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nfrom torch.utils.data import Dataset, DataLoader,SubsetRandomSampler\nimport torch.nn as nn \nimport torch.optim as optim","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T18:58:22.859045Z","iopub.execute_input":"2025-12-24T18:58:22.859552Z","iopub.status.idle":"2025-12-24T18:58:31.494775Z","shell.execute_reply.started":"2025-12-24T18:58:22.859505Z","shell.execute_reply":"2025-12-24T18:58:31.494012Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Our own custom class for datasets\nclass CreateDataset(Dataset):\n    def __init__(self, df_data, data_dir = './', transform=None):\n        super().__init__()\n        self.df = df_data.values\n        self.data_dir = data_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        img_name,label = self.df[index]\n        img_path = os.path.join(self.data_dir, img_name+'.tif')\n        image = cv2.imread(img_path)\n        if self.transform is not None:\n            image = self.transform(image)\n        return image, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T18:59:21.396893Z","iopub.execute_input":"2025-12-24T18:59:21.397221Z","iopub.status.idle":"2025-12-24T18:59:21.403585Z","shell.execute_reply.started":"2025-12-24T18:59:21.397193Z","shell.execute_reply":"2025-12-24T18:59:21.402822Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from torchvision import transforms","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T18:59:39.155377Z","iopub.execute_input":"2025-12-24T18:59:39.156124Z","iopub.status.idle":"2025-12-24T18:59:44.581308Z","shell.execute_reply.started":"2025-12-24T18:59:39.156090Z","shell.execute_reply":"2025-12-24T18:59:44.580441Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transforms_train = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.RandomHorizontalFlip(p=0.4),\n    transforms.RandomVerticalFlip(p=0.4),\n    transforms.RandomRotation(20),\n    transforms.ToTensor(),\n    # We the get the following mean and std for the channels of all the images\n    #transforms.Normalize((0.70244707, 0.54624322, 0.69645334), (0.23889325, 0.28209431, 0.21625058))\n    transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5))\n])\n\ntrain_data = CreateDataset(df_data=train_df, data_dir=train_path, transform=transforms_train)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T19:00:20.120951Z","iopub.execute_input":"2025-12-24T19:00:20.121274Z","iopub.status.idle":"2025-12-24T19:00:20.143560Z","shell.execute_reply.started":"2025-12-24T19:00:20.121245Z","shell.execute_reply":"2025-12-24T19:00:20.142635Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set Batch Size\nbatch_size = 128\n\n# Percentage of training set to use as validation\nvalid_size = 0.1\n\n# obtain training indices that will be used for validation\nnum_train = len(train_data)\nindices = list(range(num_train))\n# np.random.shuffle(indices)\nsplit = int(np.floor(valid_size * num_train))\ntrain_idx, valid_idx = indices[split:], indices[:split]\n\n# Create Samplers\ntrain_sampler = SubsetRandomSampler(train_idx)\nvalid_sampler = SubsetRandomSampler(valid_idx)\n\n# prepare data loaders (combine dataset and sampler)\ntrain_loader = DataLoader(train_data, batch_size=batch_size, sampler=train_sampler)\nvalid_loader = DataLoader(train_data, batch_size=batch_size, sampler=valid_sampler)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T19:00:52.608420Z","iopub.execute_input":"2025-12-24T19:00:52.609306Z","iopub.status.idle":"2025-12-24T19:00:52.622203Z","shell.execute_reply.started":"2025-12-24T19:00:52.609275Z","shell.execute_reply":"2025-12-24T19:00:52.621256Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"transforms_test = transforms.Compose([\n    transforms.ToPILImage(),\n    transforms.ToTensor(),\n    #transforms.Normalize((0.70244707, 0.54624322, 0.69645334), (0.23889325, 0.28209431, 0.21625058))\n    transforms.Normalize((0.5, 0.5, 0.5), (0.5, 0.5, 0.5))\n])\n\n# creating test data\nsample_sub = pd.read_csv(\"/kaggle/input/histopathologic-cancer-detection/sample_submission.csv\")\ntest_data = CreateDataset(df_data=sample_sub, data_dir=test_path, transform=transforms_test)\n\n# prepare the test loader\ntest_loader = DataLoader(test_data, batch_size=batch_size, shuffle=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-24T19:01:18.248105Z","iopub.execute_input":"2025-12-24T19:01:18.248434Z","iopub.status.idle":"2025-12-24T19:01:18.312602Z","shell.execute_reply.started":"2025-12-24T19:01:18.248406Z","shell.execute_reply":"2025-12-24T19:01:18.311671Z"}},"outputs":[],"execution_count":null}]}