{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-01-28T19:25:15.289281Z","iopub.execute_input":"2023-01-28T19:25:15.289655Z","iopub.status.idle":"2023-01-28T19:25:19.229784Z","shell.execute_reply.started":"2023-01-28T19:25:15.289627Z","shell.execute_reply":"2023-01-28T19:25:19.227406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:43:20.370284Z","iopub.execute_input":"2023-02-01T20:43:20.370762Z","iopub.status.idle":"2023-02-01T20:43:20.377943Z","shell.execute_reply.started":"2023-02-01T20:43:20.370730Z","shell.execute_reply":"2023-02-01T20:43:20.375882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import DataLoader, Dataset","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:43:20.513405Z","iopub.execute_input":"2023-02-01T20:43:20.515190Z","iopub.status.idle":"2023-02-01T20:43:22.566603Z","shell.execute_reply.started":"2023-02-01T20:43:20.515120Z","shell.execute_reply":"2023-02-01T20:43:22.565212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"csvpath = '/kaggle/input/histopathologic-cancer-detection/train_labels.csv'\nlabels = pd.read_csv(csvpath)\nlabels.head()","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:43:22.569080Z","iopub.execute_input":"2023-02-01T20:43:22.569687Z","iopub.status.idle":"2023-02-01T20:43:23.239901Z","shell.execute_reply.started":"2023-02-01T20:43:22.569636Z","shell.execute_reply":"2023-02-01T20:43:23.238284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(labels.columns.values)","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:43:23.242026Z","iopub.execute_input":"2023-02-01T20:43:23.242630Z","iopub.status.idle":"2023-02-01T20:43:23.251276Z","shell.execute_reply.started":"2023-02-01T20:43:23.242575Z","shell.execute_reply":"2023-02-01T20:43:23.250004Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(labels.iloc[:,1].value_counts())","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:43:23.255611Z","iopub.execute_input":"2023-02-01T20:43:23.256014Z","iopub.status.idle":"2023-02-01T20:43:23.275331Z","shell.execute_reply.started":"2023-02-01T20:43:23.255980Z","shell.execute_reply":"2023-02-01T20:43:23.273542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\nlabels['label'].hist()","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:43:23.277784Z","iopub.execute_input":"2023-02-01T20:43:23.278423Z","iopub.status.idle":"2023-02-01T20:43:23.587218Z","shell.execute_reply.started":"2023-02-01T20:43:23.278354Z","shell.execute_reply":"2023-02-01T20:43:23.585781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom PIL import Image, ImageDraw\nimport numpy as np\nimport os","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:43:23.589242Z","iopub.execute_input":"2023-02-01T20:43:23.589676Z","iopub.status.idle":"2023-02-01T20:43:23.602918Z","shell.execute_reply.started":"2023-02-01T20:43:23.589639Z","shell.execute_reply":"2023-02-01T20:43:23.601266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%matplotlib inline\n# get ids for malignant images\n\nmalignantIDs = []\nmalignantIDs = labels.loc[labels['label'] == 1]['id'].values\n# or malignantIDs = [labels.iloc[i,0] for i in range(len(labels)) if labels.iloc[i, 1] == 1]\npath2train = \"/kaggle/input/histopathologic-cancer-detection/train/\"\n# to show images in grayscale\ncolor = False","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:43:23.604978Z","iopub.execute_input":"2023-02-01T20:43:23.606341Z","iopub.status.idle":"2023-02-01T20:43:23.626943Z","shell.execute_reply.started":"2023-02-01T20:43:23.606279Z","shell.execute_reply":"2023-02-01T20:43:23.625542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.rcParams['figure.figsize'] = (10.0, 10.0)\nplt.subplots_adjust(wspace = 0, hspace = 0)\nnrows, ncols = 3, 3","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:43:23.628189Z","iopub.execute_input":"2023-02-01T20:43:23.628600Z","iopub.status.idle":"2023-02-01T20:43:23.639177Z","shell.execute_reply.started":"2023-02-01T20:43:23.628566Z","shell.execute_reply":"2023-02-01T20:43:23.637754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i, id_ in enumerate(malignantIDs[:nrows*ncols]):\n    file_names = os.path.join(path2train, id_ + '.tif')\n    # load image\n    img = Image.open(file_names)\n    # draw a 32*32 rectangle\n    draw = ImageDraw.Draw(img)\n    draw.rectangle(((32, 32), (64, 64)), outline = \"green\")\n    plt.subplot(nrows, ncols, i + 1)\n    if color:\n        plt.imshow(np.array(img))\n    else:\n        plt.imshow(np.array(img)[:, :])\n    plt.axis(\"off\")\n    \n    # images' shape and minimum and maximum pixel values\n    print(\"image shape:\", np.array(img).shape)\n    print(\"pixel alues range from %s to %s\" %(np.min(img), np.max(img)))","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:43:23.640898Z","iopub.execute_input":"2023-02-01T20:43:23.641273Z","iopub.status.idle":"2023-02-01T20:43:24.404531Z","shell.execute_reply.started":"2023-02-01T20:43:23.641239Z","shell.execute_reply":"2023-02-01T20:43:24.403493Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating costum dataset\nfrom PIL import Image\nimport torch\nimport torchvision.transforms as transforms\nfrom torch.utils.data import Dataset\nimport pandas as pd\ntorch.manual_seed(0)","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:43:24.407383Z","iopub.execute_input":"2023-02-01T20:43:24.407956Z","iopub.status.idle":"2023-02-01T20:43:24.691898Z","shell.execute_reply.started":"2023-02-01T20:43:24.407923Z","shell.execute_reply":"2023-02-01T20:43:24.690517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Cancerhist(Dataset):\n    def __init__(self, data_dir, transform, data_type = \"train\"):\n        # default arguments should be last arguments of the function --> nothing non-default should not come after them\n        # path to image\n        pth2img = os.path.join(data_dir, data_type)\n        # get list of images\n        filenames = os.listdir(pth2img)\n        self.imgpth = [os.path.join(pth2img, f) for f in filenames]\n        csv = data_type + \"_labels.csv\" # or csv = \"train_labels.csv\"\n        csvpth = os.path.join(data_dir, csv)\n        # get lables for the images\n        labels_df = pd.read_csv(csvpth)\n        labels_df.set_index(\"id\", inplace = True)\n        self.labels = [labels_df.loc[filename[:-4]].values[0] for filename in filenames]\n        self.transform = transform\n    def __len__(self):\n        return len(self.imgpth)\n    def __getitem__(self, idx):\n        img = Image.open(self.imgpth[idx])\n        img = self.transform(img)\n        return img, self.labels[idx]","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:43:24.693219Z","iopub.execute_input":"2023-02-01T20:43:24.693589Z","iopub.status.idle":"2023-02-01T20:43:24.704617Z","shell.execute_reply.started":"2023-02-01T20:43:24.693556Z","shell.execute_reply":"2023-02-01T20:43:24.702980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_transformer = transforms.Compose([transforms.ToTensor()])","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:43:24.706091Z","iopub.execute_input":"2023-02-01T20:43:24.706597Z","iopub.status.idle":"2023-02-01T20:43:24.722735Z","shell.execute_reply.started":"2023-02-01T20:43:24.706562Z","shell.execute_reply":"2023-02-01T20:43:24.721130Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_dir = \"/kaggle/input/histopathologic-cancer-detection/\"\nhistdataset = Cancerhist(data_dir, data_transformer, \"train\")","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:43:24.725086Z","iopub.execute_input":"2023-02-01T20:43:24.725644Z","iopub.status.idle":"2023-02-01T20:44:05.938079Z","shell.execute_reply.started":"2023-02-01T20:43:24.725559Z","shell.execute_reply":"2023-02-01T20:44:05.936857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(len(histdataset))","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:44:05.939600Z","iopub.execute_input":"2023-02-01T20:44:05.939994Z","iopub.status.idle":"2023-02-01T20:44:05.946565Z","shell.execute_reply.started":"2023-02-01T20:44:05.939962Z","shell.execute_reply":"2023-02-01T20:44:05.944897Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img, label = histdataset[9]\nprint(img.shape, torch.min(img), torch.max(img))\n","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:44:05.947824Z","iopub.execute_input":"2023-02-01T20:44:05.948642Z","iopub.status.idle":"2023-02-01T20:44:05.975972Z","shell.execute_reply.started":"2023-02-01T20:44:05.948607Z","shell.execute_reply":"2023-02-01T20:44:05.974783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(img.permute(2, 1, 0))","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:44:05.977341Z","iopub.execute_input":"2023-02-01T20:44:05.978649Z","iopub.status.idle":"2023-02-01T20:44:06.166857Z","shell.execute_reply.started":"2023-02-01T20:44:05.978611Z","shell.execute_reply":"2023-02-01T20:44:06.165678Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# let's split the dataset\nfrom torch.utils.data import random_split\n\nlenhist = len(histdataset)\nlentrain = int(0.8 * lenhist)\nlenval = lenhist - lentrain","metadata":{"execution":{"iopub.status.busy":"2023-02-01T20:55:57.657642Z","iopub.execute_input":"2023-02-01T20:55:57.658154Z","iopub.status.idle":"2023-02-01T20:55:57.664806Z","shell.execute_reply.started":"2023-02-01T20:55:57.658118Z","shell.execute_reply":"2023-02-01T20:55:57.663439Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds, val_ds = random_split(histdataset, [lentrain, lenval])\nprint(f\"train dataset length: {len(train_ds)}\")\nprint(f\"validation dataset length: {len(train_ds)}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-01T21:24:19.075352Z","iopub.execute_input":"2023-02-01T21:24:19.075822Z","iopub.status.idle":"2023-02-01T21:24:19.108406Z","shell.execute_reply.started":"2023-02-01T21:24:19.075788Z","shell.execute_reply":"2023-02-01T21:24:19.107152Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get an example\nfor x, y in train_ds:\n    print(x.shape, y)\n    plt.imshow(x.permute(1, 2, 0))\n    break","metadata":{"execution":{"iopub.status.busy":"2023-02-01T21:26:28.595497Z","iopub.execute_input":"2023-02-01T21:26:28.596233Z","iopub.status.idle":"2023-02-01T21:26:28.854849Z","shell.execute_reply.started":"2023-02-01T21:26:28.596189Z","shell.execute_reply":"2023-02-01T21:26:28.853562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get an example from the validation dataset\nfor x, y in val_ds:\n    print(x.shape, y)\n    plt.imshow(x.permute(1, 2, 0))\n    break","metadata":{"execution":{"iopub.status.busy":"2023-02-01T21:27:11.089904Z","iopub.execute_input":"2023-02-01T21:27:11.090441Z","iopub.status.idle":"2023-02-01T21:27:11.336436Z","shell.execute_reply.started":"2023-02-01T21:27:11.090385Z","shell.execute_reply":"2023-02-01T21:27:11.335351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# display a few samples from train_ds\nfrom torchvision import utils\nnp.random.seed(0)","metadata":{"execution":{"iopub.status.busy":"2023-02-01T21:30:11.085168Z","iopub.execute_input":"2023-02-01T21:30:11.085707Z","iopub.status.idle":"2023-02-01T21:30:11.091267Z","shell.execute_reply.started":"2023-02-01T21:30:11.085668Z","shell.execute_reply":"2023-02-01T21:30:11.089689Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show(img, y, color = False):\n    # convert to numpy\n    npimg = img.numpy()\n    npimg_T = np.transpose(npimg, (1, 2, 0))\n    if color == False:\n        npimg_T = npimg_T[:, :, 0]\n        plt.imshow(npimg_T, interpolation = \"nearest\", cmap = \"gray\")\n    else:\n        plt.imshow(npimg_T, interpolation = \"nearest\")\n    plt.title(f\"label: {y}\")","metadata":{"execution":{"iopub.status.busy":"2023-02-01T21:48:03.680048Z","iopub.execute_input":"2023-02-01T21:48:03.680552Z","iopub.status.idle":"2023-02-01T21:48:03.689678Z","shell.execute_reply.started":"2023-02-01T21:48:03.680517Z","shell.execute_reply":"2023-02-01T21:48:03.688297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create a grid of samples \ngrid_size = 4\nrnd_ind = np.random.randint(0, len(train_ds), grid_size)\nprint(f\"image indices: {rnd_ind}\")\nx_grid_train = [train_ds[i][0] for i in rnd_ind]\ny_grid_train = [train_ds[i][1] for i in rnd_ind]\n\nx_grid_train = utils.make_grid(x_grid_train, nrow = 4, padding = 2)\nprint(x_grid_train.shape)","metadata":{"execution":{"iopub.status.busy":"2023-02-01T21:48:05.736446Z","iopub.execute_input":"2023-02-01T21:48:05.736880Z","iopub.status.idle":"2023-02-01T21:48:05.810922Z","shell.execute_reply.started":"2023-02-01T21:48:05.736847Z","shell.execute_reply":"2023-02-01T21:48:05.809771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.rcParams['figure.figsize'] = (12.0, 7)\nshow(x_grid_train, y_grid_train)\n","metadata":{"execution":{"iopub.status.busy":"2023-02-01T21:56:09.773259Z","iopub.execute_input":"2023-02-01T21:56:09.774104Z","iopub.status.idle":"2023-02-01T21:56:10.010192Z","shell.execute_reply.started":"2023-02-01T21:56:09.774047Z","shell.execute_reply":"2023-02-01T21:56:10.008760Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create a grid of samples from validation dataset\nrand_ind_val = np.random.randint(0, len(val_ds), grid_size)\nx_grid_val = [val_ds[i][0] for i in rand_ind_val]\ny_grid_val = [val_ds[i][1] for i in rand_ind_val]\n\nx_grid_val = utils.make_grid(x_grid_val, nrow = 4, padding = 2)\nprint(x_grid_val.shape)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-02-01T21:52:24.325372Z","iopub.execute_input":"2023-02-01T21:52:24.325954Z","iopub.status.idle":"2023-02-01T21:52:24.406943Z","shell.execute_reply.started":"2023-02-01T21:52:24.325912Z","shell.execute_reply":"2023-02-01T21:52:24.405101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.rcParams['figure.figsize'] = (12.0, 7)\nshow(x_grid_val, y_grid_val)","metadata":{"execution":{"iopub.status.busy":"2023-02-01T21:56:04.703149Z","iopub.execute_input":"2023-02-01T21:56:04.705883Z","iopub.status.idle":"2023-02-01T21:56:04.942475Z","shell.execute_reply.started":"2023-02-01T21:56:04.705801Z","shell.execute_reply":"2023-02-01T21:56:04.941167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transforming the data\ntrain_transformer = transforms.Compose([\n    transforms.RandomHorizontalFlip(p = 0.5),\n    transforms.RandomVerticalFlip(p = 0.5),\n    transforms.RandomRotation(45),\n    transforms.RandomResizedCrop(96, scale = (0.8, 1.0), ratio = (1.0 , 1.0)),\n    transforms.ToTensor()\n])\n\nval_transformer = transforms.Compose([\n    transforms.ToTensor()\n])\n\ntrain_ds.transform = train_transformer\nval_ds.transform = val_transformer","metadata":{"execution":{"iopub.status.busy":"2023-02-01T22:01:08.268038Z","iopub.execute_input":"2023-02-01T22:01:08.268553Z","iopub.status.idle":"2023-02-01T22:01:08.277085Z","shell.execute_reply.started":"2023-02-01T22:01:08.268517Z","shell.execute_reply":"2023-02-01T22:01:08.275663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# creating Dataloaders\nfrom torch.utils.data import DataLoader\ntrain_dl = DataLoader(train_ds, batch_size = 32, shuffle = True)\nval_dl = DataLoader(val_ds, batch_size = 64, shuffle = False)\n\n# to check a batch on the data\nfor x, y in train_dl:\n    print(x.shape, y.shape)\n    break\n\nfor x, y in val_dl:\n    print(x.shape, y.shape)\n    break\n","metadata":{"execution":{"iopub.status.busy":"2023-02-01T22:06:12.328991Z","iopub.execute_input":"2023-02-01T22:06:12.330525Z","iopub.status.idle":"2023-02-01T22:06:13.289356Z","shell.execute_reply.started":"2023-02-01T22:06:12.330459Z","shell.execute_reply":"2023-02-01T22:06:13.288044Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}