{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nroot_dir = '/kaggle/input/histopathologic-cancer-detection'\ndirs = os.listdir(root_dir)\nprint(dirs)\ntrain_files = os.listdir(os.path.join(root_dir, \"train\"))\nprint(len(train_files))\nprint(train_files[0])\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-07-08T01:52:48.57843Z","iopub.execute_input":"2022-07-08T01:52:48.57877Z","iopub.status.idle":"2022-07-08T01:52:48.755171Z","shell.execute_reply.started":"2022-07-08T01:52:48.578741Z","shell.execute_reply":"2022-07-08T01:52:48.754084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\n\ndf = pd.read_csv(os.path.join(root_dir, 'train_labels.csv'))\nprint(df.head())\nprint(df.size)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-08T01:52:49.756437Z","iopub.execute_input":"2022-07-08T01:52:49.756852Z","iopub.status.idle":"2022-07-08T01:52:50.036061Z","shell.execute_reply.started":"2022-07-08T01:52:49.756814Z","shell.execute_reply":"2022-07-08T01:52:50.035074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport numpy as np\nimport matplotlib.pyplot as plt\n\ndef show_single_image(image_name):\n    #extract label\n    file_name = train_files[0].split(\".\")[0]\n    label_df = df[df['id']==file_name]\n    label_df = label_df.reset_index()\n\n    #show the image\n    img = Image.open(os.path.join(root_dir, 'train', train_files[0]))\n    img = np.array(img)\n    plt.imshow(img)\n    plt.xlabel(label_df['label'][0])\n    plt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-08T01:52:51.097586Z","iopub.execute_input":"2022-07-08T01:52:51.098654Z","iopub.status.idle":"2022-07-08T01:52:51.105689Z","shell.execute_reply.started":"2022-07-08T01:52:51.098609Z","shell.execute_reply":"2022-07-08T01:52:51.104771Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"show_single_image(train_files[0])","metadata":{"execution":{"iopub.status.busy":"2022-07-08T01:52:52.212631Z","iopub.execute_input":"2022-07-08T01:52:52.213467Z","iopub.status.idle":"2022-07-08T01:52:52.448074Z","shell.execute_reply.started":"2022-07-08T01:52:52.213418Z","shell.execute_reply":"2022-07-08T01:52:52.447186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from torch.utils.data import Dataset\nimport torchvision\nimport torch\n\nclass ImageDataset(Dataset):\n    def __init__(self, images, csv_files, transforms = torchvision.transforms.ToTensor()):\n        super(ImageDataset, self).__init__()\n        self.transforms = transforms\n        self.images = images\n        self.csv_files = csv_files\n        \n    def __getitem__(self, id):\n        file_name = train_files[id].split(\".\")[0]\n        label_df = df[df['id']==file_name]\n        label_df = label_df.reset_index()\n\n        #show the image\n        img = Image.open(os.path.join(root_dir, 'train', train_files[id]))\n        img = img.resize((96, 96))\n        img = np.array(img)\n        img = self.transforms(img)\n        label = torch.tensor(int(label_df['label'][0]))\n#         label = self.transforms(label_df['label'][0])\n        \n        return img, label\n    \n    def __len__(self):\n        return len(self.images)\n        ","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:40:14.44416Z","iopub.execute_input":"2022-07-08T03:40:14.444904Z","iopub.status.idle":"2022-07-08T03:40:14.460515Z","shell.execute_reply.started":"2022-07-08T03:40:14.444862Z","shell.execute_reply":"2022-07-08T03:40:14.459395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = ImageDataset(train_files, df)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:40:15.467092Z","iopub.execute_input":"2022-07-08T03:40:15.467766Z","iopub.status.idle":"2022-07-08T03:40:15.47185Z","shell.execute_reply.started":"2022-07-08T03:40:15.467728Z","shell.execute_reply":"2022-07-08T03:40:15.47094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img, label = dataset[100]\nimg = img.cpu().detach().numpy()\nimg = img.swapaxes(0, 1)\nimg = img.swapaxes(1, 2)\nprint(img.shape)\nplt.imshow(img)\nplt.xlabel(label)\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:40:16.250279Z","iopub.execute_input":"2022-07-08T03:40:16.251091Z","iopub.status.idle":"2022-07-08T03:40:16.477623Z","shell.execute_reply.started":"2022-07-08T03:40:16.251045Z","shell.execute_reply":"2022-07-08T03:40:16.476698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"indices = torch.randperm(len(dataset)).tolist()\ndataset_train = torch.utils.data.Subset(dataset, indices[:-5000])\ndataset_test = torch.utils.data.Subset(dataset, indices[-5000:])\n\nprint(len(dataset_train))\nprint(len(dataset_test))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:40:17.663642Z","iopub.execute_input":"2022-07-08T03:40:17.664574Z","iopub.status.idle":"2022-07-08T03:40:17.689625Z","shell.execute_reply.started":"2022-07-08T03:40:17.664524Z","shell.execute_reply":"2022-07-08T03:40:17.688551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataloader = torch.utils.data.DataLoader(dataset=dataset_train, batch_size=4, shuffle=True)\nval_dataloader = torch.utils.data.DataLoader(dataset=dataset_test, batch_size=4)\n\n# dataset_iter = iter(dataset)\n\n# for data in iter(dataset_iter):\n#     print(data)\n#     break\n    \ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:40:18.488961Z","iopub.execute_input":"2022-07-08T03:40:18.489687Z","iopub.status.idle":"2022-07-08T03:40:18.537144Z","shell.execute_reply.started":"2022-07-08T03:40:18.489646Z","shell.execute_reply":"2022-07-08T03:40:18.5363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch.nn as nn\nimport torch.nn.functional as F\n\nclass ClassificationModel(nn.Module):\n    def __init__(self):\n        super(ClassificationModel, self).__init__()\n        self.conv1 = nn.Conv2d(in_channels=3, out_channels=10, kernel_size=5)\n        self.pool1 = nn.MaxPool2d(kernel_size=2, stride=2)\n        self.conv2 = nn.Conv2d(in_channels=10, out_channels=15, kernel_size=5)\n        self.conv3 = nn.Conv2d(in_channels=15, out_channels=20, kernel_size=5)\n        self.fc1 = nn.Linear(in_features=8*8*20, out_features=100)\n        self.fc2 = nn.Linear(in_features=100, out_features=2)\n        \n    def forward(self, x):\n        x = self.pool1(F.relu(self.conv1(x)))\n        x = self.pool1(F.relu(self.conv2(x)))\n        x = self.pool1(F.relu(self.conv3(x)))\n        x = torch.flatten(x, 1)\n        x = F.relu(self.fc1(x))\n        x = self.fc2(x)\n#         x = torch.sigmoid(x)\n        \n        return x","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:40:19.385456Z","iopub.execute_input":"2022-07-08T03:40:19.385865Z","iopub.status.idle":"2022-07-08T03:40:19.395671Z","shell.execute_reply.started":"2022-07-08T03:40:19.385834Z","shell.execute_reply":"2022-07-08T03:40:19.394731Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = ClassificationModel()\nx = torch.randn(4, 3, 96, 96)\nprint(model(x))\n\nprint(device)\nmodel.to(torch.device('cuda:0'))","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:51:43.717462Z","iopub.execute_input":"2022-07-08T03:51:43.717839Z","iopub.status.idle":"2022-07-08T03:51:43.762757Z","shell.execute_reply.started":"2022-07-08T03:51:43.717807Z","shell.execute_reply":"2022-07-08T03:51:43.761311Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"torch.cuda.device_count()\n\ndef try_gpu(i=0):\n    if torch.cuda.device_count() >= i +1:\n        return torch.device(f'cuda:{i}')\n    else:\n        return torch.device('cpu')\n    \ntry_gpu()","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:52:44.998133Z","iopub.execute_input":"2022-07-08T03:52:44.998845Z","iopub.status.idle":"2022-07-08T03:52:45.006944Z","shell.execute_reply.started":"2022-07-08T03:52:44.998808Z","shell.execute_reply":"2022-07-08T03:52:45.005861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(model)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:40:21.261857Z","iopub.execute_input":"2022-07-08T03:40:21.26275Z","iopub.status.idle":"2022-07-08T03:40:21.268621Z","shell.execute_reply.started":"2022-07-08T03:40:21.262699Z","shell.execute_reply":"2022-07-08T03:40:21.267636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"criterion = nn.CrossEntropyLoss()\noptim = torch.optim.SGD(model.parameters(), lr=0.001, momentum = 0.9)","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:40:22.336858Z","iopub.execute_input":"2022-07-08T03:40:22.337848Z","iopub.status.idle":"2022-07-08T03:40:22.343511Z","shell.execute_reply.started":"2022-07-08T03:40:22.337798Z","shell.execute_reply":"2022-07-08T03:40:22.342551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for epoch in range(3):\n    running_loss = 0.0 \n    for i, data in enumerate(train_dataloader, 0):\n        inputs, labels = data\n#         inputs, labels = inputs.to(device), labels.to(device)\n        optim.zero_grad()\n        outputs = model(inputs)\n#         print(outputs)\n        loss = criterion(outputs, labels)\n        loss.backward()\n        optim.step()\n        \n        running_loss += loss.item()\n        \n        if i%100 == 0:\n            print(f'[{epoch + 1}, {i + 1:5d}] loss: {running_loss:.3f}')\n            running_loss = 0.0\n            \nprint(\"Finished\")","metadata":{"execution":{"iopub.status.busy":"2022-07-08T03:40:23.18748Z","iopub.execute_input":"2022-07-08T03:40:23.187831Z","iopub.status.idle":"2022-07-08T03:51:26.169296Z","shell.execute_reply.started":"2022-07-08T03:40:23.187802Z","shell.execute_reply":"2022-07-08T03:51:26.167581Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}