{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.optim import lr_scheduler\nimport torchvision\nfrom torch.utils.data import DataLoader, Dataset\nfrom torchvision import datasets, models, transforms\nimport matplotlib.pyplot as plt\nimport pickle\nimport time \nfrom skimage import io\nimport copy\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n#         break\n#     break\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-04-11T17:12:27.347397Z","iopub.execute_input":"2022-04-11T17:12:27.347797Z","iopub.status.idle":"2022-04-11T17:12:29.483097Z","shell.execute_reply.started":"2022-04-11T17:12:27.347766Z","shell.execute_reply":"2022-04-11T17:12:29.482348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\nprint(device)","metadata":{"execution":{"iopub.status.busy":"2022-04-11T17:15:28.52692Z","iopub.execute_input":"2022-04-11T17:15:28.52717Z","iopub.status.idle":"2022-04-11T17:15:28.531877Z","shell.execute_reply.started":"2022-04-11T17:15:28.527142Z","shell.execute_reply":"2022-04-11T17:15:28.531178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load data\ndf = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/train.csv')\nsample_sub = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/sample_submission.csv')\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2022-04-11T17:15:30.130728Z","iopub.execute_input":"2022-04-11T17:15:30.131352Z","iopub.status.idle":"2022-04-11T17:15:30.222196Z","shell.execute_reply.started":"2022-04-11T17:15:30.131313Z","shell.execute_reply":"2022-04-11T17:15:30.221456Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2022-04-11T17:15:32.155809Z","iopub.execute_input":"2022-04-11T17:15:32.15644Z","iopub.status.idle":"2022-04-11T17:15:32.165219Z","shell.execute_reply.started":"2022-04-11T17:15:32.156405Z","shell.execute_reply":"2022-04-11T17:15:32.164273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby('individual_id')\n","metadata":{"execution":{"iopub.status.busy":"2022-04-11T17:15:34.607034Z","iopub.execute_input":"2022-04-11T17:15:34.607576Z","iopub.status.idle":"2022-04-11T17:15:34.615693Z","shell.execute_reply.started":"2022-04-11T17:15:34.607537Z","shell.execute_reply":"2022-04-11T17:15:34.615033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EDA","metadata":{}},{"cell_type":"code","source":"# How many species\nprint(df.individual_id.unique().shape)\nresult = df.groupby('individual_id').size()\nnp.sort(result.unique())\n\n","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:21:56.703331Z","iopub.execute_input":"2022-04-11T18:21:56.703605Z","iopub.status.idle":"2022-04-11T18:21:56.74131Z","shell.execute_reply.started":"2022-04-11T18:21:56.703576Z","shell.execute_reply":"2022-04-11T18:21:56.740586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create int label for individual id\nids = list(df['individual_id'].unique() )\ndf['label'] = df['individual_id'].apply(lambda x: int(ids.index(x)))\n","metadata":{"execution":{"iopub.status.busy":"2022-04-11T17:24:52.695016Z","iopub.execute_input":"2022-04-11T17:24:52.695285Z","iopub.status.idle":"2022-04-11T17:24:57.533266Z","shell.execute_reply.started":"2022-04-11T17:24:52.695256Z","shell.execute_reply":"2022-04-11T17:24:57.532612Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# check the ids are produced coorectly\nprint(df.label.unique())\nprint(len(ids))","metadata":{"execution":{"iopub.status.busy":"2022-04-11T17:17:50.431494Z","iopub.execute_input":"2022-04-11T17:17:50.432188Z","iopub.status.idle":"2022-04-11T17:17:50.438779Z","shell.execute_reply.started":"2022-04-11T17:17:50.432149Z","shell.execute_reply":"2022-04-11T17:17:50.437921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data directory\ndata_dir = '/kaggle/input/happy-whale-and-dolphin/train_images'\n# check image if RGB or not\n# for root, dirs, files in os.walk(\"/kaggle/input/happy-whale-and-dolphin/train_images\", topdown=False):\n#    for name in files:\n#       fname = os.path.join(root, name)\n#       image = Image.open(fname)\n#       if image.mode!='RGB':\n#             print(image.mode)\n#             image.convert('RGB')\n      ","metadata":{"execution":{"iopub.status.busy":"2022-04-11T17:17:57.31358Z","iopub.execute_input":"2022-04-11T17:17:57.314176Z","iopub.status.idle":"2022-04-11T17:17:57.318318Z","shell.execute_reply.started":"2022-04-11T17:17:57.314136Z","shell.execute_reply":"2022-04-11T17:17:57.317407Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create Transforms\n\ndata_means = [0.485, 0.456, 0.406]\ndata_stds = [0.229, 0.224, 0.225]\ndata_transforms = transforms.Compose([\n        transforms.Resize((256,256)),\n        transforms.RandomHorizontalFlip(),\n        transforms.ToTensor(),\n        transforms.Normalize(data_means, data_stds)\n    ])\n","metadata":{"execution":{"iopub.status.busy":"2022-04-11T17:25:07.729376Z","iopub.execute_input":"2022-04-11T17:25:07.729947Z","iopub.status.idle":"2022-04-11T17:25:07.734752Z","shell.execute_reply.started":"2022-04-11T17:25:07.72991Z","shell.execute_reply":"2022-04-11T17:25:07.734022Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nclass CustomImageDataset(Dataset):\n    def __init__(self, df, root_dir, transform=None):\n        self.annotations = df\n        self.root_dir = root_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.annotations)\n\n    def __getitem__(self, index):\n        img_path = os.path.join(self.root_dir, self.annotations.iloc[index,0])\n        image = Image.open(img_path)\n        if image.mode != 'RGB':\n            image = image.convert('RGB')\n        label = torch.tensor(int(self.annotations.iloc[index,3]))\n\n        if self.transform:\n            image = self.transform(image)\n        return image, label;\n        \n        \n\n","metadata":{"execution":{"iopub.status.busy":"2022-04-11T17:25:08.69578Z","iopub.execute_input":"2022-04-11T17:25:08.696469Z","iopub.status.idle":"2022-04-11T17:25:08.704424Z","shell.execute_reply.started":"2022-04-11T17:25:08.696435Z","shell.execute_reply":"2022-04-11T17:25:08.703544Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"batch_size = 32\nin_channel = 3\nnum_classes = len(ids)\nlearning_rate = 1e-3\nnum_epochs = 10\n\ndataset = CustomImageDataset(df, data_dir, data_transforms)\ntrain_dataset, val_dataset = torch.utils.data.random_split(dataset,[41033,10000])\ntrain_loader = DataLoader(dataset=train_dataset, batch_size=batch_size, shuffle=True)\nval_loader = DataLoader(dataset=val_dataset, batch_size=batch_size, shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:28:27.824104Z","iopub.execute_input":"2022-04-11T18:28:27.824917Z","iopub.status.idle":"2022-04-11T18:28:27.843617Z","shell.execute_reply.started":"2022-04-11T18:28:27.824872Z","shell.execute_reply":"2022-04-11T18:28:27.842948Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nlen(ids)","metadata":{"execution":{"iopub.status.busy":"2022-04-11T17:44:05.954059Z","iopub.execute_input":"2022-04-11T17:44:05.954577Z","iopub.status.idle":"2022-04-11T17:44:05.959607Z","shell.execute_reply.started":"2022-04-11T17:44:05.95454Z","shell.execute_reply":"2022-04-11T17:44:05.958846Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load pretrained model\nmodel = models.googlenet(pretrained=True)\nmodel.to(device)","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:28:31.610002Z","iopub.execute_input":"2022-04-11T18:28:31.610721Z","iopub.status.idle":"2022-04-11T18:28:31.791325Z","shell.execute_reply.started":"2022-04-11T18:28:31.610682Z","shell.execute_reply":"2022-04-11T18:28:31.790662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# add custom output to the FC layer\nmodelOutputFeats = model.fc.in_features\nmodel.fc = nn.Linear(modelOutputFeats, len(ids), nn.Sigmoid())\nmodel = model.to(device)\n","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:28:38.172309Z","iopub.execute_input":"2022-04-11T18:28:38.172757Z","iopub.status.idle":"2022-04-11T18:28:38.354172Z","shell.execute_reply.started":"2022-04-11T18:28:38.172703Z","shell.execute_reply":"2022-04-11T18:28:38.353482Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# loss and optimizer\ncriterion = nn.CrossEntropyLoss()\noptimizer = optim.Adam(model.parameters(), lr=learning_rate)\n","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:28:40.220741Z","iopub.execute_input":"2022-04-11T18:28:40.221388Z","iopub.status.idle":"2022-04-11T18:28:40.227414Z","shell.execute_reply.started":"2022-04-11T18:28:40.221351Z","shell.execute_reply":"2022-04-11T18:28:40.226432Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train network\n\nfor epoch in range(10):\n    print(f\"Epoch {epoch}/{num_epochs - 1}\")\n#     train\n    losses = []\n    train_loss = []\n    val_loss = []\n    count = 0\n    train_correct = 0\n    val_correct = 0\n    for index,(inputs, labels) in enumerate(train_loader):\n#         get data to cuda\n#         in\n        count = count + list(inputs.shape)[0]\n        inputs = inputs.to(device)\n        labels = labels.to(device)\n        \n#         forward\n        \n        output = model(inputs)\n        _, preds = torch.max(output, 1)\n       \n        loss = criterion(output, labels)\n        losses.append(loss.to('cpu').detach().numpy())\n        train_loss.append(loss.to('cpu').detach().numpy())\n        \n#         clear gradient, backward, optimizer take step based on gradient params\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n        train_correct += torch.sum(preds == labels.data)\n  \n        \n#     validate\n    for index,(inputs, labels) in enumerate(val_loader):\n        #         get data to cuda\n        inputs = inputs.to(device)\n        labels = labels.to(device)\n        \n        output = model(inputs)\n        _, preds = torch.max(output, 1)\n        loss = criterion(output, labels)\n        val_loss.append(loss.to('cpu').detach().numpy())\n        val_correct += torch.sum(preds == labels.data)\n    train_accuracy = train_correct.to('cpu').numpy()/len(train_loader.dataset)\n    val_accuracy = val_correct.to('cpu').numpy()/len(val_loader.dataset)\n    print('Train loss: '+ str(sum(train_loss)/len(train_loss))+' Accuracy: '+ str(train_accuracy))\n    print('Val loss: '+ str(sum(val_loss)/len(val_loss))+' Accuracy: '+ str(val_accuracy))\n        \n        \n","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:28:42.616852Z","iopub.execute_input":"2022-04-11T18:28:42.617363Z","iopub.status.idle":"2022-04-12T03:32:34.660773Z","shell.execute_reply.started":"2022-04-11T18:28:42.617325Z","shell.execute_reply":"2022-04-12T03:32:34.65896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(train_loader.dataset)","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:09:39.013819Z","iopub.execute_input":"2022-04-11T18:09:39.014337Z","iopub.status.idle":"2022-04-11T18:09:39.019332Z","shell.execute_reply.started":"2022-04-11T18:09:39.0143Z","shell.execute_reply":"2022-04-11T18:09:39.018684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Train loss: '+ str(sum(train_loss)/len(train_loss))+' Accuracy: '+str(train_accuracy))\n# df.species.unique()\nsum(train_loss)/len(train_loss)","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:17:53.211385Z","iopub.execute_input":"2022-04-11T18:17:53.212022Z","iopub.status.idle":"2022-04-11T18:17:53.220967Z","shell.execute_reply.started":"2022-04-11T18:17:53.211982Z","shell.execute_reply":"2022-04-11T18:17:53.220041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sum(train_loss)","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:12:40.884098Z","iopub.execute_input":"2022-04-11T18:12:40.884351Z","iopub.status.idle":"2022-04-11T18:12:40.889941Z","shell.execute_reply.started":"2022-04-11T18:12:40.884323Z","shell.execute_reply":"2022-04-11T18:12:40.889298Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"int(train_correct.to('cpu').numpy())\n","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:14:26.507263Z","iopub.execute_input":"2022-04-11T18:14:26.507811Z","iopub.status.idle":"2022-04-11T18:14:26.513594Z","shell.execute_reply.started":"2022-04-11T18:14:26.507772Z","shell.execute_reply":"2022-04-11T18:14:26.512907Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loss","metadata":{"execution":{"iopub.status.busy":"2022-04-11T18:13:01.884209Z","iopub.execute_input":"2022-04-11T18:13:01.884754Z","iopub.status.idle":"2022-04-11T18:13:01.893779Z","shell.execute_reply.started":"2022-04-11T18:13:01.884716Z","shell.execute_reply":"2022-04-11T18:13:01.893098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}