{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-05-01T16:26:59.688838Z","iopub.execute_input":"2023-05-01T16:26:59.689915Z","iopub.status.idle":"2023-05-01T16:26:59.696670Z","shell.execute_reply.started":"2023-05-01T16:26:59.689865Z","shell.execute_reply":"2023-05-01T16:26:59.695408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport matplotlib.pyplot as plt\nimport torch.nn as nn\nimport torchvision\nfrom PIL import Image\nfrom torch import optim\nfrom torch.utils.data import Dataset, DataLoader\nimport pandas as pd\nfrom glob import glob\nfrom torchvision import datasets, transforms, models\nfrom torchvision.datasets import ImageFolder\nimport numpy as np\nfrom torchvision.transforms import transforms","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:27:15.997676Z","iopub.execute_input":"2023-05-01T16:27:15.998209Z","iopub.status.idle":"2023-05-01T16:27:16.008325Z","shell.execute_reply.started":"2023-05-01T16:27:15.998173Z","shell.execute_reply":"2023-05-01T16:27:16.005896Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PATH_train = \"/kaggle/input/happy-whale-and-dolphin/train_images\"","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:27:16.010950Z","iopub.execute_input":"2023-05-01T16:27:16.011673Z","iopub.status.idle":"2023-05-01T16:27:16.017582Z","shell.execute_reply.started":"2023-05-01T16:27:16.011623Z","shell.execute_reply":"2023-05-01T16:27:16.015668Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 1: Preprocessing\n## As a first step, let's apply some transformations to our training dataset... \n### We might need two different transforms since at inference time we won't be applying the same transformations","metadata":{}},{"cell_type":"markdown","source":"#### We are going to apply the followings to the train images:\n* Resizing (original resolution can be too heavy to work on), we must see whether to reshape it to a square size or not\n* Normalize them (normilization works pretty well in reducing the learning phase), at least I hope it's gonna work this time too xD\n* Optionally _augment_ the (train) set by applying a Random flip (either orizonthal or vertical)\n* As a final step, we're going to make a Tensor out of each image that way it's smooth working with torch","metadata":{}},{"cell_type":"code","source":"train_transform = transforms.Compose([\n    transforms.Resize((150, 150)),\n    #transforms.Normalize(***), # I need to see how do I compute the normalization values (RGB channels)\n    #transforms.RandomHorizontalFlip(),\n    transforms.Grayscale(num_output_channels=3), # we do this because some of the images are in greyscale, i'm trying to bring the few of them to 3channels\n    transforms.ToTensor()\n])\n\n\ntest_transform = transforms.Compose([\n    transforms.Resize((150, 150)),\n    #transforms.Normalize(***), # I need to see how do I compute the normalization values (RGB channels)\n    #transforms.RandomHorizontalFlip(),\n    transforms.Grayscale(num_output_channels=3), # we do this because some of the images are in greyscale, i'm trying to bring the few of them to 3channels\n    transforms.ToTensor()\n])","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:27:16.019577Z","iopub.execute_input":"2023-05-01T16:27:16.021034Z","iopub.status.idle":"2023-05-01T16:27:16.028920Z","shell.execute_reply.started":"2023-05-01T16:27:16.020988Z","shell.execute_reply":"2023-05-01T16:27:16.027712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### *Later on we'll define the transforms for the testset, for now we don't give a f**k*","metadata":{}},{"cell_type":"markdown","source":"# Step 2: EDA on the csv file","metadata":{}},{"cell_type":"markdown","source":"#### Loading the csv file containing all the labels for our training set","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv('/kaggle/input/happy-whale-and-dolphin/train.csv')","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:27:16.032289Z","iopub.execute_input":"2023-05-01T16:27:16.033091Z","iopub.status.idle":"2023-05-01T16:27:16.134741Z","shell.execute_reply.started":"2023-05-01T16:27:16.033053Z","shell.execute_reply":"2023-05-01T16:27:16.133705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#### Let's count the number of categories\n##### The categories are going to be the unique values on the column 'species'","metadata":{}},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:27:16.136898Z","iopub.execute_input":"2023-05-01T16:27:16.137642Z","iopub.status.idle":"2023-05-01T16:27:16.154570Z","shell.execute_reply.started":"2023-05-01T16:27:16.137603Z","shell.execute_reply":"2023-05-01T16:27:16.153683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Let's check the amount of unique values (unique individuals) inside the individual_ID column","metadata":{}},{"cell_type":"markdown","source":"#### We have around 50k images and 15k different individuals.","metadata":{}},{"cell_type":"code","source":"len(df['individual_id']), len(df['individual_id'].unique())","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:27:16.156737Z","iopub.execute_input":"2023-05-01T16:27:16.157346Z","iopub.status.idle":"2023-05-01T16:27:16.173406Z","shell.execute_reply.started":"2023-05-01T16:27:16.157309Z","shell.execute_reply":"2023-05-01T16:27:16.172067Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# getting a list containing unique values inside the 'species' column.\nspecies = df['species'].unique()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:27:16.174662Z","iopub.execute_input":"2023-05-01T16:27:16.174952Z","iopub.status.idle":"2023-05-01T16:27:16.183629Z","shell.execute_reply.started":"2023-05-01T16:27:16.174926Z","shell.execute_reply":"2023-05-01T16:27:16.182312Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(species), species","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:27:16.185094Z","iopub.execute_input":"2023-05-01T16:27:16.185810Z","iopub.status.idle":"2023-05-01T16:27:16.193471Z","shell.execute_reply.started":"2023-05-01T16:27:16.185769Z","shell.execute_reply":"2023-05-01T16:27:16.192237Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### We are dealing with 30 different species! DAMN...\n#### Too many to be left with their alphabetical name, let's categorize them ","metadata":{}},{"cell_type":"code","source":"# this command returns the same old column (species) but instead of having the name of the species, \n## we have the number associated to each species\nnew_species = df['species'].astype('category').cat.codes  \n\n\n# Then, we overwrite the old column, with this brand new numerical column\ndf['species'] = new_species\n\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:27:16.197591Z","iopub.execute_input":"2023-05-01T16:27:16.198237Z","iopub.status.idle":"2023-05-01T16:27:16.217488Z","shell.execute_reply.started":"2023-05-01T16:27:16.198198Z","shell.execute_reply":"2023-05-01T16:27:16.216726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"##### We must keep the old species names somehow to make it clear at inference time.","metadata":{}},{"cell_type":"markdown","source":"### Now that we have all the labels for each image, it's time to pair them all and instantiate a DataLoader ","metadata":{}},{"cell_type":"markdown","source":"## Defining a function that preprocesses the images and pairs them up with the associated label from the dataframe","metadata":{}},{"cell_type":"code","source":"from torchvision.io import read_image\nimport os\nclass CustomImageDataset(Dataset):\n    def __init__(self, annotations_file, img_dir, transform=None):\n        self.img_labels = annotations_file\n        self.img_dir = img_dir\n        self.transform = transform\n\n\n    def __len__(self):\n        return len(self.img_labels)\n\n    def __getitem__(self, idx):\n        img_path = os.path.join(self.img_dir, self.img_labels.iloc[idx, 0])\n        image = Image.open(img_path)\n        label = self.img_labels.iloc[idx, 1]\n        \n        if self.transform:\n            image = self.transform(image)\n        return image, label","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:28:09.872907Z","iopub.execute_input":"2023-05-01T16:28:09.873845Z","iopub.status.idle":"2023-05-01T16:28:09.881278Z","shell.execute_reply.started":"2023-05-01T16:28:09.873792Z","shell.execute_reply":"2023-05-01T16:28:09.880183Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"myDATA = CustomImageDataset(df, \"/kaggle/input/happy-whale-and-dolphin/train_images\", transform=train_transform)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:28:09.886904Z","iopub.execute_input":"2023-05-01T16:28:09.887426Z","iopub.status.idle":"2023-05-01T16:28:09.892768Z","shell.execute_reply.started":"2023-05-01T16:28:09.887382Z","shell.execute_reply":"2023-05-01T16:28:09.891752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_train, image_val = torch.utils.data.random_split(myDATA, (len(myDATA) - 1600, 1600))","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:28:09.895166Z","iopub.execute_input":"2023-05-01T16:28:09.895878Z","iopub.status.idle":"2023-05-01T16:28:09.906944Z","shell.execute_reply.started":"2023-05-01T16:28:09.895842Z","shell.execute_reply":"2023-05-01T16:28:09.905924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loader = DataLoader(image_train, batch_size=16, shuffle=True)\ntest_loader = DataLoader(image_val, batch_size=16, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:28:09.908925Z","iopub.execute_input":"2023-05-01T16:28:09.909608Z","iopub.status.idle":"2023-05-01T16:28:09.915565Z","shell.execute_reply.started":"2023-05-01T16:28:09.909571Z","shell.execute_reply":"2023-05-01T16:28:09.914547Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"next(iter(train_loader))","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:28:09.918750Z","iopub.execute_input":"2023-05-01T16:28:09.919546Z","iopub.status.idle":"2023-05-01T16:28:11.603246Z","shell.execute_reply.started":"2023-05-01T16:28:09.919507Z","shell.execute_reply":"2023-05-01T16:28:11.602233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Step 3: Create DataLoader","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:28:11.607613Z","iopub.execute_input":"2023-05-01T16:28:11.609969Z","iopub.status.idle":"2023-05-01T16:28:11.616243Z","shell.execute_reply.started":"2023-05-01T16:28:11.609929Z","shell.execute_reply":"2023-05-01T16:28:11.615170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## At this point, we have the dataloader for the training set, we'll easily do the same for test data.","metadata":{}},{"cell_type":"markdown","source":"# Step 4: *MODEL TIME!*","metadata":{}},{"cell_type":"markdown","source":"### As a first thing we are going to define a simple CNN with not too many layer, that will probably underfit the data but it's needed just to understand the dataset and the task","metadata":{}},{"cell_type":"code","source":"class CNN1(nn.Module):\n    \n    \n    def __init__(self):\n        \n        super().__init__()\n        \n        # Elements of the convolutional layers\n        self.conv1 = nn.Conv2d(3, 16, 3)\n        self.conv2 = nn.Conv2d(16, 32, 3)\n        self.conv3 = nn.Conv2d(32, 64, 3)\n\n        self.maxpool = nn.MaxPool2d((3,3))\n        \n        \n        \n        # DENSE LAYERS\n        self.fc1 = nn.Linear(7200, 128)\n        self.fc2 = nn.Linear(128, 30)\n        \n        self.softmax = nn.Softmax()\n        \n        self.relu = nn.ReLU()\n        self.dropout = nn.Dropout(0.4)\n        \n        \n    def forward(self, x):\n        \n        out = self.maxpool(self.relu(self.conv1(x)))  # first convolution + relu activation\n        \n        out = self.maxpool(self.relu(self.conv2(out)))\n        \n        #out = self.maxpool(self.relu(self.conv3(out)))\n        \n        #out = self.maxpool(self.relu(self.conv4(out)))\n        \n        fl = nn.Flatten()\n        \n        x = fl(out)\n        \n        out = self.relu(self.fc1(x))\n        \n        out = self.dropout(out)\n        \n        out = self.fc2(out)\n        \n        #print('fino a qui tutto bene')\n        \n        return out\n        \n        ","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:28:11.621425Z","iopub.execute_input":"2023-05-01T16:28:11.623910Z","iopub.status.idle":"2023-05-01T16:28:11.636596Z","shell.execute_reply.started":"2023-05-01T16:28:11.623874Z","shell.execute_reply":"2023-05-01T16:28:11.635543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = CNN1()","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:28:11.641490Z","iopub.execute_input":"2023-05-01T16:28:11.643735Z","iopub.status.idle":"2023-05-01T16:28:11.664839Z","shell.execute_reply.started":"2023-05-01T16:28:11.643677Z","shell.execute_reply":"2023-05-01T16:28:11.663753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs, labels = next(iter(train_loader))","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:28:11.669080Z","iopub.execute_input":"2023-05-01T16:28:11.671314Z","iopub.status.idle":"2023-05-01T16:28:13.796016Z","shell.execute_reply.started":"2023-05-01T16:28:11.671278Z","shell.execute_reply":"2023-05-01T16:28:13.794960Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"logits = model(imgs)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:28:13.799250Z","iopub.execute_input":"2023-05-01T16:28:13.799640Z","iopub.status.idle":"2023-05-01T16:28:13.969455Z","shell.execute_reply.started":"2023-05-01T16:28:13.799603Z","shell.execute_reply":"2023-05-01T16:28:13.968445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"loss = nn.CrossEntropyLoss()\n\ncriterion = optim.AdamW(model.parameters(), lr=0.001)","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:28:13.971062Z","iopub.execute_input":"2023-05-01T16:28:13.971444Z","iopub.status.idle":"2023-05-01T16:28:13.979367Z","shell.execute_reply.started":"2023-05-01T16:28:13.971405Z","shell.execute_reply":"2023-05-01T16:28:13.978207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#for param_group in criterion.param_groups:\n#        param_group['lr'] = 0.001","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:56:36.378615Z","iopub.execute_input":"2023-05-01T16:56:36.379336Z","iopub.status.idle":"2023-05-01T16:56:36.384303Z","shell.execute_reply.started":"2023-05-01T16:56:36.379299Z","shell.execute_reply":"2023-05-01T16:56:36.382971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(model, iteratore, loss_f, optimizer):\n    k = 0\n    num_batches = 400\n    tot_loss = 0\n    \n    model.train()\n    \n    for i, batch in enumerate(iteratore):\n        \n        \n        \n        k += 1\n        if k > 400:\n            break\n        \n        \n        img, label = batch[0], batch[1]\n        \n        label = label.type(torch.LongTensor)\n        img, label = img.cuda(), label.cuda()\n        \n        \n        \n        out = model(img)\n        \n        loss = loss_f(out, label)\n        \n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n        \n        tot_loss += loss.item()\n        \n        if i % 100 == 0:\n            print(f\"{i}-th  batch. ATM the loss is {loss}\\n\")\n        \n        \n    print(f\"Average TRAIN loss: {tot_loss/num_batches}\")","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:28:13.980873Z","iopub.execute_input":"2023-05-01T16:28:13.981407Z","iopub.status.idle":"2023-05-01T16:28:13.992189Z","shell.execute_reply.started":"2023-05-01T16:28:13.981368Z","shell.execute_reply":"2023-05-01T16:28:13.991167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def test(model, iteratore, loss_f):\n    \n    \n    num_batches = len(iteratore)\n    test_loss = 0\n    \n    model.eval()\n    with torch.no_grad():\n        for img, labels in iteratore:\n            labels  = labels.type(torch.LongTensor)\n            img, labels = img.cuda(), labels.cuda()\n\n            pred = model(img)\n            test_loss += loss_f(pred, labels).item()\n\n    test_loss = test_loss / num_batches\n\n    print(f\"Average TEST loss: {test_loss}\")\n    \n    return test_loss","metadata":{"execution":{"iopub.status.busy":"2023-05-01T18:30:14.149430Z","iopub.execute_input":"2023-05-01T18:30:14.149736Z","iopub.status.idle":"2023-05-01T18:30:14.183773Z","shell.execute_reply.started":"2023-05-01T18:30:14.149706Z","shell.execute_reply":"2023-05-01T18:30:14.182862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.cuda()\n\nmin_loss = 100\n\nfor epoch in range(10):\n    print(f\"----- epoch {epoch} -----\")\n    train(model, train_loader, loss, criterion)\n    t_loss = test(model, test_loader, loss)\n    \n    if t_loss < min_loss:\n        min_loss = t_loss\n        torch.save(model, 'checkpoint.pth')\n    print(\"\\n\")\n    \n    ","metadata":{"execution":{"iopub.status.busy":"2023-05-01T16:56:45.268188Z","iopub.execute_input":"2023-05-01T16:56:45.268554Z","iopub.status.idle":"2023-05-01T17:10:35.996027Z","shell.execute_reply.started":"2023-05-01T16:56:45.268521Z","shell.execute_reply":"2023-05-01T17:10:35.994868Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Step 5: *Evaluation Metrics*","metadata":{}},{"cell_type":"markdown","source":"## The metric of evaluation suggested by the author of the competition is the Mean Average Precision since the number of predicted labels for each test image is up to 5.\n#### ex: label to predict is A, our prediction will be an array of 5 elements [A, A, A, A, A] = 1.0 (precision)\n\n\nhttps://www.kaggle.com/competitions/happy-whale-and-dolphin/overview/evaluation\n\n\n\n\n\n\n### Now I'm wondering what do we do with the species column if all we need to predict is the id of the individual...\n### As a first guess I'd say that we could first guess the species to prune out quite a lot of individuals... @elena ??","metadata":{}},{"cell_type":"markdown","source":"","metadata":{}}]}