{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"#This architcture is a unique one. I name it Arpagus.\n#This project will mainly focus on object localization in images.\n#Created by: Xanta, 05-01-2021\n#14:50","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:47:59.729186Z","iopub.execute_input":"2021-07-31T04:47:59.729542Z","iopub.status.idle":"2021-07-31T04:47:59.734722Z","shell.execute_reply.started":"2021-07-31T04:47:59.729497Z","shell.execute_reply":"2021-07-31T04:47:59.733853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\"\"\"\nToday's objective:\n-Introduce a mask into the error function that only optimizes the bounding box on the gound truth\nexisting grid cell with the highest iou.\n\n\n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:48:00.060035Z","iopub.execute_input":"2021-07-31T04:48:00.060392Z","iopub.status.idle":"2021-07-31T04:48:00.067638Z","shell.execute_reply.started":"2021-07-31T04:48:00.060358Z","shell.execute_reply":"2021-07-31T04:48:00.066747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#each index here has more than 400 examples. totally 8_879 datasize.\n#however as 8_879 is a bad factor. I've taken 8_878 as the no of examples.\n#with about 46 images per batch totalling 193 mini batches.","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:48:00.207246Z","iopub.execute_input":"2021-07-31T04:48:00.207615Z","iopub.status.idle":"2021-07-31T04:48:00.2113Z","shell.execute_reply.started":"2021-07-31T04:48:00.207573Z","shell.execute_reply":"2021-07-31T04:48:00.210291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#32 images, 16 x 16 boxes.","metadata":{"execution":{"iopub.status.busy":"2021-07-31T04:48:00.440098Z","iopub.execute_input":"2021-07-31T04:48:00.440433Z","iopub.status.idle":"2021-07-31T04:48:00.444349Z","shell.execute_reply.started":"2021-07-31T04:48:00.440401Z","shell.execute_reply":"2021-07-31T04:48:00.44334Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#import pytorch, matplotlib\n\"\"\"\nLoad the required libraries.\n\"\"\"\nimport torch as tor\nfrom matplotlib import image\nfrom matplotlib import pyplot as plt\nfrom os import listdir\nfrom PIL import Image\n#from scipy.io import loadmat as mat\nimport xml.etree.ElementTree as ET\nfrom torchvision import transforms\nfrom matplotlib import patches as pat\nfrom numpy import insert, array, zeros, concatenate\nflatten = tor.nn.Flatten\nconv2d = tor.nn.Conv2d\nlinear = tor.nn.Linear\nsequential = tor.nn.Sequential\nbatch_norm_2d = tor.nn.BatchNorm2d\nbatch_norm_1d = tor.nn.BatchNorm1d\n#softplus = tor.nn.Softplus()\nsigmoid = tor.nn.Sigmoid()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:16:09.863016Z","iopub.execute_input":"2021-08-03T13:16:09.863368Z","iopub.status.idle":"2021-08-03T13:16:11.364284Z","shell.execute_reply.started":"2021-08-03T13:16:09.863335Z","shell.execute_reply":"2021-08-03T13:16:11.363472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Things to do:\n\n-Learn how random crop works and use it to randomly crop the images but letting the bounding boxes \nstay the same. \n\n-Change the training code to be more pytorch friendly by removing the for loops and making the images\nall to become same sized.","metadata":{}},{"cell_type":"markdown","source":"New Idea. 2 Step training. First let it train on distance loss for about 10-20 epochs. Then train it on IoU loss on about the next 10-20 epochs","metadata":{}},{"cell_type":"markdown","source":">Model Architecture : Arphagus;\n\n>Model Version   : 0;\n\n>Model Name  : Jyam;\n\n>Parameters  : 83_15_614;\n\n1. **Padd layers**: There will be about 3x predicted bounding boxes for max bounding boxes there exits in\nthe dataset. *Completed*\n2. **Mini-batched**: Each mini batch will be randomly assingned a size out of 3 sizes. i. e 480p (854x480), 720p (1280x720), 1080p (1920x1080). All the resolutions have approximately consistent \naspect ratios of 16:9, The aspect ratio will be randomly inverted throught training. i.e from (825x480) to (480x854), 16:9 to 9:16. *Completed*","metadata":{}},{"cell_type":"markdown","source":"x_min, y_min, x_max, y_max ---> x_mid, y_mid, w, h\n","metadata":{}},{"cell_type":"code","source":"padd = tor.nn.ZeroPad2d\npath = '../input/the-oxfordiiit-pet-dataset/images/images'\nif tor.cuda.is_available() == True:\n    device = 'cuda'\nelse:\n    device = 'cpu'","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:16:11.365954Z","iopub.execute_input":"2021-08-03T13:16:11.366382Z","iopub.status.idle":"2021-08-03T13:16:11.428826Z","shell.execute_reply.started":"2021-08-03T13:16:11.366343Z","shell.execute_reply":"2021-08-03T13:16:11.427802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" #class RCrop:\n\"\"\"\nTakes in a image of any size or aspect ratio and crops a region of the image with \nspecified size randomly. Also, checks if the bounding box falls in the region of cropped\nimage and return either a zero confident tensor (if there is no bounding box there) or a one \nconfident tensor (if there is a bounding box there).\n\"\"\"\n    \n#    def __init__(self, crop_size=256):\n#        self.crop_size = crop_size\n    \n#    def crop(self, img, lbl)\n#        x_range, y_range = img.size()[-2:]\n#        x_coor, y_coor = tor.randint(x_range), tor.randint(y_range)\n        \n#        transforms.functional.crop()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:16:11.430337Z","iopub.execute_input":"2021-08-03T13:16:11.430944Z","iopub.status.idle":"2021-08-03T13:16:11.441372Z","shell.execute_reply.started":"2021-08-03T13:16:11.430902Z","shell.execute_reply":"2021-08-03T13:16:11.440486Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Flip:\n    \"\"\"\n    Take in a image and returns a flipped image. Either horizontal, vertical or both vertical\n    and horizontal flip. Based on its index.\n    \"\"\"\n    def __init__(self, data_len = 3686-32):\n        self.data_len = data_len\n        \n    \"\"\"\n    Rescales the index variable (idx) so that it doesn't raise a (list index out of range) while \n    loading the image into working memory and saves the transformations that are to be applied on \n    the image.\n    \"\"\"\n    def check(self, input_):\n        idx = (input_).clone()\n        \n        self.trnsfm = [transforms.functional.hflip, transforms.functional.vflip]\n        \n        if idx < self.data_len: #normal\n            self.bool = -1\n            \n        elif (idx >= self.data_len) and (idx < (self.data_len*2)): #horizontal flip.\n            idx -= self.data_len\n            self.bool = 0\n        \n        elif (idx >= (self.data_len*2)) and (idx < (self.data_len*3)): #vertical flip.\n            idx -= (self.data_len*2)\n            self.bool = 1\n            \n        elif idx >= (self.data_len*3) and idx < (self.data_len*4): #both horizontal and vertical.\n            idx -= (self.data_len*3)\n            self.bool = 2\n            \n        else:\n            idx = None\n            self.bool = None\n            \n        return idx\n    \"\"\"\n    Applies the saved transformations on the image. \n    \"\"\"\n    def flip(self, image, label):\n        if self.bool < 2 and self.bool != -1:\n            image = self.trnsfm[self.bool](image)\n            label[self.bool] = 1-(label[self.bool]+label[self.bool+2])\n        \n        elif self.bool == 2:\n            image = self.trnsfm[1](self.trnsfm[0](image))\n            label[0], label[1] = 1-(label[0]+label[2]), 1-(label[1]+label[3])\n            \n        return image, label","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:16:11.443117Z","iopub.execute_input":"2021-08-03T13:16:11.443584Z","iopub.status.idle":"2021-08-03T13:16:11.461468Z","shell.execute_reply.started":"2021-08-03T13:16:11.443543Z","shell.execute_reply":"2021-08-03T13:16:11.460726Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Loader_2:\n    def __init__(self, file_path, mini_bch_size, grid_size, device):\n        self.trn_files = [[], []] #training data.\n        #In first list the images will be saved.\n        #in the second one the labels will be saved.\n\n        self.noise_percent = 0.20\n        self.val_files = [[], []] #testing data, no labels will exist.\n        #same as the train but will be used for evaluation band testing.\n        self.device=device\n        self.test_files = [[], []]\n        \n        \"\"\"\n        Unpacking the dataset into train, validation and test dataset.\n        \"\"\"\n        for file in listdir(file_path):\n            u = file.rsplit('.jpg', 1)[0]\n            u += '.xml'\n            try:\n                #creating tree\n                tree = ET.parse('../input/the-oxfordiiit-pet-dataset/annotations/annotations/xmls/' + u)\n\n                # getting the parent tag of\n                # the xml document\n                root = tree.getroot()\n\n                label = []\n                for t in root[5][4]:\n                    label.append(int(t.text)) #xmin, ymin, xmax, ymax\n\n                u = tor.tensor([label[0], label[1], label[2]-label[0], label[3]-label[1]], dtype=tor.float32)\n                #xmin, ymin, xmax --> width, ymax --> height\n                \n                if len(self.trn_files[0]) != (3686-32):\n                    self.trn_files[0].append(file)\n                    self.trn_files[1].append(u)\n                else:\n                    self.val_files[0].append(file)\n                    self.val_files[1].append(u)\n                \n                \n            except:\n                self.test_files[0].append(file)\n            \n        self.mini_bch_size=mini_bch_size\n        self.path = file_path + '/'\n        \n        \n        \"\"\"\n        Initialize various other variables like grid size, no of duplicate boxes, and also the default\n        image and label (Zero) tensor\n        \"\"\"\n        self.grid_size = grid_size\n        \n        self.dup_boxs = 2\n        self.default_img = tor.zeros(self.mini_bch_size, 3, 256, 256, dtype=tor.float32, device=device)\n        self.default_lbl = tor.zeros(self.mini_bch_size, (int(256/self.grid_size)), (int(256/self.grid_size)), self.dup_boxs, 5, dtype=tor.float32, device=device)\n        \n        #create the coordinates of grid.\n        self.grid_coor = []\n        for i in range(int(256/self.grid_size)):\n            self.grid_coor.append([self.grid_size*(i+1)])\n            \n        u = tor.tensor(self.grid_coor.copy()*len(self.grid_coor)).view(len(self.grid_coor), len(self.grid_coor), 1)\n        \n        \n        \"\"\"\n        Randomizes the indexes inorder to increase accuracy of the dataset. Initializes the\n        virtual yolo grid, and the transforms that are to be applied.\n        \"\"\"\n        self.flipdexs = 2\n        self.idxs = tor.randperm(len(self.trn_files[1])*self.flipdexs)\n        self.grid_coor = tor.tensor(self.grid_coor, device = self.device)\n        self.grid_mat_coor = tor.cat((u, tor.transpose(u, 0, 1)), -1).to(self.device)\n        \n        self.sizes = tor.tensor(range(0, 5), dtype=tor.int64, device=self.device)\n        self.trnsfm = [transforms.ToTensor(), transforms.Resize((256, 256)), Flip(len(self.trn_files[1]))]\n        \n    def load_bch(self, epoch, trn_or_val):\n        #midpoint = (256/2)+64, (256/2)-64\n        \"\"\"\n        Check if the function is working in train, validation or test mode.\n        \"\"\"\n        if trn_or_val == 0:\n            q = self.trn_files\n            mini_bch_idxs = self.idxs[(epoch)*self.mini_bch_size:(epoch+1)*self.mini_bch_size]\n            noise_percent = self.noise_percent\n            \n        elif trn_or_val == 0.5:\n            q = self.val_files\n            mini_bch_idxs = tor.arange(len(q[0]))[:self.mini_bch_size]\n            noise_percent = 0\n            #the validation dataset is consistant for every epoch.\n    \n        else:\n            q = self.test_files\n            mini_bch_idxs = tor.arange(len(q[0]))[:self.mini_bch_size]\n        \n        \n        \n        \"\"\"\n        Randex means random index taken out of the a small part of self.idxs as mini_bch_idx.\n        This loop essentially loads the image and label from the randomly selected index to make\n        a mini batch.\n        \"\"\"\n        img_bch, lbl_bch = self.default_img.clone(), self.default_lbl.clone()\n        \n        u = 0\n        for randex in mini_bch_idxs:\n            if tor.randint(10, (1,)) < 2:\n                img_noise = (0)\n            else:\n                img_noise = (noise_percent)\n\n            i = ((self.trnsfm[2]).check(randex)).clone()\n            \n            #read the image.\n            img_ = (image.imread(self.path+(q[0][i]))).copy()   \n            \n            img_.flags.writeable = True\n            #this is just to not trigger the warning message from pytorch.\n            \n            h_or_w = 1*(max(img_.shape[0], img_.shape[1]) == img_.shape[1])\n            #0 when height > width.\n            #1 when width > height.\n\n            old_w, old_h = img_.shape[0], img_.shape[1]\n            #real img's width and height.\n            \n            missing_w, missing_h = 0, 0\n\n            #After analyzing the behaviour of the code I've come to the conclusion that\n            #the shape of a numpy array is (height, width, channels).\n            \n            if h_or_w == 0:\n                #when height is more than width.\n                \n                missing_w = round((img_.shape[0] - img_.shape[1])/2)\n                #pixels that are to be added. This needs to be done in order to \n                #not get a distortion while resizing the image. At left and right.\n                \n                padd_ = padd((missing_w, missing_w, 0, 0))\n                \n            else:\n                #when its the other way around.\n                missing_h = round((img_.shape[1] - img_.shape[0])/2)\n                padd_ = padd((0, 0, missing_h, missing_h))\n\n            #load the required label's clone.\n            label = (q[1][i]).clone().float()\n            \n            #print(label)\n            #(x + missing width ) * (1/ (old_w + 2missing height))\n            #convert the label to match the now, square padded image.\n            label[0], label[1] = (label[0]+missing_w).clone()*(1/(old_w+(2*missing_h))), (label[1]+missing_h).clone()*(1/(old_h+(2*missing_w)))\n            label[2], label[3] = label[2]*(1/(old_w+(2*missing_h))), label[3]*(1/(old_h+(2*missing_w)))\n            \n            #transform the image into a tensor, resize the image and finally\n            #flip the image according to its index value.\n            img_bch[u], label = (self.trnsfm[2]).flip(self.trnsfm[1](padd_(self.trnsfm[0](img_))), label)\n            img_bch[u] = ((img_bch[u]*(1-img_noise))+(tor.randn(img_bch[u].size(), device=self.device)*img_noise))\n\n            #change the bbox's x, y coordinates to be representive of the midpoint of bbox \n            #than the edge.\n            label[0], label[1] = label[0] + (label[2]*0.5), label[1]+(label[3]*0.5)\n            \n            #yolofy the labels. I. e normalize the labels according to its respective grid cell.\n            yolo_label = self.yolofy(tor.cat((tor.tensor([1,]).to(self.device), (label).to(self.device)*256)), u)\n            \n            lbl_bch[u][int(yolo_label[0])-1][int(yolo_label[1])-1][0] = (yolo_label[2])\n            lbl_bch[u][int(yolo_label[0])-1][int(yolo_label[1])-1][1][0] = 1.0\n            \n            u += 1\n                \n        if (epoch+1) == round(len(self.idxs)/self.mini_bch_size):\n            self.idxs = tor.randperm(len(self.trn_files[1])*self.flipdexs)\n    \n        return img_bch.to(self.device), lbl_bch.to(self.device)\n\n    def re_labels(self, tensor):\n        \"\"\"\n        This function makes the yolo friendly labels into matplotlib friendly labels.\n        \"\"\"\n        item = tensor.clone()\n        #item size is (16, 16, 5, 3)\n        #16x16 boxes, with 5 variables with 3 duplicate boxes.\n        \n        #subtract grid and multiply by grid then multiply by boole.\n        grid = (tor.index_select(self.grid_mat_coor.clone(), -1, self.sizes[0]).unsqueeze(-1), tor.index_select(self.grid_mat_coor.clone(), -1, self.sizes[1]).unsqueeze(-1))\n        \n        #grid (16, 3), bool_coor (16, 1)\n        c, x, y, w, h = item[...,0:1], item[...,1:2], item[...,2:3], item[...,3:4], item[...,4:5]\n        w, h = tor.mul(w, 256), tor.mul(h, 256)\n        \n        #convert the middle co-ordinates to edge coordinates for the bounding box.\n        x, y = (tor.mul((1 - x), grid[0]) - (w/2), tor.mul((1 - y), grid[1]) - (h/2))\n        \n        return tor.cat((c, x, y, w, h), -1)\n        \n    def yolofy(self, temp, u):\n        \"\"\"\n        This function makes the matplotlib friendly labels into yolo friendly labels.\n        \"\"\"\n        #boolean for x and y. i. e the range in which they exist. if x is 95 then range is 96 if x is 105 then\n        #range is 128. etc.\n        bool_x, bool_y = (tor.stack([(1*(temp[1] < self.grid_coor)) - (1*((temp[1]+self.grid_size) < self.grid_coor)) - (1*(temp[1]==0)), (1*(temp[2] < self.grid_coor)) - (1*((temp[2]+self.grid_size) < self.grid_coor)) - (1*(temp[2]==0))]))\n        \n        #taking the index. if range is 32 then index is 1-(1) or if range is 128 then index is 3-(1)\n        grid_x, grid_y = (tor.max(self.grid_coor*bool_x, 0)[0])/self.grid_size, (tor.max(self.grid_coor*bool_y, 0)[0]/self.grid_size)\n        \n        #normalized_x_y coor\n        temp[1], temp[2] = tor.max(((self.grid_coor - temp[1])/self.grid_coor)*(bool_x), 0)[0], tor.max(((self.grid_coor - temp[2])/self.grid_coor)*(bool_y), 0)[0]\n        temp[3], temp[4] = temp[3]/256, temp[4]/256\n\n        return grid_y.long(), grid_x.long(), temp","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:45:02.832251Z","iopub.execute_input":"2021-08-03T13:45:02.832629Z","iopub.status.idle":"2021-08-03T13:45:02.936646Z","shell.execute_reply.started":"2021-08-03T13:45:02.832593Z","shell.execute_reply":"2021-08-03T13:45:02.935633Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Abs(tor.nn.Module):\n    \"\"\"\n    Absolute value activation function.\n    \"\"\"\n    def __init__(self):\n        super().__init__()\n    \n    def forward(self, input):\n        return tor.abs(input)\n#activation function.\n\ndef mish(x): return (x * tor.tanh(softplus(x)))\ndef swish(x): return (x * sigmoid(x))\n\nclass Mish(tor.nn.Module):\n    \"\"\"\n    Mish activation function\n    \"\"\"\n    def __init__(self):\n        super().__init__()\n        \n    def forward(self, input): \n        return mish(input)\n\nclass Swish(tor.nn.Module):\n    \"\"\"\n    Swish activation function\n    \"\"\" \n    def __init__(self):\n        super().__init__()\n    \n    def forward(self, input): \n        return swish(input)\n\ndef iou_1(pred_coor, lbl_coor, pred_wh, lbl_wh, device='cuda'):\n    lbl_coor = lbl_coor+ 1*(lbl_coor)\n    lbl_wh = lbl_wh+ 1*(lbl_wh)\n\n    X  = ( (2*pred_coor[0]) + (pred_wh[0]/2) ) * ( (2*pred_coor[1]) + (pred_wh[1]/2) )\n    X_ = ( (2*lbl_coor[0])  + (lbl_wh[0]/2)  ) * ( (2*lbl_coor[1])  + (lbl_wh[1]/2)  )\n    I_h = tor.min(pred_coor[1]+pred_wh[1], lbl_coor[1]+lbl_wh[1]) + tor.min(pred_coor[1], lbl_coor[1])\n    I_w = tor.min(pred_coor[0]+pred_wh[0], lbl_coor[0]+lbl_wh[0]) + tor.min(pred_coor[0], lbl_coor[0])\n\n    I = I_h * I_w\n    U = X + X_ - I\n    IoU = (I/U)\n\n    return IoU\n\ndef iou_2(a_coor, b_coor, a_wh, b_wh, device='cuda'):\n    #(for a or b minimun of all x coor vector + all width coordinate vector) - (for a or b maximum pf all x coor vector)\n\n    #This IoU function was customly created and implemented by me. Unlike in earlier versions\n    #where I used a IoU function implemented by someone online, I did this because I wanted to\n    #A. Optimize the error function for calculating error in batches instead of a single pair.\n    #B. There were a few fatal flaws with the earlier implementation.\n    inst_x = tor.min((a_coor[0]+(a_wh[0]/2)), (b_coor[0]+(b_wh[0]/2))) - tor.max((a_coor[0]-(a_wh[0]/2)), (b_coor[0]-(b_wh[0]/2)))\n    inst_y = tor.min((a_coor[1]+(a_wh[1]/2)), (b_coor[1]+(b_wh[1]/2))) - tor.max((a_coor[1]-(a_wh[1]/2)), (b_coor[1]-(b_wh[1]/2)))\n\n    inst = tor.max(tor.zeros(inst_x.size(), device=device), inst_x) * tor.max(tor.zeros(inst_y.size(), device=device), inst_y)\n    uni = ((a_wh[1]*a_wh[0]) + (b_wh[1]*b_wh[0]) - inst)\n\n    return inst/(uni + 1*(inst==0))","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:16:11.729901Z","iopub.execute_input":"2021-08-03T13:16:11.730187Z","iopub.status.idle":"2021-08-03T13:16:11.756436Z","shell.execute_reply.started":"2021-08-03T13:16:11.730158Z","shell.execute_reply":"2021-08-03T13:16:11.755328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You pathetic skull cracked dumbo. You have confused the confidence metric with the class confidence metric you idiot. Sike. They are both DIFFERENT!!! i. e confidence cannot and should not be negative!!!","metadata":{}},{"cell_type":"code","source":"class YoloLoss(tor.nn.Module):\n    def __init__(self, device):\n        super().__init__()\n        self.sizes = tor.tensor(range(0, 5), dtype=tor.int64, device=device)\n        \n    def forward(self, preds, lbls, train=1):\n        conf_pred = preds[...,0:1]\n        conf_lbl = lbls[...,0:1]\n        \n        coor_pred = tor.stack((preds[...,1:2], preds[...,2:3])) \n        coor_lbl = tor.stack((lbls[...,1:2], lbls[...,2:3]))\n        \n        wh_pred = tor.stack(((preds[...,3:4]), preds[...,4:5]))\n        wh_lbl = tor.stack((lbls[...,3:4], lbls[...,4:5]))\n        \n        sqr_er = tor.nn.MSELoss(reduction='none')\n        \n        iou__ = iou_1(coor_pred, coor_lbl, wh_pred, wh_lbl)\n        max_mask = (((tor.max(iou__, -2, True)[0])==iou__)*conf_lbl).float()\n        \n        if train == 1:\n            coor_pred.register_hook(lambda grad: (grad * max_mask.float()))\n            wh_pred.register_hook(lambda grad: (grad * max_mask.float()))\n        \n        L_1, L_2 = 5, 0.5\n        \n        coor_er = L_1*(sqr_er(coor_pred[0], coor_lbl[0]) + sqr_er(coor_pred[1], coor_lbl[1])).sum()\n\n        #print((conf_lbl*tor.max(0, iou__))[0].sum(), (conf_lbl*tor.max(0, iou__)).sum())\n        wh_er = L_1*(sqr_er(tor.sign(wh_pred[0])*tor.sqrt(tor.abs(wh_pred[0]) + 1e-10), tor.sqrt(wh_lbl[0])) \n                   + sqr_er(tor.sign(wh_pred[1])*tor.sqrt(tor.abs(wh_pred[1]) + 1e-10), tor.sqrt(wh_lbl[1]))).sum()\n        \n        conf_er = (sqr_er(conf_pred*max_mask, conf_lbl*iou__) + L_2*(sqr_er(conf_pred*(1 - max_mask), conf_lbl*iou__))).sum()\n        \n        return (coor_er+wh_er+conf_er)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T14:04:17.987695Z","iopub.execute_input":"2021-08-03T14:04:17.988177Z","iopub.status.idle":"2021-08-03T14:04:18.017744Z","shell.execute_reply.started":"2021-08-03T14:04:17.988129Z","shell.execute_reply":"2021-08-03T14:04:18.016564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#the model\nclass Arpagus(tor.nn.Module): \n    def __init__(self, device):\n        super(Arpagus, self).__init__()\n    \n    \n        self.device = device\n        #these are the 3 convolutional synapses; Same convolution;\n        self.conv_layer = sequential(\n                                    conv2d(3, 4, (3), padding=5, stride=2),\n                                    batch_norm_2d(4),\n                                    Swish(),\n            \n                                    conv2d(4, 8, (3), padding=5, stride=2),\n                                    batch_norm_2d(8),\n                                    Swish(),\n            \n                                    conv2d(8, 4, (1)),\n                                    batch_norm_2d(4),\n                                    Swish(),\n            \n                                    conv2d(4, 8, (3), padding=5, stride=2),\n                                    batch_norm_2d(8),\n                                    Swish(),\n                            \n                                    conv2d(8, 16, (3), padding=5, stride=2),\n                                    batch_norm_2d(16),\n                                    Swish(),\n            \n                                    conv2d(16, 8, (1)),\n                                    batch_norm_2d(8),\n                                    Swish(),\n            \n                                    conv2d(8, 16, (3), padding=5, stride=2),\n                                    batch_norm_2d(16),\n                                    Swish(),\n                            \n                                    conv2d(16, 32, (3), padding=5, stride=2),\n                                    batch_norm_2d(32),\n                                    Swish(),\n            \n                                    conv2d(32, 16, (1)),\n                                    batch_norm_2d(16),\n                                    Swish(),\n            \n                                    conv2d(16, 32, (3), padding=5, stride=2),\n                                    batch_norm_2d(32),\n                                    Swish(),\n                            \n                                    conv2d(32, 64, (3), padding=5, stride=2),\n                                    batch_norm_2d(64),\n                                    Swish(),\n            \n                                    conv2d(64, 32, (1)),\n                                    batch_norm_2d(32),\n                                    Swish(),\n            \n                                    conv2d(32, 64, (3), padding=5, stride=2),\n                                    batch_norm_2d(64),\n                                    Swish(),\n                            \n                                    conv2d(64, 128, (3), padding=5, stride=2),\n                                    batch_norm_2d(128),\n                                    Swish(),\n            \n                                    conv2d(128, 64, (1)),\n                                    batch_norm_2d(64),\n                                    Swish(),\n            \n                                    conv2d(64, 128, (3), padding=5, stride=2),\n                                    batch_norm_2d(128),\n                                    Swish(),\n                            \n                                    conv2d(128, 256, (3), padding=5, stride=2),\n                                    batch_norm_2d(256),\n                                    Swish(),\n            \n                                    conv2d(256, 128, (1)),\n                                    batch_norm_2d(128),\n                                    Swish(),\n            \n                                    conv2d(128, 256, (3), padding=5, stride=2),\n                                    batch_norm_2d(256),\n                                    Swish(),\n            \n                                    conv2d(256, 512, (3), padding=5, stride=2),\n                                    batch_norm_2d(512),\n                                    Swish(),\n            \n                                    conv2d(512, 256, (1)),\n                                    batch_norm_2d(256),\n                                    Swish(),\n                                    \n                                    flatten(1, -1),\n            \n                                    linear(256*9*9, 8*8*2*5),\n                )\n        \n        #the loss\n        self.Loss_1 = YoloLoss(device)\n        \n        #the optimizer\n        self.Optimizer = tor.optim.AdamW(self.parameters(), lr=(2e-4))#, momentum=0.9)#tor.optim.SGD(self.parameters(), lr=1e-5, momentum=0.9, weight_decay=1e-4)#\n        \n    #takes training data and gives bunch of predictions.\n    def forward(self, data, bch_size=16):\n        output = (self.conv_layer(data).view(bch_size, 8, 8, 2, 5))\n        return output\n        \n    def backprop(self, preds, lbls, val_or_trn):\n        \n        if val_or_trn == 1:\n            error_1 = self.Loss_1(preds, lbls, 1)\n            \n            error_1.backward()\n            self.Optimizer.step()\n            \n            #zeroing the gradients.\n            self.Optimizer.zero_grad()\n\n        else:\n            error_1 = self.Loss_1(preds, lbls, 0)\n            \n            \n        return error_1","metadata":{"execution":{"iopub.status.busy":"2021-08-03T14:04:19.499783Z","iopub.execute_input":"2021-08-03T14:04:19.500121Z","iopub.status.idle":"2021-08-03T14:04:19.531191Z","shell.execute_reply.started":"2021-08-03T14:04:19.500072Z","shell.execute_reply":"2021-08-03T14:04:19.529878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Jiyam = Arpagus(device=device)\n#trained_model = tor.load('../input/project-seeq/Arpagus Jiyam.mdl', map_location=device) #  #tor.load('../input/project-t-look/Arpagus Etirun', map_location=device)\n#Jiyam.load_state_dict(trained_model['state_dict']) #Arpagus() #Model name: Arphagus Etirun\n\nprint('Model created...')\nJiyam.to(device) #changing the model save device to gpu from cpu.\n\n#Jiyam.Optimizer.load_state_dict(trained_model['optimizer'])\n#Jiyam.Scheduler.load_state_dict(trained_model['scheduler'])\n# #loading the state of optimizer from saved version.","metadata":{"execution":{"iopub.status.busy":"2021-08-03T14:04:35.043247Z","iopub.execute_input":"2021-08-03T14:04:35.043612Z","iopub.status.idle":"2021-08-03T14:04:35.264478Z","shell.execute_reply.started":"2021-08-03T14:04:35.043579Z","shell.execute_reply":"2021-08-03T14:04:35.263434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Jiyam.Optimizer.param_groups[0]['lr'] = 1e-5\n#Jiyam.Optimizer.param_groups[0]['lr']","metadata":{"execution":{"iopub.status.busy":"2021-08-03T14:04:35.269718Z","iopub.execute_input":"2021-08-03T14:04:35.269971Z","iopub.status.idle":"2021-08-03T14:04:35.27376Z","shell.execute_reply.started":"2021-08-03T14:04:35.269945Z","shell.execute_reply":"2021-08-03T14:04:35.272849Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bch_size = 16\nData_Loader = Loader_2(path, bch_size, 256/(8), device)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:53:08.895531Z","iopub.execute_input":"2021-08-03T13:53:08.895854Z","iopub.status.idle":"2021-08-03T13:53:14.750543Z","shell.execute_reply.started":"2021-08-03T13:53:08.895823Z","shell.execute_reply":"2021-08-03T13:53:14.749694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Starting training...')\nepochs = 40\ntor.autograd.set_detect_anomaly(True)\n\ntot_error= [[], []] #0 == train error, 1 == validation error\n\nbst_minima = 10**5\n#10**5 #Remember to change this every time you load a new model.\n#very IMPORTANT!!!\n\nJiyam.train(True)\n\nchances = 100\nq = round(len(Data_Loader.idxs)/bch_size)\n\n#iteration phase. about n number of iterations.\nfor epoch in range(1, 1+epochs):\n    error_trn = 0\n    \n    for mini_bch_idx in range(q):\n        mini_data_bch, mini_label_bch = Data_Loader.load_bch(mini_bch_idx, 0)\n        \n        #this is backpropagating per batch.\n        #k = (Jiyam.backprop((Jiyam.forward(mini_data_bch, bch_size)), mini_label_bch, 1)).item()\n        #print(mini_label_bch[0].view(-1, 5).sum(-2))\n        error_trn += (Jiyam.backprop((Jiyam.forward(mini_data_bch, bch_size)), mini_label_bch, 1)).detach().item()\n        #add the batch error into total batch error.\n        \n        #print(str(mini_bch_idx+1)+'.', k)\n        #print(mini_bch_idx, k)\n        \n        del mini_data_bch, mini_label_bch\n        tor.cuda.empty_cache()\n    \n    print()\n    print(f'{epoch}: Train error:{error_trn/q}')\n    tot_error[0].append(error_trn/q)\n    \n    #The validation error is consistent for about 2 decimal places. After that it \n    #is chaotic. Due to loading a varied batch size.  \n    val_mini_data_bch, val_mini_label_bch = Data_Loader.load_bch(0, 0.5)\n\n    #load the a single mini batch (data and labels) into memory.\n    Jiyam.eval()\n    with tor.no_grad():\n        val_predictions = Jiyam.forward(val_mini_data_bch, bch_size)\n        error_val = (Jiyam.backprop(val_predictions, val_mini_label_bch, 0)).item()\n    Jiyam.train(True)\n    \n    \n    print(f'{epoch}: Val error: {error_val/len(val_mini_data_bch)}')\n    tot_error[1].append(error_val/len(val_mini_data_bch))\n\n    if epoch > 2 and tot_error[1][-1] > tot_error[1][-2]:\n        if chances == 0:\n            break\n\n        else:\n            chances -= 1\n            continue\n\n\n\n    if (error_val/len(val_mini_data_bch)) < bst_minima:\n        print('Saving model...')\n        checkpoint = {'model': 'Jiyam',\n                      'version': 'Y4',\n                      'state_dict': Jiyam.state_dict(),\n                      'optimizer' : Jiyam.Optimizer.state_dict(),\n                      'validation error': tot_error[1],\n                      'training error': tot_error[0],\n                      'tot_epochs_trained' : epoch}\n        try:\n            checkpoint['tot_epochs_trained'] += trained_model['tot_epochs_trained']\n        except:\n            pass\n\n        bst_minima = (error_val/len(val_mini_data_bch))\n        tor.save(checkpoint, './Arpagus Jiyam.mdl')\n\n    del val_predictions, val_mini_data_bch, val_mini_label_bch\n    print('\\n')\n    tor.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:53:14.755287Z","iopub.execute_input":"2021-08-03T13:53:14.7556Z","iopub.status.idle":"2021-08-03T13:53:18.341568Z","shell.execute_reply.started":"2021-08-03T13:53:14.755571Z","shell.execute_reply":"2021-08-03T13:53:18.338415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### plt.plot(trained_model['training error'])","metadata":{"execution":{"iopub.status.busy":"2021-07-24T08:53:14.132333Z","iopub.execute_input":"2021-07-24T08:53:14.13272Z","iopub.status.idle":"2021-07-24T08:53:14.137739Z","shell.execute_reply.started":"2021-07-24T08:53:14.132688Z","shell.execute_reply":"2021-07-24T08:53:14.136542Z"}}},{"cell_type":"code","source":"#plot(trained_model['validation error'])","metadata":{"execution":{"iopub.status.busy":"2021-08-02T07:42:30.573117Z","iopub.execute_input":"2021-08-02T07:42:30.573524Z","iopub.status.idle":"2021-08-02T07:42:30.577727Z","shell.execute_reply.started":"2021-08-02T07:42:30.573473Z","shell.execute_reply":"2021-08-02T07:42:30.576782Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"3696.5 images for train and test on avg.","metadata":{}},{"cell_type":"code","source":"def filter__(tensor, device='cpu'):\n    output = []\n    \n    for i in tensor:\n        item = i.view(8*8, 2, 5)\n        pp_max = []\n        \n        for j in (item.mean(-2)):\n            num = j[0]\n            j[1:] = j[1:]\n            \n            if num >= 0.5:\n                pp_max.append(j)\n                continue\n        \n        r = sorted(pp_max, key=lambda x: x[0], reverse=True)\n        \n        if len(r) != 0:\n            output.append((tor.stack(r)))\n    \n    return output","metadata":{"execution":{"iopub.status.busy":"2021-08-02T07:42:11.200851Z","iopub.execute_input":"2021-08-02T07:42:11.201379Z","iopub.status.idle":"2021-08-02T07:42:11.210814Z","shell.execute_reply.started":"2021-08-02T07:42:11.201323Z","shell.execute_reply":"2021-08-02T07:42:11.210106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def n_maxs(preds, device='cuda'):\n    bounding_box = []\n    m = []\n    \n    for main_bbox in range(len(preds)): #take the main box.\n        for side_bbox in range(len(preds[main_bbox:])): #take another box which is not the main box.\n            \n            coor_pred = tor.stack((preds[main_bbox][1], preds[main_bbox][2]))\n            wh_pred = tor.stack((preds[main_bbox][3], preds[main_bbox][4]))\n            \n            coor_lbl = tor.stack((preds[side_bbox][1], preds[side_bbox][2]))\n            wh_lbl = tor.stack((preds[side_bbox][3], preds[side_bbox][4]))\n            \n            iou_ = iou_2(coor_pred, coor_lbl, wh_pred, wh_lbl, device)\n            #calculate iou between main and side box.\n            \n            if iou_ < 0.5: #if iou is less than 0.5 and is more than 0.\n                k_1 = preds[main_bbox][0]\n                k_2 = preds[side_bbox][0]\n                #then look at their confiedence scores.\n                \n                if k_1 > k_2 and (preds[main_bbox]).mean() not in m:\n                    bounding_box.append(preds[main_bbox])\n                    m.append((preds[main_bbox]).mean())\n                    \n                elif k_1 < k_2 and (preds[side_bbox]).mean() not in m:\n                    bounding_box.append(preds[side_bbox])\n                    m.append((preds[side_bbox]).mean())\n    \n    return bounding_box","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:39:04.774087Z","iopub.execute_input":"2021-08-03T13:39:04.774428Z","iopub.status.idle":"2021-08-03T13:39:04.787646Z","shell.execute_reply.started":"2021-08-03T13:39:04.774385Z","shell.execute_reply":"2021-08-03T13:39:04.786675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"a = Loader_2(path, 32, grid_size=256/(8), device=device)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T13:53:58.943536Z","iopub.execute_input":"2021-08-03T13:53:58.943875Z","iopub.status.idle":"2021-08-03T13:54:00.795773Z","shell.execute_reply.started":"2021-08-03T13:53:58.943842Z","shell.execute_reply":"2021-08-03T13:54:00.794929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"imgs, labels = a.load_bch(0, 0.5)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T14:02:04.798713Z","iopub.execute_input":"2021-08-03T14:02:04.799084Z","iopub.status.idle":"2021-08-03T14:02:05.100566Z","shell.execute_reply.started":"2021-08-03T14:02:04.799047Z","shell.execute_reply":"2021-08-03T14:02:05.099649Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"Jiyam.to(device)\nJiyam.eval()\nx = 8\nwith tor.no_grad():\n    loaded_labels = (Jiyam(imgs, bch_size=32)).view(32, x, x, 2, 5)\nprint(Jiyam.training)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T14:03:16.4197Z","iopub.execute_input":"2021-08-03T14:03:16.420054Z","iopub.status.idle":"2021-08-03T14:03:16.441083Z","shell.execute_reply.started":"2021-08-03T14:03:16.420019Z","shell.execute_reply":"2021-08-03T14:03:16.440049Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"index = tor.randint(32, (1,))\n#0, 1, 3, 4, 7, 15, 13, 18, 19, 16\n#bad apples: 0, 1, 19, 16, 10, 11, 9, 8, 7, 4\n#good apples: 20, 21, 22, 23, 14, 3, 2\n\nk_image = ((imgs[index]).cpu())\nnorm_image = (((imgs[index]).cpu()).squeeze().permute(1, 2, 0))\n\nk = tor.stack(sorted((a.re_labels(labels[index]).view(-1, 5)), key=lambda x: x[0], reverse=True))\nu = tor.stack(sorted((a.re_labels(loaded_labels[index])).view(-1, 5), key=lambda x: x[0], reverse=True))","metadata":{"execution":{"iopub.status.busy":"2021-08-03T14:03:16.649392Z","iopub.execute_input":"2021-08-03T14:03:16.649776Z","iopub.status.idle":"2021-08-03T14:03:16.696071Z","shell.execute_reply.started":"2021-08-03T14:03:16.649739Z","shell.execute_reply":"2021-08-03T14:03:16.695357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nfor item in u[:2]:#(labels[index].view(-1, 5)):\n    box = pat.Rectangle((item[1], item[2]), item[3], item[4], linewidth=2, edgecolor='red', facecolor='none')\n    ax.add_patch(box)\n\nax.imshow(norm_image)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T14:03:16.834808Z","iopub.execute_input":"2021-08-03T14:03:16.835131Z","iopub.status.idle":"2021-08-03T14:03:17.017136Z","shell.execute_reply.started":"2021-08-03T14:03:16.8351Z","shell.execute_reply":"2021-08-03T14:03:17.016248Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, ax = plt.subplots()\nprint(k[:5])\nfor item in k:#(labels[index].view(-1, 5)):\n    box = pat.Rectangle((item[1], item[2]), item[3], item[4], linewidth=2, edgecolor='red', facecolor='none')\n    ax.add_patch(box)\n\nax.imshow(norm_image)","metadata":{"execution":{"iopub.status.busy":"2021-08-03T14:03:17.026455Z","iopub.execute_input":"2021-08-03T14:03:17.026727Z","iopub.status.idle":"2021-08-03T14:03:17.46682Z","shell.execute_reply.started":"2021-08-03T14:03:17.026699Z","shell.execute_reply":"2021-08-03T14:03:17.465639Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"#condition = True\n#dum = u.clone()#(tor.stack(sorted(u, key=lambda x: x[0], reverse=True)))#, device)\n#while condition:\n#    print(len(dum))\n#    if len(dum) <= 1:\n#        condition=False\n#        break\n\n#    dum = n_maxs(dum, device)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#false negative\n#with a confidence of less than 0.5 but a iou of more than 0.5","metadata":{"execution":{"iopub.status.busy":"2021-08-03T06:27:26.785662Z","iopub.execute_input":"2021-08-03T06:27:26.786544Z","iopub.status.idle":"2021-08-03T06:27:26.794677Z","shell.execute_reply.started":"2021-08-03T06:27:26.786496Z","shell.execute_reply":"2021-08-03T06:27:26.793585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}