{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Optimizers ⚙️\n\n<img src = \"https://i.imgflip.com/2s1kjh.jpg\" width = 400>\n\nOptimizers are `algorithms` that `help deep learning models` learn `more effectively`. They do this by `updating the model's parameters` in a way `that minimizes the loss function`. The most common optimizers are \n* $Stochastic$ $Gradient$ $Descent$ $SGD$\n* $AdaGrad$ \n* $RMSProp$\n* $Adam$ ","metadata":{}},{"cell_type":"code","source":"! pip install segmentation_models_pytorch\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n        \nimport segmentation_models_pytorch as smp\n\nimport cv2\nimport os\nimport numpy as np \nfrom PIL import Image\nimport tqdm\nimport json\n\nimport tensorflow as tf\n\n\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\nimport torch\nfrom torchvision import transforms\nfrom torch.utils.data import DataLoader, Dataset\nimport torch.nn as nn\n\ntrain_dir = \"/kaggle/input/hubmap-hacking-the-human-vasculature/train\"\ntest_dir = \"/kaggle/input/hubmap-hacking-the-human-vasculature/test\"\n\na = A.Compose(\n            [\n                A.Resize(width = 512 , height = 512) , \n                A.Normalize(\n                    mean = [0 , 0 , 0] , \n                    std = [1 , 1 , 1] , \n                    max_pixel_value = 255\n        ) , ToTensorV2()\n            ]\n)\n\nclass hubmapDataset(Dataset):\n    \n    def __init__(self, image_dir, labels_file , augments = False):\n        \n        with open(labels_file, 'r') as json_file:\n            self.json_labels = [\n                json.loads(line) \n                for line in json_file\n            ]\n\n        self.image_dir = image_dir\n#         self.transform = transform\n        self.augments = augments\n\n    __len__ = lambda self : len(self.json_labels)    \n        \n    def __getitem__(self, idx):\n        \n        image_path = os.path.join(self.image_dir, f\"{self.json_labels[idx]['id']}.tif\")\n        image = Image.open(image_path)\n        \n        if self.augments:\n            \n            image = a(image = image)[\"image\"]\n        \n        mask = np.zeros((512, 512), dtype=np.float32)\n\n        for annot in self.json_labels[idx]['annotations']:\n\n            cords = annot['coordinates']\n            \n            if annot['type'] == \"blood_vessel\":\n                \n                for cord in cords:\n                    \n                    rr, cc = np.array([i[1] for i in cord]), np.asarray([i[0] for i in cord])\n                    \n                    mask[rr, cc] = 1\n\n        image = torch.tensor(np.array(image), dtype=torch.float32).permute(2, 0, 1)  # Shape: [C, H, W]\n        mask = torch.tensor(mask, dtype=torch.float32)\n\n#         if self.transform:\n#             image = self.transform(image)\n\n        return image, mask\n\ntrain_dataset = hubmapDataset(image_dir = train_dir, labels_file = '../input/hubmap-hacking-the-human-vasculature/polygons.jsonl')\ntrain_dataloader = DataLoader(train_dataset, batch_size = 4, shuffle = True)\n\nfor data in tqdm.tqdm(train_dataloader , total = len(train_dataloader)):\n    img , mask = data\n    \nimg = img[-1 , -1 , : , :]\nmask = mask[-1 , : , :]\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"jupyter":{"source_hidden":true},"execution":{"iopub.status.busy":"2023-06-24T04:10:30.841645Z","iopub.execute_input":"2023-06-24T04:10:30.842026Z","iopub.status.idle":"2023-06-24T04:11:47.026673Z","shell.execute_reply.started":"2023-06-24T04:10:30.841998Z","shell.execute_reply":"2023-06-24T04:11:47.025906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets assume we have $2$ images like ","metadata":{}},{"cell_type":"code","source":"img , mask","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:50.521661Z","iopub.execute_input":"2023-06-24T04:11:50.523834Z","iopub.status.idle":"2023-06-24T04:11:50.558343Z","shell.execute_reply.started":"2023-06-24T04:11:50.523803Z","shell.execute_reply":"2023-06-24T04:11:50.557410Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"If we try to visualize them ","metadata":{}},{"cell_type":"code","source":"print(\"Orginal Image\")\ntransforms.ToPILImage()(img)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:51.718457Z","iopub.execute_input":"2023-06-24T04:11:51.718813Z","iopub.status.idle":"2023-06-24T04:11:51.750529Z","shell.execute_reply.started":"2023-06-24T04:11:51.718787Z","shell.execute_reply":"2023-06-24T04:11:51.749299Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Masked Image\")\ntransforms.ToPILImage()(mask)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:53.623705Z","iopub.execute_input":"2023-06-24T04:11:53.624774Z","iopub.status.idle":"2023-06-24T04:11:53.637301Z","shell.execute_reply.started":"2023-06-24T04:11:53.624735Z","shell.execute_reply":"2023-06-24T04:11:53.636431Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets assume we want to calculate how different this image is from the other image. \n\nWe know that images are just a collection of pixels. Lets asusme we denote every pixel with a number. Both of these are a `black and white`. So neither we have the complexity of different channels nor we are bounded by a large range of values $(0 , 255)$. For laziness, we choose an image to be `black and white`. \n\nSo we can represent our images in the form of large `matrices`. \n\nLets get not much deep into this. We will assume that `God just gave us a way to calculate how different is an image from its mask.`","metadata":{}},{"cell_type":"code","source":"class God_Method(nn.Module):\n    \n    def __init__(self):\n        \n        super(God_Method,self).__init__()\n        \n        self.diceloss = smp.losses.DiceLoss(mode='binary')\n        self.binloss = smp.losses.SoftBCEWithLogitsLoss(reduction = 'mean' , smooth_factor = 0.1)\n\n    def forward(self, output, mask):\n        \n        output = torch.squeeze(output)\n        mask = torch.squeeze(mask)\n        \n        dice = self.diceloss(output , mask)\n        bce = self.binloss(output , mask)\n        \n        loss = dice * 0.7 + bce * 0.3\n        \n        return loss","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:55.201887Z","iopub.execute_input":"2023-06-24T04:11:55.202397Z","iopub.status.idle":"2023-06-24T04:11:55.208586Z","shell.execute_reply.started":"2023-06-24T04:11:55.202368Z","shell.execute_reply":"2023-06-24T04:11:55.207295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"our_method = God_Method()\nour_method(img , mask)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:11:55.624999Z","iopub.execute_input":"2023-06-24T04:11:55.625350Z","iopub.status.idle":"2023-06-24T04:11:55.668670Z","shell.execute_reply.started":"2023-06-24T04:11:55.625320Z","shell.execute_reply":"2023-06-24T04:11:55.667753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So we can say that according to the `God Method`, the difference between `img` and `mask` is around $50.7774$ (as of this run). Lets give this `difference` a fancy name `Loss`. Also we will give this `God Method` a different name as `Loss Function`.\n\nWe can assume that the less the `Loss Function` will be, the more similar the `image` will be with its `mask`. So if we try to create a model that takes `image` as input and lets say give an output similar to the dimensions of the image be `output_image`, will be a good model if the `Loss` between the `output_image` and the `mask` is less, if the models intention is to predict the `mask`\n\nWe can interpret that as the `Loss` will decrease the accuracy of the model will increase. So by calculating the `Loss` we can improve our model. \n\nNow hear me out, models are just a `group` of `matrices`. Matrices consists of numbers. So if we try to change our numbers in the matrices according to a `function` $f(x)$ that is dependent on the `Loss` $(x)$. Then might be at some point we can reach a `unique set of numbers` that gives the lowest `Loss`\n\nLets break it in more simples terms \n\n# 1 | Basic Terminologies ✏️\n\nFirst lets get a deep dive into some of the basic terminologies\n\n* $Slope$\n* $Diffrentiation$\n* $Intercept$","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt \nimport seaborn as sns \n\nfrom IPython.display import IFrame","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:56.072332Z","iopub.execute_input":"2023-06-24T04:11:56.073091Z","iopub.status.idle":"2023-06-24T04:11:56.278525Z","shell.execute_reply.started":"2023-06-24T04:11:56.073060Z","shell.execute_reply":"2023-06-24T04:11:56.276909Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets assume we have data like this","metadata":{}},{"cell_type":"code","source":"features = np.array([x for x in range(0 , 200 , 1)])\ntarget = np.array([x for x in range(0 , 400 , 2)])","metadata":{"_kg_hide-output":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:56.477475Z","iopub.execute_input":"2023-06-24T04:11:56.477834Z","iopub.status.idle":"2023-06-24T04:11:56.483510Z","shell.execute_reply.started":"2023-06-24T04:11:56.477805Z","shell.execute_reply":"2023-06-24T04:11:56.482254Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features , target","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:56.990546Z","iopub.execute_input":"2023-06-24T04:11:56.991849Z","iopub.status.idle":"2023-06-24T04:11:57.002489Z","shell.execute_reply.started":"2023-06-24T04:11:56.991800Z","shell.execute_reply":"2023-06-24T04:11:57.000793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets assume there is a connection between the `target` , and `features`. By human instacne we know that every element in `target` is just a `double` of the corresponding element in `features`, or $target  = 2XFeatures$. \n\nLets assume we change the target a little bit...","metadata":{}},{"cell_type":"code","source":"target = np.array([x + 1 for x in range(0 , 400 , 2)])","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:57.054990Z","iopub.execute_input":"2023-06-24T04:11:57.055353Z","iopub.status.idle":"2023-06-24T04:11:57.061142Z","shell.execute_reply.started":"2023-06-24T04:11:57.055324Z","shell.execute_reply":"2023-06-24T04:11:57.059834Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"features , target","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:57.093190Z","iopub.execute_input":"2023-06-24T04:11:57.093546Z","iopub.status.idle":"2023-06-24T04:11:57.100553Z","shell.execute_reply.started":"2023-06-24T04:11:57.093519Z","shell.execute_reply":"2023-06-24T04:11:57.099684Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Now what could be the trend here..., We can see the code above and with the help of that we can say. That `target` value is just the `double + 1` of the corresponding element in `features`. or $target = 2Xfeatures + 1$\n\nTill now the problem was really easy to solve, and thats why we used the brain only, But these are just examples. As we move closer to the real world. The examples/problems get difficulat and we find it harder to find proper trends in the two `arrays`. Thats we try to teach machine, how to find trend in the data. The formula we had before $target = 2Xfeature + 1$ is subjective to only one problem or a similar problem. But this formula can be generlized by the equation of `straight line`, which is $y = mx + b$\n\nSo what does this line means ???\n\nLets first try to plot the data we had on a scatter plot ","metadata":{}},{"cell_type":"code","source":"plt.scatter(features , target)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:57.157005Z","iopub.execute_input":"2023-06-24T04:11:57.157446Z","iopub.status.idle":"2023-06-24T04:11:57.407802Z","shell.execute_reply.started":"2023-06-24T04:11:57.157413Z","shell.execute_reply":"2023-06-24T04:11:57.406485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"You can see we got a sequence of dots that resembles kind of straight line. \n\nLets assume we have a line that tries to capture most of the points on this, like this","metadata":{}},{"cell_type":"code","source":"plt.scatter(features , target)\nplt.plot([0 , 200] , [0 , 400] , \"yellow\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:57.409249Z","iopub.execute_input":"2023-06-24T04:11:57.409581Z","iopub.status.idle":"2023-06-24T04:11:57.637949Z","shell.execute_reply.started":"2023-06-24T04:11:57.409552Z","shell.execute_reply":"2023-06-24T04:11:57.636599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Again by human intution we found the `best fit line`. But what if we want to generalize the things and kind of do not find the best fit line...?\n\nFirst of all lets get a little bit more deep into equation $y = mx + b$\n\nSo what does these terms resembles in this eqution. \n* `m` is the slope of the line\n\n## 1.1 | Slope of A function\n\n<img src = \"https://encrypted-tbn0.gstatic.com/images?q=tbn:ANd9GcRJCTVoq09OqEdc6MWsy7UJu2w9kLZ_J-IqxjgBDsn_gqHzXt_NWi4DaC-JZ3sCiTJ94jA&usqp=CAU\" width = 400>\n\nSlope of a function shows how steep a function is, or the direction of a function at a given point on the curve.\n\nLets assume we have this curve $y = 4x^2$ the slope of this curve will be $y = 8x$","metadata":{}},{"cell_type":"code","source":"IFrame(\"https://www.desmos.com/calculator/zluqu5vyuh\" , 1000 , 300)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:57.640167Z","iopub.execute_input":"2023-06-24T04:11:57.640582Z","iopub.status.idle":"2023-06-24T04:11:57.647204Z","shell.execute_reply.started":"2023-06-24T04:11:57.640547Z","shell.execute_reply":"2023-06-24T04:11:57.646088Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So how do we calculate the `slope` of a line???\n\nLets assume we have a function $y = f(x)$,. To find the slope of a function, we simply diffrenctiate the function, thus, the slope of this line will be $y^` = f^`(x)$\n\n## 1.2 | Diffrentiation\n\n<img src = \"https://preview.redd.it/differentiation-meme-v0-jt8ka3lrly2a1.png?auto=webp&s=c7261bc1ce1c7f57c2366e095d87db0478d1492d\" width = 400>\n\nDiffrentaition can be explained as getting a small value of a function.\n\nlets assume we have a function `y = sin(x)`\n\nA small strip at that function will demonstrate taking a derivative of that function `sin(x)`\n\nTaking about the function we had taken before that is $y = 4x^2$\n\nTaking its derivative we will get $$y = 8x$$ ($x{n^`} = nx^{n-1}$)\n\nSo the slope of $y = 4x^2$ can be represnted as ","metadata":{}},{"cell_type":"code","source":"IFrame(\"https://www.desmos.com/calculator/hrguwktg9q\" , 1000 , 300)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:57.650006Z","iopub.execute_input":"2023-06-24T04:11:57.650365Z","iopub.status.idle":"2023-06-24T04:11:57.659225Z","shell.execute_reply.started":"2023-06-24T04:11:57.650342Z","shell.execute_reply":"2023-06-24T04:11:57.658353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**If you want to know more aboud diffrentiation, here is [3Blue1Brown](https://www.youtube.com/@3blue1brown/featured) => [Essence Of Calculas](https://www.youtube.com/playlist?list=PLZHQObOWTQDMsr9K-rj53DwVRMYO3t5Yr)**\n\nSo now we have a basic idea of `slope`\n\n## 1.3 | Intercept\n\n<img src = \"https://media.makeameme.org/created/teacher-whats-the-e2c025118e.jpg\" width = 400>\n\nNow what `b` represents in the data. Usually it is called the `intercept`. Consider this graph of the equation $y = x$ or $y = 1x + b$ ","metadata":{}},{"cell_type":"code","source":"IFrame(\"https://www.desmos.com/calculator/gai0veg5fh\" , 1000 , 300)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:57.660480Z","iopub.execute_input":"2023-06-24T04:11:57.661224Z","iopub.status.idle":"2023-06-24T04:11:57.678187Z","shell.execute_reply.started":"2023-06-24T04:11:57.661194Z","shell.execute_reply":"2023-06-24T04:11:57.676912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This line passes the axis at $(0 , 0)$. These coordinatesare called as the `intercepts` of this line. If we make `b` or `intercept` as $1$. The line will then pass from $(1 , -1)$. Basically the `intercept` moves a line in a plane. \n\nWith that being said, Lets also undertand how the `slope` changes the line. If we make `m` as $2$. The line will rotate anti-clockwise. So as we increase the value of `m` or `slope`. The line moves anti-clockwise, And so the vice-versa, If we decrease the value of `m`, The slope will move in the clockwise direction. \n\nIn short tweeking the values of `m` and `b` or `slope` and `intercept`. We can move the line in any direction and in any way we want `as long as it resembles a straight line`. We still cannot bend the line \n\nSo now we have any data, we just need to difine the values of `slope` and `intercpet`. And we can get the best fit line. But still the question arises how do we generalize the values of these tuning parametes. \n\nIn simple word we can say, How can we find a relation between the data we have and these tuning parameters. So that we only need to define that relationship and then we can easily predict the values.\n\nLets think that the value assigned to the line is this ","metadata":{}},{"cell_type":"code","source":"plt.scatter(features , target)\nplt.plot([0 , 300] , [0 , 400] , \"yellow\")","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:57.679740Z","iopub.execute_input":"2023-06-24T04:11:57.680103Z","iopub.status.idle":"2023-06-24T04:11:57.901412Z","shell.execute_reply.started":"2023-06-24T04:11:57.680073Z","shell.execute_reply":"2023-06-24T04:11:57.900064Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"If we test this line on the training data only, we will find that this line is not correct. It is predicting points incorrect, We know that the best fit line we drew first, will predict points wiht lowest incorrect ones. For example the line we just defined if asked the corresponding value of $200$, it will say $250$. But rather it was $400$. There was some `error`, some `loss`, or some `cost` with the `actual` and `predicted` values.\n\nThis is the same as the loss we calucalted above for image\n\nFor measuring this loss, what we can do is find the difference between the `actual value` and the `predicted value`. A best fit line will give the lowest value of this difference.\n\nOne can deifne loss as $$Loss = actual - predicted$$\n\nWe only took the example of one value. but there are a large group of values. that can show the same trait, For that we can change the formula to \n\nLets denote $actual$ as $a$ and $predicted$ as $p$\n\n$$Loss = (a_1 - p_1) + (a_2 - p_2) + (a_3 - p_3) + ... + (a_n - p_n)$$\n\nor $$Loss = \\sum\\limits_{i = 1}^{n}a_i - p_i$$ or $$Loss = \\sum\\limits_{i = 1}^{n}(y_i - \\hat y_i)$$\n\nWhenever you see $\\hat y$, think of it as the `predicted value`\n\nNow lets assume we have data like this and a random line is drawn like this \n\n<img src = \"https://cdn-media-1.freecodecamp.org/images/MNskFmGPKuQfMLdmpkT-X7-8w2cJXulP3683\" width = 400>\n\nIf you look closely, a lot of error terms will tend to cancel out each other. We can also get into a state where the line is `not the best fit`, but still gives $0$ error. With the `Loss` we defined before, we are not chossing a `best fit line`. Rather we are chossing a line that is in the `middle` of those points. One way to counter this is to add a `modulus` function like this $$Loss = \\sum\\limits_{i = 1}^{n}|y - \\hat y|$$\n\nBut what is a modulus function. The function is nothing but converts, any negative numbers to postive. For example \n$|-1| = 1$","metadata":{}},{"cell_type":"code","source":"IFrame(\"https://www.desmos.com/calculator/kamxotjra2\" , 1000 , 300)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-06-24T04:11:57.903429Z","iopub.execute_input":"2023-06-24T04:11:57.903939Z","iopub.status.idle":"2023-06-24T04:11:57.912702Z","shell.execute_reply.started":"2023-06-24T04:11:57.903898Z","shell.execute_reply":"2023-06-24T04:11:57.911194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"But there is a problem with this function. A `modulus` is not diffrentiable. You might be thinking that why are we even seeing that part, like we we care for that. Why would you even diffrentiate a loss function. \n\nWe actually diffrentiate loss function in further steps, thats why we will not use the modulus function. \n\nAnother way of doing so is to, square the loss function like this $Loss = (y - \\hat y)^2$\n\nIts cool, its good and we can even diffrentiate this...\n\nNow we have a basic idea that we need to compute `m` and `b` for the lowest loss values. Now we should come to know how we can do this \n\nWhat if we somehow interelate the `losses` and `m and b`. \n\n\n# 2 | SGD 🔱\n\n<img src = \"https://i.redd.it/qrxyr8t2m1u51.jpg\" width = 500>\n\nLets assume we intialize the parameters randomly, like this \n","metadata":{}},{"cell_type":"code","source":"weights = np.random.randn(1)\nbiases = np.random.randn(1)\n\nweights , biases","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:11:57.914981Z","iopub.execute_input":"2023-06-24T04:11:57.915423Z","iopub.status.idle":"2023-06-24T04:11:57.928166Z","shell.execute_reply.started":"2023-06-24T04:11:57.915385Z","shell.execute_reply":"2023-06-24T04:11:57.926943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So the predicitions will be according to the equation $y = mx + b$","metadata":{}},{"cell_type":"code","source":"pred = weights * 30 + biases\npred","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:11:57.930783Z","iopub.execute_input":"2023-06-24T04:11:57.931152Z","iopub.status.idle":"2023-06-24T04:11:57.939339Z","shell.execute_reply.started":"2023-06-24T04:11:57.931125Z","shell.execute_reply":"2023-06-24T04:11:57.938145Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And its way far than what we had expected. ","metadata":{}},{"cell_type":"code","source":"loss = (pred - 60)\nloss","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:11:57.941036Z","iopub.execute_input":"2023-06-24T04:11:57.941434Z","iopub.status.idle":"2023-06-24T04:11:57.955117Z","shell.execute_reply.started":"2023-06-24T04:11:57.941403Z","shell.execute_reply":"2023-06-24T04:11:57.953921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our main motive is to reduce this loss as much as possible,. \n\nWhat if we subtract a small subset of the derivative of this loss from the parameters like this. The derivative of the loss will show us the steepness of the curve, and thus doing so might get us to the valeus of minimum loss. So how do we find the derivative of this function $Loss = (y - \\hat y)^2$. What we know is $\\hat y = mx + b$. COmputing this value in we get $$Loss = (y - mx - b)^2$$ Now we can diffrentiate the function\n\n## Diffrentiating wrt `b`\n$$\\frac {dLoss}{db}= \\frac {d}{db}(y - mx - b)^2$$\n$$= 2(y - mx - b)(-1)$$\n\n## Diffrentiating wrt `m`\n$$\\frac {dLoss}{dm} = \\frac {d}{dm}(y - mx - b)^2$$\n$$= 2(y - mx - b)(-x)$$","metadata":{}},{"cell_type":"code","source":"weights -= (-2* (60 - weights*30 - biases)) * 0.001\nbiases -= (2 * 30 * (60 - weights * 30 - biases)) * 0.01","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:11:57.956703Z","iopub.execute_input":"2023-06-24T04:11:57.957066Z","iopub.status.idle":"2023-06-24T04:11:57.966901Z","shell.execute_reply.started":"2023-06-24T04:11:57.957038Z","shell.execute_reply":"2023-06-24T04:11:57.965369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And if we then try to predict the values ","metadata":{}},{"cell_type":"code","source":"loss = (60 - (weights * 30 + biases))\nloss","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:11:57.969167Z","iopub.execute_input":"2023-06-24T04:11:57.969470Z","iopub.status.idle":"2023-06-24T04:11:57.981832Z","shell.execute_reply.started":"2023-06-24T04:11:57.969444Z","shell.execute_reply":"2023-06-24T04:11:57.981150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Our losses have been decreased, so lets do it again ","metadata":{}},{"cell_type":"code","source":"weights -= -2 * loss * 0.01\nbiases -= -2 * loss * 0.01\n\nloss = (60 - (weights * 30 + biases))\nloss","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:11:57.983794Z","iopub.execute_input":"2023-06-24T04:11:57.984101Z","iopub.status.idle":"2023-06-24T04:11:57.994791Z","shell.execute_reply.started":"2023-06-24T04:11:57.984078Z","shell.execute_reply":"2023-06-24T04:11:57.993927Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So now we know if we do this iteratively, we will minimise the loss, and iteratively we will reach the optimal values of `weights` or `m` and `biases` or `m`\n\nLets say we have runn this again and again for around 100 times ","metadata":{}},{"cell_type":"code","source":"for _ in range(100):\n    weights -= -2 * loss * 0.01\n    biases -= -2 * loss * 0.01\n    \n    loss = (60 - (weights * 30 + biases))","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:11:58.020187Z","iopub.execute_input":"2023-06-24T04:11:58.020778Z","iopub.status.idle":"2023-06-24T04:11:58.027035Z","shell.execute_reply.started":"2023-06-24T04:11:58.020752Z","shell.execute_reply":"2023-06-24T04:11:58.026128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Lets now see the values","metadata":{}},{"cell_type":"code","source":"weights , biases , loss","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:11:58.073655Z","iopub.execute_input":"2023-06-24T04:11:58.074219Z","iopub.status.idle":"2023-06-24T04:11:58.082646Z","shell.execute_reply.started":"2023-06-24T04:11:58.074182Z","shell.execute_reply":"2023-06-24T04:11:58.081257Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Though we have biases as high, but we have almost achived value of `weights`\n\nAlso we have made our `loss ~ 0` (as of this run)\n\nLets do this all again, and now we will also try to plot a graph","metadata":{}},{"cell_type":"code","source":"weights = abs(np.random.randn(1))\nbiases = abs(np.random.randn(1))\n\nlosses = []\n\nfor _ in range(100):\n    weights -= -2 * loss * 0.01\n    biases -= -2 * loss * 0.01\n    \n    loss = (60 - (weights * 30 + biases))\n    losses.append(loss)","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:11:58.142850Z","iopub.execute_input":"2023-06-24T04:11:58.143657Z","iopub.status.idle":"2023-06-24T04:11:58.150340Z","shell.execute_reply.started":"2023-06-24T04:11:58.143627Z","shell.execute_reply":"2023-06-24T04:11:58.149021Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.lineplot(np.array(losses))","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:11:58.185684Z","iopub.execute_input":"2023-06-24T04:11:58.186246Z","iopub.status.idle":"2023-06-24T04:11:58.444663Z","shell.execute_reply.started":"2023-06-24T04:11:58.186215Z","shell.execute_reply":"2023-06-24T04:11:58.443106Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"As we can see we have greatly decreased our losses \n\n## 2.1 | Functionalities\nWe have made out our **SGD**, now we need to add some functionalities to it. We can get funcitonalites form **[Tensorflow](https://www.tensorflow.org/)=>[Keras](https://keras.io/about/)=>[Optimizers](https://keras.io/api/optimizers/)=>[Experimental](https://www.tensorflow.org/api_docs/python/tf/keras/optimizers/experimental)=>[SGD](https://www.tensorflow.org/api_docs/python/tf/keras/optimizers/experimental/SGD)**. The type of optimization tensorflow uses is a little bit different. It initializes and keeps the computation in the optimizer function and do the `fit` and `predict` on another function that uses this function. This is done so as to make a generalized optimizer for both `Linear` and `Sequential` models. But we will try to use the vanilla gradient descent and combine the initialization and the `fit` methods.\n\n|Name|Atrribute|Info|Input Type|Default|Check Point\n|---|---|---|---|---|---\n|$List$ $Of$ $Columns$|||||✅\n|$Learning$ $Rate$|`learning_rate`|A Tensor, floating point value, or a schedule that is a **[tf.keras.optimizers.schedules.LearningRateSchedule](https://www.tensorflow.org/api_docs/python/tf/keras/optimizers/schedules/LearningRateSchedule)**, or a callable that takes no arguments and returns the actual value to use.|`float`|$0.001$|✅\n|$Momentum$|`momentum`|float hyperparameter >= 0 that accelerates gradient descent in the relevant direction and dampens oscillations.|`float`|$0$|✅\n|$Nestrov$|`nesterov`| Whether to apply Nesterov momentum|`bool`|$False$|✅\n|$Weight Decay$|`weight_decay`|If set, weight decay is applied|`float`|$None$|✅\n|$Clip$ $Norm$|`clipnorm`|If set, the gradient of each weight is individually clipped so that its norm is no higher than this value.|`Float`|$0$|✅\n|$EMA$|`use_ema`|EMA consists of computing an exponential moving average of the weights of the model (as the weight values change after each training batch), and periodically overwriting the weights with their moving average.|`Bool`|`False`|✅\n|$EMA$ $Momentum$|`ema_momentum`|Only used if use_ema=True. This is the momentum to use when computing the EMA of the model's weights: `new_average = ema_momentum * old_average + (1 - ema_momentum) * current_variable_value`.|`Float`|$0.99$|✅\n\n\n\n## 2.1.1 | List Of Columns \n\nThis function will only work if there are only $2$ columns, one `feature` and the other one `target`. What if the user gives out a list of columns. For this we nee dto take two different arguemnts form the user and work on them differently","metadata":{}},{"cell_type":"code","source":"def SGD(X , y):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n\n    losses = []\n\n    for _ in range(100):\n\n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum((pred - y) ** 2)\n        losses.append(loss)\n        \n        weights -= -2 * loss * 0.01\n        biases -= -2 * loss * 0.01\n\n    return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:11:58.633067Z","iopub.execute_input":"2023-06-24T04:11:58.633592Z","iopub.status.idle":"2023-06-24T04:11:58.639716Z","shell.execute_reply.started":"2023-06-24T04:11:58.633562Z","shell.execute_reply":"2023-06-24T04:11:58.638122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.1.2 | Learning Rate\n\nYou remeber we were taking a small part of the loss, not the whole loss. The parameter that defines how much loss we are taking is called the **Learning Rate** \n\nYou might be thinking that it is not that important, but it is really important concept, a higher learning rate has a high chance that you will never converge to the model, a low very low learning rate means that you will take a very long time to converge for the model \n\nHere is a very good image that explains the importance of the learning rate \n\n<img src = \"https://www.researchgate.net/profile/Hajar-Feizi/publication/341609757/figure/fig2/AS:894745802977280@1590335431623/Changes-in-the-loss-function-vs-the-epoch-by-the-learning-rate-40.png\" width = 400>\n\nIt will be really easy for us to apply this functinality, we just need to change some varaibales","metadata":{}},{"cell_type":"code","source":"def SGD(X , y , learning_rate = 0.01):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n\n    losses = []\n\n    for _ in range(100):\n\n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum((pred - y) ** 2)\n        losses.append(loss)\n        \n        weights -= -2 * loss * learning_rate\n        biases -= -2 * loss * learning_rate\n\n    return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:11:59.475834Z","iopub.execute_input":"2023-06-24T04:11:59.477057Z","iopub.status.idle":"2023-06-24T04:11:59.484284Z","shell.execute_reply.started":"2023-06-24T04:11:59.477003Z","shell.execute_reply":"2023-06-24T04:11:59.482735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.1.3 | Momentum \n\nThe probelem with gradient descent is this \n\n<img src = \"https://winder.ai/blog/2017/img/gradient_descent.svg\">\n\nNotice that before we were taking larger steps and as we reached the minimum value, are step size shortened. It is kind of a blessing as well as sometimes a curse for us. The problem is while getting very near to the minimum value, the step size gets so small. that it merely becomes $0$. not $0$ (that means we have reahed the minimum value). To cunter this we introduce momentum to the formula \n\nLets assume you are going to a place that you dont know. Like you dont know where it exists. So what you do is you ask people in the way that where is the place, and they show you the directions. Lets assume the directions are only right $(->)$ and left $(<-)$. So you ask $1^{st}$ person and he says to go to the right $(->)$, then you ask the $2^{nd}$ person and he also asks you to go right $(->)$, so you gain a confidence that you are going right. so you increase your speed. you might skip the $3^{rd}$ person and directly ask the $4^{th}$. \n\nLets assume the same situation from start. So you ask the $1^{st}$ person and he says to go to the right $(->)$ and then you ask the $2^{nd}$ person and he says to go left. But your inner instinct says that you are in the correct direction, but due to the influence of the $2^{nd}$ person, you will go slowly.\n\nSo you increase and decrease you speed on the basis of the previously gained knowledge. This is the same concept `momentum`, tries to implement. \n\nOne way of doing so is to add the commulative sum of all the gradients we achived previously like this \n\n$$w_{n+1} = w_n - \\frac {dLoss}{dw} \\alpha + (\\sum\\limits_{i = 1}^{n}w_i)$$\n\n$$b_{n+1} = b_n - \\frac {dLoss}{db} \\alpha + (\\sum\\limits_{i = 1}^{n}b_i)$$\n\nBut there are majorly $2$ probelems with this formula $:-$\n* Rather than fastening the gradients a little bit, it will fasten them exponentially.\n* This formula values every gradient equal, \n\nTo rectify this probelm we take the weighted average sum of all the gradients. or we actually multiply the sum with some constant. we change the formula a little bit \n\n$$w_{n+1} = w_n - (\\beta w_m + \\alpha(1 - \\beta) \\frac {dLoss}{dw})$$\n\n$$b_{n+1} = b_n - (\\beta b_m + \\alpha(1 - \\beta) \\frac {dLoss}{db})$$\n","metadata":{}},{"cell_type":"code","source":"def SGD(X , y , learning_rate = 0.01 , momentum = 0):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n\n    m_weights = 0\n    m_biases = 0\n\n    losses = []\n\n    for _ in range(100):\n\n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum(pred - y)\n        losses.append(loss)\n\n        m_weights = momentum * m_weights + (1 - momentum) * (-2 * loss)\n        m_biases = momentum * m_biases + (1 - momentum) * (-2 * loss)\n        \n        weights -= m_weights[epochs + 1] * 0.01\n        biases -= m_biases[epochs + 1] * 0.01\n\n    return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:12:00.223166Z","iopub.execute_input":"2023-06-24T04:12:00.223546Z","iopub.status.idle":"2023-06-24T04:12:00.231308Z","shell.execute_reply.started":"2023-06-24T04:12:00.223518Z","shell.execute_reply":"2023-06-24T04:12:00.229979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.1.4 | Nestrov\nWhile SGD with momentum can be effective at overcoming oscillations in the cost function, NAG has been shown to converge faster and more reliably in many cases, and thats why we will give the functionality of this function to \n$$w^n = w^{n-1} - \\beta w_v^{n-1} + (1-\\beta)(w^{n-1} - \\beta w_v^{n-1})$$\n$$b^n = b^{n-1} - \\beta b_v^{n-1} + (1-\\beta)(b^{n-1} - \\beta b_v^{n-1})$$","metadata":{}},{"cell_type":"code","source":"def SGD(X , y , learning_rate = 0.01 , momentum = 0 , nestrov = False):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n    \n    m_weights = 0\n    m_biases = 0\n    \n    losses = []\n    \n    for _ in range(100):\n    \n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum((pred - y) ** 2)\n        losses.append(loss)\n        \n        if nestrov :\n\n            m_weights = (momentum * m_weights + ((1 - momentum) * (weights - momentum * m_weights))) * -2 * loss \n            m_biases = (momentum * m_biases + ((1 - momentum) * (weights - momentum * m_biases))) * -2 * loss\n\n        else :\n\n            m_weights = momentum * m_weights + (1 - momentum) * -2 * loss\n            m_biases = momentum * m_biases + (1 - momentum) * -2 * loss\n\n        weights -= m_weights * learning_rate\n        biases -= m_biases * learning_rate\n\n    return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:12:00.697742Z","iopub.execute_input":"2023-06-24T04:12:00.698163Z","iopub.status.idle":"2023-06-24T04:12:00.709108Z","shell.execute_reply.started":"2023-06-24T04:12:00.698131Z","shell.execute_reply":"2023-06-24T04:12:00.707354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.1.5 | Weight Decay \nWeight decay is a powerful regularization technique that can help to prevent overfitting and improve the generalization performance of machine learning models trained with SGD optimization.\n$$u = u + \\gamma u$$","metadata":{}},{"cell_type":"code","source":"def SGD(X , y , learning_rate = 0.01 , momentum = 0 , nestrov = False , weight_decay = None):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n    \n    m_weights = 0\n    m_biases = 0\n    \n    losses = []\n    \n    for _ in range(100):\n    \n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum((pred - y) ** 2)\n        losses.append(loss)\n        \n        if nestrov :\n\n            m_weights = (momentum * m_weights + (1 - momentum) * (weights - momentum * m_weights)) * -2 * loss \n            m_biases = (momentum * m_biases + (1 - momentum) * (weights - momentum * m_biases)) * -2 * loss\n\n        else :\n\n            m_weights = momentum * m_weights + (1 - momentum) * -2 * loss\n            m_biases = momentum * m_biases + (1 - momentum) * -2 * loss\n\n        weights -= (m_weights + weight_decay * m_weights) * learning_rate\n        biases -= (m_biases + weight_decay * m_biases) * learning_rate\n\n    return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:12:01.321999Z","iopub.execute_input":"2023-06-24T04:12:01.322329Z","iopub.status.idle":"2023-06-24T04:12:01.329906Z","shell.execute_reply.started":"2023-06-24T04:12:01.322301Z","shell.execute_reply":"2023-06-24T04:12:01.328926Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.1.6 | Clip Norm\nSometimes there is a chance that the weights go skyrocketting, which is generally considered a bad idea, parameters in the range of $(−1,1)$ are considered to be good. Thats why sometimes we use clip-norm , that generates a upper baseline for the parameters.","metadata":{}},{"cell_type":"code","source":"def SGD(X , y , learning_rate = 0.01 , momentum = 0 , nestrov = False , weight_decay = None , clip_norm = None):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n    \n    m_weights = 0\n    m_biases = 0\n    \n    losses = []\n    \n    for _ in range(100):\n    \n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum((pred - y) ** 2)\n        losses.append(loss)\n        \n        if nestrov :\n\n            m_weights = (momentum * m_weights + (1 - momentum) * (weights - momentum * m_weights)) * -2 * loss \n            m_biases = (momentum * m_biases + (1 - momentum) * (weights - momentum * m_biases)) * -2 * loss\n\n        else :\n\n            m_weights = momentum * m_weights + (1 - momentum) * -2 * loss\n            m_biases = momentum * m_biases + (1 - momentum) * -2 * loss\n\n        if clip_norm != None:\n            \n            weights = np.clip(weights , weights , clip_norm)\n            biases = np.clip(biases , biases , clip_norm)\n\n        weights -= (m_weights + weight_decay * m_weights) * learning_rate\n        biases -= (m_biases + weight_decay * m_biases) * learning_rate\n\n    return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:12:02.015310Z","iopub.execute_input":"2023-06-24T04:12:02.015986Z","iopub.status.idle":"2023-06-24T04:12:02.027077Z","shell.execute_reply.started":"2023-06-24T04:12:02.015953Z","shell.execute_reply":"2023-06-24T04:12:02.025411Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.1.7 | Use EMA\nWether to use the estimated momentum average or not. The major use of this function is when `ema_momentum` is given\n\n It is a technique used to smooth out the noisy gradients that are produced by Stochastic Gradient Descent (SGD). This can help to improve the convergence of SGD and make it more robust to noise.","metadata":{}},{"cell_type":"code","source":"def SGD(X , y , learning_rate = 0.01 , momentum = 0 , nestrov = False , weight_decay = None , clip_norm = None , ema = False):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n    \n    m_weights = 0\n    m_biases = 0\n    \n    losses = []\n    \n    for _ in range(100):\n    \n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum((pred - y) ** 2)\n        losses.append(loss)\n        \n        if nestrov :\n\n            m_weights = (momentum * m_weights + (1 - momentum) * (weights - momentum * m_weights)) * -2 * loss \n            m_biases = (momentum * m_biases + (1 - momentum) * (weights - momentum * m_biases)) * -2 * loss\n\n        else :\n\n            m_weights = momentum * m_weights + (1 - momentum) * -2 * loss\n            m_biases = momentum * m_biases + (1 - momentum) * -2 * loss\n\n        if ema:pass\n\n        if clip_norm != None:\n            \n            weights = np.clip(weights , weights , clip_norm)\n            biases = np.clip(biases , biases , clip_norm)\n\n        if clip_value != None:\n            \n            m_weights = np.clip(m_weights , m_weights , clip_value)\n            m_biases = np.clip(m_biases , m_biases , clip_value)\n\n        weights -= (m_weights + weight_decay * m_weights) * learning_rate\n        biases -= (m_biases + weight_decay * m_biases) * learning_rate\n\n    return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:12:02.549534Z","iopub.execute_input":"2023-06-24T04:12:02.549882Z","iopub.status.idle":"2023-06-24T04:12:02.558519Z","shell.execute_reply.started":"2023-06-24T04:12:02.549844Z","shell.execute_reply":"2023-06-24T04:12:02.557549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.1.8 | EMA Momentum\nIn RMSprop, the exponential moving average (EMA) of the gradient with momentum is used to improve convergence and prevent oscillations during training. The EMA of the gradient with momentum is updated at each iteration and incorporates information about the previous gradients and the current gradient to produce a more stable and consistent update direction. This helps to smooth out the noise in the gradient estimates and helps the optimizer to move more directly towards the minimum of the loss function.","metadata":{}},{"cell_type":"code","source":"def SGD(X , y , learning_rate = 0.01 , momentum = 0 , nestrov = False , weight_decay = None , clip_norm = None , ema = False , ema_momentum = 0.99):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n    \n    m_weights = 0\n    m_biases = 0\n    \n    losses = []\n    \n    for _ in range(100):\n    \n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum((pred - y) ** 2)\n        losses.append(loss)\n        \n        if nestrov :\n\n            m_weights = (momentum * m_weights + (1 - momentum) * (weights - momentum * m_weights)) * -2 * loss \n            m_biases = (momentum * m_biases + (1 - momentum) * (weights - momentum * m_biases)) * -2 * loss\n\n        else :\n\n            m_weights = momentum * m_weights + (1 - momentum) * -2 * loss\n            m_biases = momentum * m_biases + (1 - momentum) * -2 * loss\n\n        if ema:\n            m_weights = ema_momentum * m_weights + (1 - ema_momentum) * -2 * loss\n            m_biases = ema_momentum * m_biases + (1 - ema_momentum) * -2 * loss\n\n        if clip_norm != None:\n            \n            weights = np.clip(weights , weights , clip_norm)\n            biases = np.clip(biases , biases , clip_norm)\n\n        if clip_value != None:\n            \n            m_weights = np.clip(m_weights , m_weights , clip_value)\n            m_biases = np.clip(m_biases , m_biases , clip_value)\n\n        weights -= (m_weights + weight_decay * m_weights) * learning_rate\n        biases -= (m_biases + weight_decay * m_biases) * learning_rate\n\n    return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:12:03.088916Z","iopub.execute_input":"2023-06-24T04:12:03.089298Z","iopub.status.idle":"2023-06-24T04:12:03.099243Z","shell.execute_reply.started":"2023-06-24T04:12:03.089270Z","shell.execute_reply":"2023-06-24T04:12:03.097942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.2 | Methods\n\nNow to add more functionalities, we will add different methods to our alogirthm, You can acces the list of methods we are going to use from the same link before we used for `Functionalities`. \n\n|Method|Attribute|Info|Applied\n|---|---|---|---\n|$$Build$$|`build`|Initialize optimizer variables. SGD optimizer has one variable momentums, only set if self.momentum is not 0.|✅\n|$Compute_-Gradients$|`compute_gradients`|Compute gradients of loss on trainable variables.|✅\n|$Minimize$|`minimize`|Minimize loss by updating `var_list`.|✅\n|$Update$ $Step$|`update_step`|Update step given gradient and the associated model variable.|✅\n\n## 2.2.1 | Build\nThis method initializes all the values in the `SGD`, you can understand this as a constructor, but not perfectly as the one. \n\nWe will give the user a functionality, that they can intialize there own variables. At the starting the `weights` and `baises` are `None`. If the user passes their own, the function first checks that they are or correct shape and if True assigns the values, else initialize random values ","metadata":{}},{"cell_type":"code","source":"class SGD:\n    \n    def __init__(self , \n                 X , y , \n                 learning_rate = 0.01 , momentum = 0 , \n                 nestrov = False , weight_decay = 0 , \n                 clip_norm = None , clip_value = None , \n                 use_ema = False , ema_momentum = 0):\n        \n        self.X = X\n        self.y = y\n        self.learning_rate = learning_rate\n        self.momentum = momentum\n        self.nestrov = nestrov\n        self.weight_decay = weight_decay\n        self.clip_norm = clip_norm\n        self.clip_value = clip_value\n        self.use_ema = use_ema\n        self.ema_momentum = ema_momentum\n    \n    def build(self , weights = None , biases = None):\n        \n        if weights == None:\n            \n            self.weights = abs(np.random.randn(self.X.shape[0]))\n        \n        else :\n            \n            if weights.shape[0] == self.X.shape[1]:\n                \n                self.weights = weights\n            \n            else :\n                \n                self.weights = abs(np.random.randn(self.X.shape[1]))\n                \n                raise UserWarning(\"Values do not match with the `X` , intializing random Values\")\n                \n        \n        if biases == None:\n            \n            self.biases = abs(np.random.randn(1))\n        \n        else :\n            \n            if biases.shape[0] == 1:\n                \n                self.biases = biases\n            \n            else :\n                \n                self.biases = abs(np.random.randn(1))\n                \n                raise UserWarning(\"Values do not match with the `X` , intializing random Values\")\n    \n        self.m_weights = 0\n        self.m_biases = 0\n    \n        self.losses = []\n    \n        for _ in range(100):\n        \n            pred = np.sum((weights * self.X).T) + biases\n            \n            loss = np.sum((pred - self.y) ** 2)\n            self.losses.append(loss)\n            \n            if self.nestrov :\n\n                self.m_weights = (self.momentum * self.m_weights + (1 - self.momentum) * (self.weights - self.momentum * self.m_weights)) * -2 * loss \n                self.m_biases = (self.momentum * self.m_biases + (1 - self.momentum) * (self.biases - self.momentum * self.m_biases)) * -2 * loss\n\n            else :\n\n                self.m_weights = self.momentum * self.m_weights + (1 - self.momentum) * -2 * loss\n                self.m_biases = self.momentum * self.m_biases + (1 - self.momentum) * -2 * loss\n            \n            if self.use_ema:\n                \n                self.m_weights = self.ema_momentum * self.m_weights + (1 - self.ema_momentum) * -2 * loss\n                self.m_biases = self.ema_momentum * self.m_biases + (1 - self.ema_momentum) * -2 * loss\n                \n            if self.clip_norm != None:\n                \n                weights = np.clip(weights , weights , self.clip_norm)\n                biases = np.clip(biases , biases , self.clip_norm)\n            \n            if self.clip_value != None:\n                \n                self.m_weights = np.clip(self.m_weights , self.m_weights , self.clip_value)\n                self.m_biases = np.clip(self.m_biases , self.m_biases , self.clip_value)\n            \n            weights -= (self.m_weights + self.weight_decay * self.m_weights) * self.learning_rate\n            biases -= (self.m_biases + self.weight_decay * self.m_biases) * self.learning_rate\n\n        return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:12:03.165096Z","iopub.execute_input":"2023-06-24T04:12:03.165473Z","iopub.status.idle":"2023-06-24T04:12:03.182467Z","shell.execute_reply.started":"2023-06-24T04:12:03.165444Z","shell.execute_reply":"2023-06-24T04:12:03.180701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.2.2 | Compute Gradients\n\nCompuute the loss function and gradients for the losses ","metadata":{}},{"cell_type":"code","source":"class SGD:\n    \n    def __init__(self , \n                 X , y , \n                 learning_rate = 0.01 , momentum = 0 , \n                 nestrov = False , weight_decay = 0 , \n                 clip_norm = None , clip_value = None , \n                 use_ema = False , ema_momentum = 0):\n        \n        self.X = X\n        self.y = y\n        self.learning_rate = learning_rate\n        self.momentum = momentum\n        self.nestrov = nestrov\n        self.weight_decay = weight_decay\n        self.clip_norm = clip_norm\n        self.clip_value = clip_value\n        self.use_ema = use_ema\n        self.ema_momentum = ema_momentum\n    \n    def build(self , weights = None , biases = None):\n        \n        if weights == None:\n            \n            self.weights = abs(np.random.randn(self.X.shape[0]))\n        \n        else :\n            \n            if weights.shape[0] == self.X.shape[1]:\n                \n                self.weights = weights\n            \n            else :\n                \n                self.weights = abs(np.random.randn(self.X.shape[1]))\n                \n                raise UserWarning(\"Values do not match with the `X` , intializing random Values\")\n                \n        \n        if biases == None:\n            \n            self.biases = abs(np.random.randn(1))\n        \n        else :\n            \n            if biases.shape[0] == 1:\n                \n                self.biases = biases\n            \n            else :\n                \n                self.biases = abs(np.random.randn(1))\n                \n                raise UserWarning(\"Values do not match with the `X` , intializing random Values\")\n    \n        self.m_weights = 0\n        self.m_biases = 0\n    \n        self.losses = []\n\n    def comput_gradients(self , weights , biases):\n        \n            pred = np.sum((weights * self.X).T) + biases\n            \n            loss = np.sum((pred - self.y) ** 2)\n            \n            yield loss\n            \n            self.losses.append(loss)\n            \n        #     if self.nestrov :\n\n        #         self.m_weights = (self.momentum * self.m_weights + (1 - self.momentum) * (self.weights - self.momentum * self.m_weights)) * -2 * loss \n        #         self.m_biases = (self.momentum * self.m_biases + (1 - self.momentum) * (self.biases - self.momentum * self.m_biases)) * -2 * loss\n\n        #     else :\n\n        #         self.m_weights = self.momentum * self.m_weights + (1 - self.momentum) * -2 * loss\n        #         self.m_biases = self.momentum * self.m_biases + (1 - self.momentum) * -2 * loss\n            \n        #     if self.use_ema:\n                \n        #         self.m_weights = self.ema_momentum * self.momentum + (1 - self.ema_momentum) * -2 * loss\n        #         self.m_biases = self.ema_momentum * self.momentum + (1 - self.ema_momentum) * -2 * loss\n                \n        #     if self.clip_norm != None:\n                \n        #         weights = np.clip(weights , weights , self.clip_norm)\n        #         biases = np.clip(biases , biases , self.clip_norm)\n            \n        #     if self.clip_value != None:\n                \n        #         self.m_weights = np.clip(self.m_weights , self.m_weights , self.clip_value)\n        #         self.m_biases = np.clip(self.m_biases , self.m_biases , self.clip_value)\n            \n        #     weights -= (self.m_weights + self.weight_decay * self.m_weights) * self.learning_rate\n        #     biases -= (self.m_biases + self.weight_decay * self.m_biases) * self.learning_rate\n\n        # return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:12:03.233989Z","iopub.execute_input":"2023-06-24T04:12:03.234324Z","iopub.status.idle":"2023-06-24T04:12:03.245385Z","shell.execute_reply.started":"2023-06-24T04:12:03.234298Z","shell.execute_reply":"2023-06-24T04:12:03.244626Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.2.3 | Minimize \nMinimze the loss values","metadata":{}},{"cell_type":"code","source":"class SGD:\n    \n    def __init__(self , \n                 X , y , \n                 learning_rate = 0.01 , momentum = 0 , \n                 nestrov = False , weight_decay = 0 , \n                 clip_norm = None , clip_value = None , \n                 use_ema = False , ema_momentum = 0):\n        \n        self.X = X\n        self.y = y\n        self.learning_rate = learning_rate\n        self.momentum = momentum\n        self.nestrov = nestrov\n        self.weight_decay = weight_decay\n        self.clip_norm = clip_norm\n        self.clip_value = clip_value\n        self.use_ema = use_ema\n        self.ema_momentum = ema_momentum\n    \n    def build(self , weights = None , biases = None):\n        \n        if weights == None:\n            \n            self.weights = abs(np.random.randn(self.X.shape[0]))\n        \n        else :\n            \n            if weights.shape[0] == self.X.shape[1]:\n                \n                self.weights = weights\n            \n            else :\n                \n                self.weights = abs(np.random.randn(self.X.shape[1]))\n                \n                raise UserWarning(\"Values do not match with the `X` , intializing random Values\")\n                \n        \n        if biases == None:\n            \n            self.biases = abs(np.random.randn(1))\n        \n        else :\n            \n            if biases.shape[0] == 1:\n                \n                self.biases = biases\n            \n            else :\n                \n                self.biases = abs(np.random.randn(1))\n                \n                raise UserWarning(\"Values do not match with the `X` , intializing random Values\")\n    \n        self.m_weights = 0\n        self.m_biases = 0\n    \n        self.losses = []\n\n    def comput_gradients(self , weights , biases):\n        \n            pred = np.sum((weights * self.X).T) + biases\n            \n            loss = np.sum((pred - self.y) ** 2)\n            \n            yield loss\n            \n            self.losses.append(loss)\n            \n        #     if self.nestrov :\n\n        #         self.m_weights = (self.momentum * self.m_weights + (1 - self.momentum) * (self.weights - self.momentum * self.m_weights)) * -2 * loss \n        #         self.m_biases = (self.momentum * self.m_biases + (1 - self.momentum) * (self.biases - self.momentum * self.m_biases)) * -2 * loss\n\n        #     else :\n\n        #         self.m_weights = self.momentum * self.m_weights + (1 - self.momentum) * -2 * loss\n        #         self.m_biases = self.momentum * self.m_biases + (1 - self.momentum) * -2 * loss\n            \n        #     if self.use_ema:\n                \n        #         self.m_weights = self.ema_momentum * self.momentum + (1 - self.ema_momentum) * -2 * loss\n        #         self.m_biases = self.ema_momentum * self.momentum + (1 - self.ema_momentum) * -2 * loss\n    def minimize(self):                \n        if self.clip_norm != None:\n            \n            weights = np.clip(weights , weights , self.clip_norm)\n            biases = np.clip(biases , biases , self.clip_norm)\n        \n        if self.clip_value != None:\n            \n            self.m_weights = np.clip(self.m_weights , self.m_weights , self.clip_value)\n            self.m_biases = np.clip(self.m_biases , self.m_biases , self.clip_value)\n        \n        weights -= (self.m_weights + self.weight_decay * self.m_weights) * self.learning_rate\n        biases -= (self.m_biases + self.weight_decay * self.m_biases) * self.learning_rate\n\n        # return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:12:03.340565Z","iopub.execute_input":"2023-06-24T04:12:03.340928Z","iopub.status.idle":"2023-06-24T04:12:03.354836Z","shell.execute_reply.started":"2023-06-24T04:12:03.340901Z","shell.execute_reply":"2023-06-24T04:12:03.353740Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.2.4 | Update_Step","metadata":{}},{"cell_type":"code","source":"class SGD:\n    \n    def __init__(self , \n                 X , y , \n                 learning_rate = 0.01 , momentum = 0 , \n                 nestrov = False , weight_decay = 0 , \n                 clip_norm = None , clip_value = None , \n                 use_ema = False , ema_momentum = 0):\n        \n        self.X = X\n        self.y = y\n        self.learning_rate = learning_rate\n        self.momentum = momentum\n        self.nestrov = nestrov\n        self.weight_decay = weight_decay\n        self.clip_norm = clip_norm\n        self.clip_value = clip_value\n        self.use_ema = use_ema\n        self.ema_momentum = ema_momentum\n    \n    def build(self , weights = None , biases = None):\n        \n        if weights == None:\n            \n            self.weights = abs(np.random.randn(self.X.shape[0]))\n        \n        else :\n            \n            if weights.shape[0] == self.X.shape[1]:\n                \n                self.weights = weights\n            \n            else :\n                \n                self.weights = abs(np.random.randn(self.X.shape[1]))\n                \n                raise UserWarning(\"Values do not match with the `X` , intializing random Values\")\n                \n        \n        if biases == None:\n            \n            self.biases = abs(np.random.randn(1))\n        \n        else :\n            \n            if biases.shape[0] == 1:\n                \n                self.biases = biases\n            \n            else :\n                \n                self.biases = abs(np.random.randn(1))\n                \n                raise UserWarning(\"Values do not match with the `X` , intializing random Values\")\n    \n        self.m_weights = 0\n        self.m_biases = 0\n    \n        self.losses = []\n\n    def comput_gradients(self , weights , biases):\n        \n            pred = np.sum((weights * self.X).T) + biases\n            \n            loss = np.sum((pred - self.y) ** 2)\n            \n            yield loss\n            \n            self.losses.append(loss)\n            \n    def update_step(self , loss):\n\n        if self.nestrov :\n\n            self.m_weights = (self.momentum * self.m_weights + (1 - self.momentum) * (self.weights - self.momentum * self.m_weights)) * -2 * loss \n            self.m_biases = (self.momentum * self.m_biases + (1 - self.momentum) * (self.biases - self.momentum * self.m_biases)) * -2 * loss\n\n        else :\n\n            self.m_weights = self.momentum * self.m_weights + (1 - self.momentum) * -2 * loss\n            self.m_biases = self.momentum * self.m_biases + (1 - self.momentum) * -2 * loss\n\n        if self.use_ema:\n\n            self.m_weights = self.ema_momentum * self.m_weights + (1 - self.ema_momentum) * -2 * loss\n            self.m_biases = self.ema_momentum * self.m_biases + (1 - self.ema_momentum) * -2 * loss\n    def minimize(self):                \n        if self.clip_norm != None:\n            \n            weights = np.clip(weights , weights , self.clip_norm)\n            biases = np.clip(biases , biases , self.clip_norm)\n        \n        if self.clip_value != None:\n            \n            self.m_weights = np.clip(self.m_weights , self.m_weights , self.clip_value)\n            self.m_biases = np.clip(self.m_biases , self.m_biases , self.clip_value)\n        \n        weights -= (self.m_weights + self.weight_decay * self.m_weights) * self.learning_rate\n        biases -= (self.m_biases + self.weight_decay * self.m_biases) * self.learning_rate\n\n        # return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:12:03.793990Z","iopub.execute_input":"2023-06-24T04:12:03.794367Z","iopub.status.idle":"2023-06-24T04:12:03.810014Z","shell.execute_reply.started":"2023-06-24T04:12:03.794336Z","shell.execute_reply":"2023-06-24T04:12:03.808473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 2.3 | SGD Final Source Code","metadata":{}},{"cell_type":"code","source":"class SGD:\n    \n    def __init__(self , \n                 X , y , \n                 learning_rate = 0.01 , momentum = 0 , \n                 nestrov = False , weight_decay = 0 , \n                 clip_norm = None , clip_value = None , \n                 use_ema = False , ema_momentum = 0):\n        \n        self.X = X\n        self.y = y\n        self.learning_rate = learning_rate\n        self.momentum = momentum\n        self.nestrov = nestrov\n        self.weight_decay = weight_decay\n        self.clip_norm = clip_norm\n        self.clip_value = clip_value\n        self.use_ema = use_ema\n        self.ema_momentum = ema_momentum\n    \n    def build(self , weights = None , biases = None):\n        \n        if weights == None:\n            \n            self.weights = abs(np.random.randn(self.X.shape[0]))\n        \n        else :\n            \n            if weights.shape[0] == self.X.shape[1]:\n                \n                self.weights = weights\n            \n            else :\n                \n                self.weights = abs(np.random.randn(self.X.shape[1]))\n                \n                raise UserWarning(\"Values do not match with the `X` , intializing random Values\")\n                \n        \n        if biases == None:\n            \n            self.biases = abs(np.random.randn(1))\n        \n        else :\n            \n            if biases.shape[0] == 1:\n                \n                self.biases = biases\n            \n            else :\n                \n                self.biases = abs(np.random.randn(1))\n                \n                raise UserWarning(\"Values do not match with the `X` , intializing random Values\")\n    \n        self.m_weights = 0\n        self.m_biases = 0\n    \n        self.losses = []\n\n    def comput_gradients(self , weights , biases):\n        \n            pred = np.sum((weights * self.X).T) + biases\n            \n            loss = np.sum((pred - self.y) ** 2)\n            \n            yield loss\n            \n            self.losses.append(loss)\n    def update_step(self , loss):    \n        \n        if self.nestrov :\n\n            self.m_weights = (self.momentum * self.m_weights + (1 - self.momentum) * (self.weights - self.momentum * self.m_weights)) * -2 * loss \n            self.m_biases = (self.momentum * self.m_biases + (1 - self.momentum) * (self.biases - self.momentum * self.m_biases)) * -2 * loss\n\n        else :\n\n            self.m_weights = self.momentum * self.m_weights + (1 - self.momentum) * -2 * loss\n            self.m_biases = self.momentum * self.m_biases + (1 - self.momentum) * -2 * loss\n        \n        if self.use_ema:\n            \n            self.m_weights = self.ema_momentum * self.m_weights + (1 - self.ema_momentum) * -2 * loss\n            self.m_biases = self.ema_momentum * self.m_weights + (1 - self.ema_momentum) * -2 * loss\n    \n    def minimize(self):                \n        if self.clip_norm != None:\n            \n            weights = np.clip(weights , weights , self.clip_norm)\n            biases = np.clip(biases , biases , self.clip_norm)\n        \n        if self.clip_value != None:\n            \n            self.m_weights = np.clip(self.m_weights , self.m_weights , self.clip_value)\n            self.m_biases = np.clip(self.m_biases , self.m_biases , self.clip_value)\n        \n        weights -= (self.m_weights + self.weight_decay * self.m_weights) * self.learning_rate\n        biases -= (self.m_biases + self.weight_decay * self.m_biases) * self.learning_rate\n\n        return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:12:04.768227Z","iopub.execute_input":"2023-06-24T04:12:04.768606Z","iopub.status.idle":"2023-06-24T04:12:04.786646Z","shell.execute_reply.started":"2023-06-24T04:12:04.768574Z","shell.execute_reply":"2023-06-24T04:12:04.784779Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 3 | RMS Prop\nSo what is this `RMSProp` and why the hell do we need this thing \n\nSo actually there were some problems with `SGD`. I just searched on ChatGPT and found that \n* SGD was applying a fixed learning rate in every situation, and RMSProp adapts the learning rate for the situation\n* SGD is not really good with sparse inputs, and RMSProp works good with sparse inputs too\nWe know a little bit about `SGD` before the basic vanilla formula for SGD is \n$$p_{n} = p_{n-1} - \\frac {dLoss}{dp}\\alpha$$\nThe formula for `RMSProp` is simple as hell\n****\n$$v_t = \\beta v_{t-1} + (1 - \\beta)\\frac{dLoss}{dp}$$\n$$u = \\frac {n}{\\sqrt{v_t + e}}\\frac{dLoss}{dp}$$\n$$p_n = p_{n-1} - u$$\n****\nWe know everything in this fomrula except of this one guy $e$, what is this $e$ doing here. Lets assume at some point the weight of one feature has became purely $0$. One way to think about this is as that feature doesnt contribute litrally anything to target values, then as we are dividing by `weight`  which is $0$, we will get error, as division by $0$ is not possible, thats why we add a terms $e$ in the weights. the $e$ is so small that when the weights have some value, it doesnt really make any sense, and if the values are $0$. it prevents from $0$ division\nRemeber this code...?\n\nWe can modify this a little bit and we can find the code for `RMSProp`","metadata":{}},{"cell_type":"code","source":"def SGD(X , y , learning_rate = 0.01 , momentum = 0):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n\n    m_weights = [0] * (100 + 1)\n    m_biases = [0] * (100 + 1)\n\n    predic = []\n    losses = []\n\n    for epochs in range(100):\n\n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum(pred - y)\n        losses.append(loss)\n\n        m_weights[epochs + 1] = momentum * m_weights[epochs] + (1 - momentum) * (-2 * loss)\n        m_biases[epochs + 1] = momentum * m_biases[epochs] + (1 - momentum) * (-2 * loss)\n        \n        weights -= m_weights[epochs + 1] * 0.01\n        biases -= m_biases[epochs + 1] * 0.01\n\n    return weights , biases , losses","metadata":{"_kg_hide-input":true,"jupyter":{"source_hidden":true}},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def rms_prop(X , y):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n\n    m_weights = 0\n    m_biases = 0\n\n    losses = []\n\n    for epochs in range(100):\n\n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum(pred - y)\n        losses.append(loss)\n\n        l_weights = 0 * l_weights + (1 - 0) * (-2 * loss)\n        l_biases = 0 * l_biases + (1 - 0) * (-2 * loss)\n        \n        weights -= 1/np.sqrt(l_weights + 1e-6) * 0.01\n        biases -= 1/np.sqrt(l_biases + 1e-6) * 0.01\n\n    return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:12:49.609098Z","iopub.execute_input":"2023-06-24T04:12:49.609497Z","iopub.status.idle":"2023-06-24T04:12:49.617354Z","shell.execute_reply.started":"2023-06-24T04:12:49.609467Z","shell.execute_reply":"2023-06-24T04:12:49.615992Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"And now we have our code for the `RMSProp` \n\n## 3.1 | Functionalities\nNow lets add some functionalieties to our code as we did before to make it more usable\n\n|Name|Atrribute|Info|Input Type|Default|Check Point\n|---|---|---|---|---|---\n|$Learning$ $Rate$|`learning_rate`|either a floating point value, or a **[tf](https://www.tensorflow.org/)=>[keras](https://keras.io/)=>[optimizers](https://www.tensorflow.org/api_docs/python/tf/keras/optimizers)=>[schedules](https://www.tensorflow.org/api_docs/python/tf/keras/optimizers/schedules)=>[LearningRateSchedule](https://www.tensorflow.org/api_docs/python/tf/keras/optimizers/schedules/LearningRateSchedule)** instance.|`float`|$0.001$|✅\n|$Rho$|`rho`|Discounting factor for the old gradients|`Float`|$0.9$ |✅\n|$Momentum$|`momentum`|The optimizer tracks the momentum value, with a decay rate equals to $1$ - momentum.|`Float`|$0.0$|✅\n|$Weight$ $Decay$|`weight_decay`|If set, weight decay is applied|`Float`|`None` |✅\n|$Epsilon$|`epsilon`|A small constant for numerical stability. This epsilon is \"epsilon hat\" in the Kingma and Ba paper (in the formula just before Section 2.1), not the epsilon in Algorithm 1 of the paper|`Float`|$1e-7$|✅\n|$Clip-Norm$|`clipnorm`|If set, the gradient of each weight is individually clipped so that its norm is no higher than this value|`Float`|$0.0$ |✅\n## 3.1.1 | Learning Rate\n\nIt is the learning rate with how the model learns, we had a discussion about this when we were creating the `SGD`. Just scroll a little up and you will find that ","metadata":{}},{"cell_type":"code","source":"def rms_prop(X , y , learning_rate = 0.01):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n\n    m_weights = 0\n    m_biases = 0\n    predic = []\n\n    for epochs in range(100):\n\n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum(pred - y)\n        losses.append(loss)\n\n        l_weights = 0 * l_weights + (1 - 0) * (-2 * loss)\n        l_biases = 0 * l_biases + (1 - 0) * (-2 * loss)\n        \n        weights -= 1/np.sqrt(l_weights + 1e-6) * learning_rate\n        biases -= 1/np.sqrt(l_biases + 1e-6) * learning_rate\n\n    return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:21:11.344473Z","iopub.execute_input":"2023-06-24T04:21:11.344843Z","iopub.status.idle":"2023-06-24T04:21:11.352720Z","shell.execute_reply.started":"2023-06-24T04:21:11.344813Z","shell.execute_reply":"2023-06-24T04:21:11.351935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.1.2 | Rho\nThis is the $\\beta$ we saw in the formula, we had made our algortihtm such that the $\\beta$ was $0$, now we will replace that $0$ with a vraible, that the user can change","metadata":{}},{"cell_type":"code","source":"def rms_prop(X , y , learning_rate = 0.01 , rho = 0.9):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n\n    m_weights = 0\n    m_biases = 0\n\n    predic = []\n\n    for epochs in range(100):\n\n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum(pred - y)\n        losses.append(loss)\n\n        l_weights = rho * l_weights + (1 - rho) * (-2 * loss)\n        l_biases = rho * l_biases + (1 - rho) * (-2 * loss)\n        \n        weights -= 1/np.sqrt(l_weights + 1e-6) * learning_rate\n        biases -= 1/np.sqrt(l_biases + 1e-6) * learning_rate\n\n    return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:21:37.569923Z","iopub.execute_input":"2023-06-24T04:21:37.570926Z","iopub.status.idle":"2023-06-24T04:21:37.578985Z","shell.execute_reply.started":"2023-06-24T04:21:37.570889Z","shell.execute_reply":"2023-06-24T04:21:37.577201Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.1.3 | Momentum\n\nMomentum allows the optimizer to build up velocity in directions with consistent gradients, leading to faster convergence. It can also help to reduce oscillations in the optimization process by smoothing out variations in the gradient updates. The combination of RMSprop and momentum can lead to more efficient and effective training of deep neural networks.\n\n$$m = \\gamma m + (1 - \\gamma)\\frac {dLoss}{dp}$$","metadata":{}},{"cell_type":"code","source":"def rms_prop(X , y , learning_rate = 0.01 , rho = 0.9 , momentum = 0):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n\n    m_weights = 0\n    m_biases = 0\n\n    u_weights = 0\n    u_biases = 0\n\n    predic = []\n\n    for epochs in range(100):\n\n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum(pred - y)\n        losses.append(loss)\n\n        u_weights = rho * l_weights + (1 - rho) * (-2 * loss)\n        u_biases = rho * l_biases + (1 - rho) * (-2 * loss)\n\n        m_weights = momentum * m_weights + (1 - momentum) * loss\n        m_biases = momentum * m_biases + (1 - momentum) * loss\n        \n        weights -= (1/np.sqrt(u_weights + 1e-6) * 1 / np.sqrt(m_weights + 1e-6)) * learning_rate\n        biases -= (1/np.sqrt(u_biases + 1e-6) * 1 / np.sqrt(m_boases + 1e-6)) * learning_rate\n\n    return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:21:56.483782Z","iopub.execute_input":"2023-06-24T04:21:56.484185Z","iopub.status.idle":"2023-06-24T04:21:56.492732Z","shell.execute_reply.started":"2023-06-24T04:21:56.484155Z","shell.execute_reply":"2023-06-24T04:21:56.491381Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.1.4 | Epsilon\n\nIn RMSprop, epsilon is a small constant used to prevent division by zero and improve numerical stability during the calculation of the adaptive learning rate. It is added to the denominator of the weight update equation to ensure that the divisor is always positive. Epsilon is typically set to a small value, such as 1e-8, and has a minimal impact on the optimization process.","metadata":{}},{"cell_type":"code","source":"def rms_prop(X , y , learning_rate = 0.01 , rho = 0.9 , momentum = 0 , epsilon = 1e-7):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n\n    m_weights = 0\n    m_biases = 0\n\n    u_weights = 0\n    u_biases = 0\n\n    predic = []\n\n    for epochs in range(100):\n\n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum(pred - y)\n        losses.append(loss)\n\n        u_weights = rho * l_weights + (1 - rho) * (-2 * loss)\n        u_biases = rho * l_biases + (1 - rho) * (-2 * loss)\n\n        m_weights = momentum * m_weights + (1 - momentum) * loss\n        m_biases = momentum * m_biases + (1 - momentum) * loss\n        \n        weights -= (1/np.sqrt(u_weights + epsilon) * 1 / np.sqrt(m_weights + epsilon)) * learning_rate\n        biases -= (1/np.sqrt(u_biases + epsilon) * 1 / np.sqrt(m_boases + epsilon)) * learning_rate\n\n    return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:22:25.034196Z","iopub.execute_input":"2023-06-24T04:22:25.034618Z","iopub.status.idle":"2023-06-24T04:22:25.044316Z","shell.execute_reply.started":"2023-06-24T04:22:25.034588Z","shell.execute_reply":"2023-06-24T04:22:25.042930Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 3.1.5 | Clipnorm\nThe `clipnorm` parameter in `RMSprop` is used to prevent the gradient from becoming too large during training, which can cause instability or divergence. When clipnorm is set, the gradient is clipped to a maximum norm value, effectively scaling the gradient if its norm exceeds the specified value. This helps to ensure that the updates to the weights are not too large and that the optimization process remains stable.","metadata":{}},{"cell_type":"code","source":"def rms_prop(X , y , learning_rate = 0.01 , rho = 0.9 , momentum = 0 , epsilon = 1e-7 , clip_norm = None):\n    \n    weights = abs(np.random.randn(X.shape[1]))\n    biases = abs(np.random.randn(1))\n\n    m_weights = 0\n    m_biases = 0\n\n    u_weights = 0\n    u_biases = 0\n\n    predic = []\n\n    for epochs in range(100):\n\n        pred = np.sum((weights * X).T) + biases\n        \n        loss = np.sum(pred - y)\n        losses.append(loss)\n\n        u_weights = rho * l_weights + (1 - rho) * (-2 * loss)\n        u_biases = rho * l_biases + (1 - rho) * (-2 * loss)\n\n        m_weights = momentum * m_weights + (1 - momentum) * loss\n        m_biases = momentum * m_biases + (1 - momentum) * loss\n\n        if clip_norm != None:\n            \n            weights = np.clip(weights , weights , clip_norm)\n            biases = np.clip(biases , biases , clip_norm)\n            \n        weights -= (1/np.sqrt(u_weights + epsilon) * 1 / np.sqrt(m_weights + epsilon)) * learning_rate\n        biases -= (1/np.sqrt(u_biases + epsilon) * 1 / np.sqrt(m_boases + epsilon)) * learning_rate\n\n    return weights , biases , losses","metadata":{"execution":{"iopub.status.busy":"2023-06-24T04:22:45.868816Z","iopub.execute_input":"2023-06-24T04:22:45.869241Z","iopub.status.idle":"2023-06-24T04:22:45.877590Z","shell.execute_reply.started":"2023-06-24T04:22:45.869213Z","shell.execute_reply":"2023-06-24T04:22:45.876653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**THIS IS NOT THE FULL IMPLEMENTATION, IT STILL LACKS MANY FUNCTIONALITIES AND IS VULENRABLE TO MANY EDGE CASES, WE WILL IMPROVE THIS IN THE UPCOMING VERSIONS**\n\n**PLEASE COMMENT DOWN IF I DID ANY MISTAKES, OR IF CAN MAKE THIS MORE CONNECTED TO THE GROUND, OR SUGGESTIONS. YOUR ASSISTS ARE HIGHLY APPRECIABLE**\n\n**THATS IT FOR TODAY GUYS**\n\n**HOPE YOU UNDERSTOOD AND LIKED MY WORK**\n\n**DONT FORGET TO MAKE AN UPVOTE $:)$**\n\n<img src = \"https://i.imgflip.com/19aadg.jpg\">\n\n**PEACE OUT !!!**","metadata":{}}]}