{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":4104,"databundleVersionId":46661,"sourceType":"competition"},{"sourceId":265751,"sourceType":"datasetVersion","datasetId":110097},{"sourceId":2269470,"sourceType":"datasetVersion","datasetId":1366461}],"dockerImageVersionId":30162,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Diabetic Retinopathy \n\nDiabetic retinopathy (DR), also known as diabetic eye disease, is a medical condition in which damage occurs to the retina due to diabetes mellitus. It is a leading cause of blindness. Diabetic retinopathy affects up to 80 percent of those who have had diabetes for 20 years or more. Diabetic retinopathy often has no early warning signs. **Retinal (fundus) photography with manual interpretation is a widely accepted screening tool for diabetic retinopathy**, with performance that can exceed that of in-person dilated eye examinations. \n\nThe below figure shows an example of a healthy patient and a patient with diabetic retinopathy as viewed by fundus photography ([source](https://www.biorxiv.org/content/biorxiv/early/2018/06/19/225508.full.pdf)):\n\n![image.png](data:image/jpeg;base64,/9j/4AAQSkZJRgABAQAAAQABAAD/2wCEAAkGBxITEhUUExIVFRUXGBcXFhgYFxcYGRoaGRoXFhcYGRUYHiggGRolGxUWITElJSkrLi4uFx8zODMtNygtLisBCgoKDg0OGxAQGy8lICYtLTU1Ly0tLS0vMDUtLS0tLS0vLS0tLS0vLS8tLS0tLy8tLS01NS0tLS0tLS0tNS0tLf/AABEIAKUBMgMBIgACEQEDEQH/xAAcAAABBQEBAQAAAAAAAAAAAAAAAgMEBQYBBwj/xABFEAACAQIDBQUECAMFBwUAAAABAgMAEQQSIQUGMUFREyJhcYEykaGxByNCUnLB0fAUYoIzkrLh8RVDU1RzosIWNGODs//EABoBAQACAwEAAAAAAAAAAAAAAAACAwEEBQb/xAAwEQACAQMDAgMHBAMBAAAAAAAAAQIDBBESITEFQVGx8BMiYXGBkdEyocHhIzNCFP/aAAwDAQACEQMRAD8A9xooooAooooDM764nFp/C/wtyxn762urosUrmNjY5c2UKG5MVqh2PvRjcsAaJiWjzMHil7U5klkLl1sihCioVIuSeIuoPodNYnEogzOyqOpNqN4MpZ4PPBvVtURs5ghLKrPlEE/ey4eHFZQc/E9q0V7e0pNtMtStq7yYwzTQxoyqrw5XWJwwtiIEkUklg6skjG9l0BPjV9id78MpspaQ/wAq6e9rCoEm+33YD/UwHyBrXldUY8yNqFhcS3UH9dvMg7QxWOG0WKtKMOMRBHfigVogzL2PZ6qznL2gfusdRYGqxt7NpSRhzH2Qy4nMiwyZi4wxkjizXJV0cEZvtGwsCLG4bfx1YBsOMp0BEnPoQVqwh30i+3HIvlZvlrWFeUX/ANE5dOuYrOnyKqfenGpmvEMp7VUbsZWKmOaKJWk7wDB1lLX7tshOovbWbt4558LDNInZySRqzpYjKxGq2bUWPWjA7bgl0SVSfunun3GrGr4yUllPJpyhKDxJYZ2iiipEQooooAooooAooooArK764pkfDDtp4YmM3aNCpZriMmO9kb7XAW1Nh4VqqKA8zw29e0IzGssDmRnwwkDROVs0WD7bKykCMhppGIs3sNwsbdj3rx8auvZ9rZ3HaNDKOyHbTKgkBZRLdFSxUiwIJvpf0quM4AuTYDmaA87k3kx6yylo2OVS6RCOSyg4aJ7FgLygSM/RrowHQWO2Mfi5cJhmhYtJJiAjmNZMOHQdqCfrEdokORTcgjUWNiDV1i96cLHp2mc9EBb48PjVbLvsn2YWPmwHyvVErmlHmRtQsq891B+XmUmy95No/VQmMljC5Z3jbOJPr9LiysYjEiHu2fMGBW4BTBvNtARllQSN2AlLPDLlZkwySsqotirM5Zeet9OVWWI36kUXGHBHP6w3t19mpWH34QgFoXAP3SG/SoK8ov8A68yx9NuUs6f3X5KrFb54xO1ZoUUKyqA0ct4r4qHDJ2jXtKZI5GlXJa2S2t71xN7sczpGIdXSYFxBKBdVxBhnQs3sP2MfcIuO1AzcK00e3sHiAEdl4q2WVbd5WDqe9pcMoI6ECrxWB51sQnGazF5NSdOdN4mmvmea7N3m2giPK9pkvmOaGRCFjw0E0mQkgAH65QCPb5/Zrd7AxUk2HjlkAVpBnCi4srEmMG/2shW/jepWLwscqFJEV0bRlYAqeeoPGngKkQO0UUUAUUUUAUUUUAUUUUAUUUUAU3NKFBZiABqSTYDzNN43FLGhdzZVFyf3xNeX7wbblxbkXKwDgnC55FiOJ+A+Na1xcxorfnwN2zsp3MttkuX67l7tjfcsSmGAsNDK3M9EX8z7qzU0zyNmdi7dWN/d09KZWK2lrCnFFcKtczqv3ment7SlQXuL69xQ01pCknjoPnQq3PhypwVTqNjgR2CnxHmaZYsnVl5g8QPA8xUi3SuG/wC/0pqMpnTYjSrHZ28OJgIs3aIOKOT8H4qfeKpmw33WK9NdPceVLikPsvo3wI6j9KshVlB5iyupRhUjiSyviem7B3hhxQ7hyuPajawdfTmPEaVcCvGmVlIeNisi6qw0I/UVv90951xKhHsswFyOAcDiVHXqK7Nreqo9MufM85f9NdFe0p7x/df0aaiuXrtb5yQooooAooooArhNBNefb370O5MOGfKODuOJHMKeQ8eflVNavGlHMjZtrWdxPTD7+Bcbwb4pE3ZwgSy89e4n4iOJ8B8KxmP2nNMbyOza8Bog/p4fOoccAUd0Wrt7X/WuFXup1eePA9Ra2NKgtll+L5/oczcLWvcVztDmsD5n/L0pOUtyt1/Q1x1YW/L9ePrWvk3MIWFF7fmdeFcMRFyhI6qeHp0pMLkknnp5jTh8Kcs1tD+tYyMioZQwvax4EHkRUrBbSnh1ikI/lPeQ/wBH6WNGEkgVHzDvtz4VAVimjG68m5+TfrVqk44lF7lTjGpmLjt8e56BsDe+OZhHKBFKeAJ7j/gY8/A6+daYGvHJoQwsa026e9RQiDEve1lSQ+PAOenj766ltf6npqfc4d90rSnUo8d1+DfUVwGu10zhhRRRQBRRRQBRRRQBXCa7WZ362t2MORTZ5e6PBftH8vWoVKipxcn2LKNJ1ZqEe5lN79ufxEmRT9SjWH854Fj4ch69apl0PgfnTRcAeooYE25C/rb8q81VqOpJykeyoUo0oKEeCQ01tPd1pLO1uHxHypK6X+P+tKUc6oZsIcjm5HQ9DToamreF66E6G3nqKGNhw0CkC9tb+mtdAY8PjagBpADXHjVx8iOI8qcWO3j86bcDy8uNEE/ARFLY5WPe5H7w/WoJJR45FbKVY5SOTA6VJnXMLEnwJHyYVFxcfcK34d5W8RrrVsHuYnhrHievbtbZXExBtA47rr0P6HjVvXkm6G1TDOjm4jkCq48+Deh+BNetLXobWt7WG/KPH3tv7GphcPj8HaKKK2TTCiioW2cesELyt9kaDqToo9SRWG0llmYxcmkjMb9bdKj+HjNiR9Yw+yDwXzPy86w6aAEctT+dSYF7YuzyAMxLG/MnU00MM5UkWAA49fKvO3FSVWWp8dj11pShQhoXPf4nTMBz0NJMjch79P8AOmoxbre1tfy6CljXyrUZvoewaZiQzhONr8OvEHjQ4AJAIPXn6611Jbcr3o7IcRofC1Zclgj33IsuEs4mF8wUhgDoy8SCOZHEevWpq2axGt+FufSkqG56+X6UzC7RuAtwCTl0Fg3ErryOpHqOlZzq5IY0vK7kmU246eFN51Olrg8f8xUqVmaxa2YUxKBzFv3yqL2exKL23J8WFiEeQXJGubp4Xqh2pDfODyAJ8RzqX/ENlsrG3W1/86gxtqbnMDob6n/TWrtSeNiEYyjnLzk9B3C292iiB2u6i6MeLILaE8yt/dWyrwrZWKeJlZD34n7p6/5WJB9a9r2bjFmiSReDqCPDqPQ6V2rKvrjpfK8jzPU7ZUqmqPD8yVRRRW8c0KKKKAKKKKA4a8k322l2mLcC5Edox001b/uJ91esTyZVLHgAT7hevBllLkueLEsfU3PzrndRniCidbpNPM5Tfb+SVGL2vqb+7yp8jUeR/f76UzEKeYmuHI9GmdKUvMBpeo5auFredRSMvJKEgpyNhVcJPGpET1lowmTlFKNNI9KBqOSQvJ1oEfhRnqVglBOtZISlpWREeH0uRVftaIFSemvxGlaAgE25VU7bh4hTcAi46g+NTSK6VTMsMhxyi1rXH70r1PdrGGXDRsfatla/VdPjYH1ryuO4+ySeOugrcfR9iWImRgBlZWFjf2gR/wCNdDp82quPE5/V6SdLUuz/AKNhRRRXcPNhWB+k3aVuygF9frG9O6vxzH0rfV45vxie0x8vRMkY9FBP/czVp30sUseJ0Om09VfL7blYhuNeHT9amO5yBbmwtUKIeNSS1hxtyvXAkepXYWqXvx6+6nImC63193xpqxIPnw9CfdpSQ1r3sfOq+STeR0tc24+puSfKlo+oBHEfpxqEZCG1uDyA0/YrhxRWVoipy2DI/Xk6W8NDf+bwrOM8EHLDSLTXmD+/CuyxKy2N/TQ9QR4ggH0rkE1jc/DjQZCT4fvWoptMzzsJhkJBDWzLoRwv0YeB+Go5V2RDY90f3hTeIXUMASwHD7y8189LjxHiae9sKVIIup1GhHG9T2e5HLWw3FDcqOAIJ58efCkDDakE68j15MflVoAL2sGBs3Ai3RgDxHlXdoRgAAHUjjpobHQdKnjOyK41cyKDB6AW4i4Pv1NehbgY0skkZFgpDL5Nx+I+NeeICLAob6c9Lak69b1qNy8U64pFIAV1ZfaJPDMOPlW1Zz01l8SjqdNToy+G/wBj0miuCu16A8mFFFFAFFFFAV+8D2w05HKKT/Ca8Pwz8K9z23Fmw8y9Y5B71NeC4fgPKuX1FbxO50j9Mvmi1QinA96gpyqReuRJHcQ63GmJDenQ3Go+JQMcg8C34eFv6tR5XrEUJPBGwUjNdiuUFjkHMroAxPibkeFqsIr0gRG9OoLVlvJGLeB8NUhG0piJbU7a2o4dKqZZEcvXYnsa5SawSayiXisUVUsp15eZ0FU22oXOHl75zlGu/EjQ3sKeDFnFjoup6X5Vza09opPwN8QQKuhlSRRJYi0SI5VZbk3sLedqvvo2mJxE4J+wh18GP61h53KvppmF/UaGtv8ARcCZJ2I4JGAfMsfyrds1/mj67Gj1H/TL13PRaKKK7x5gK8N29LfF4gn/AI0nwYj8q9yrwzeWPLjcSP8A5WP97vD51odQXuL5nV6T/sl8v5OYfUgHhcVdbTwKRxAjVidOnX8qzkY08eXSpZnYqASbDhfhXG2SeUd9puSaeB0P3DwDHj4rzH76UxIelOxvccLgc/Dnf4VDjTKzZnJRzdbj2DzUniVY8OnDmKrjHktlLT2Bgb3v/pXZQbZhxU5vP7wHmt/W1L7M08iWotiDeR2Nja4N9Lj1qVCw14nyqBhBYstuFio/lb9DcegqcCfsnUfvWoTXZEovKFtwvb/WoN2VrFbKxup07rnUjwvqR436ipqsSNeV6bmiDKVPA8fLw6Hx5WpF45JOOV8RazcWW4tlUak8eP6+lRJMb9YgkJvLn7P7oKAadcxGvoaQXJ+r4NfW3Mff8Li/qCK5jswOYfZykW/l1+IzD1q6K3w/XgVPaOV68SVh5QWOb1vxU+XTXjUvYs1sdh7cO0t7waocZPe0g4X49VOg/I1a7m3bGQC17PfysrGrqK9+PzRRdv8Axy+T8j2QV2uCu16M8iFFFFAFFFFAJcXFeB4zDGKWSI/Ydl9x0+Fq99NeU/SZswx4kTD2Zl1/GgsfeuU+hrRvoZgpeB1OlVdNVxfdeRmYwalR2NRIn0p+9u9r0PlXEkj0S3HZ5Aouwvy08dALdb6UQwiw1GZu8x8fDwAsBUcuWa4tlU2Hi3An04ed6eYWN7kdeFvOotYWCSWdyUFFDGwNhf8Ad7DxplFPU0qJjwIvUTOk7s2ZnijZlszKGZehI1FTD5VAwbkRpofZHTp51J7Runx/SkuWZgvdWRwMbcKTMWsLaEn9mmJcUV0IuTwA/elP4drDMzAt8AKwkWDqRhQAPWq3ahzJJ0CsT520qZJIW0XhzP6VF2gAInAH2W+RqUX7yIyjiLGNpRnRhxX5c69D+jHDWw7ycpH7v4VAHzLVg5QWIAFySAo6k6AV7BsHZ4w+HihH2FAJ6n7R9Teun06DlNyfb+TkdYqKNJQ7t+RYUUUV2TzQGvIPpGwpTHM3KVEf1HcP+Ee+vX6xP0obMz4dZgNYm1/A2h9xyn31rXcNVJ/A3en1NFdZ77evqebxj1qRGeRqLC1SOPW/EVwJI9SsEmMkDTlw/Slo0ZBuAQQQRyueAI6XqOJLi4+R+VdeIgA3GouCOB8P8jVTRPCYQDIcp1+6eoHI/wAwHvGvI1JU2qKyZhqT100II4eRFEEjAlWsSLcNLjky+B6ciLdCZNZ3MaexIlAXK9yMp72n2T7VvKwP9NSwdDpqT16VFZ/A+tj6caRhZGtltqump5cVPu08wai1lGVHDJakjlpRc20FqYkmZRc2tXcPIXN2ICjgvP1rCRaAhIAfjJ8xzXw8PECkYhg1ipve3uHy41cWR4mK2zCqML2bdQ51PRzy8A3z/FVjWEimLy2yOIB2Zj+7oPLivwIHoa0n0YQF8Sz2/s0Ib8RNl+Gas9O9nB5MMp8xdl+Gb4V6T9HWy+ygaQ6NM2f+kDKn5n1rcsoudVP6+vqaXUpqnbyXd7evoawUUUV3jygUUUUAUUUUAVTb17HGKw7x/a9qM9HXh6HUHzq5oNYlFSWGShNwkpLlHz8LqxVgQwJUg8QRoQfWlSSnRV0Y8+g5t6fMit39I27PHFxDgCZ16qP94PEAa+Avyrz/AAhv3jxPAdByHnzPifCuDWoulLc9TbXEa0E4/X4E1O6LAadOlKje+h5UkGuGx5XrVaN1eA8j208dKdR9fdUPIOGvvpcYYcx63qLRPA/gW+rXyFcMxYlUPm3IeXjUDBSM6AXso004nU1OimCjKBcjkPz6VmSw2YhwhSYcXsLn7xPE+tKkjUA6VxTbnqeJomewB8dag+S5McMltKibRf6t/wALfKlO16VDgJJxIijRUZnbkqgHW/U8hU6cHKSSKq1SMINvY1O4WyDJN27DuRkhfF+v9IPvPhXo4qPs7CJFGqILKo0HzJ8TUmvR29FUoKJ468uXcVXPt2+QUUUVeaoU1iYFdWRhdWBUjqDoadooDwnbezWws7wtewPcP3kPsn3aHxBqfj9nBIkkU+1a4Neh767tjFxd2wmS5jbr1QnobDyOteQQyzd5Jrhkd1yk6qAbAEcjXFuKCpt7bdj0tpde2Ud91yTATy9fLjXVlsbEaHr+tWcCwNhzmv2o9kCqyNGYHu3txrSlDB0Yz1J9sCyQCenh86XIwIGgzAkq1z7iOangR+YFRgnmPWlIhGot43v+VV4LHHKwyVBMGvbQ/aHRuYvzHQ9DTbOBKnAZu6RxsD7JPgGt/eNQppJM3ctmGhtzHTz6Hr5mn4WSxvchhY/ePIi3Ig+4ipKKW5jDe2dy7xGwwiBzJmkJsOg8hUEwKOVMYfFOy99j3TYjx+95kEH1pXbXzAHW1YqYb2WCVFTx7zyLjksvhSZ+8pBFwRY+tNmS49KsdlY0KBGsRklZrKBre/y0/WkIZfJmrPTFtLJC2Vs2TEv2I1ZBnY9VU3B8CxGXzJr2fBMpRSnslQV8rC3wqi2ZsE4ePOtjPcu1tA1+MWv2bWt4gHrVhsKdSrIvBTdfwPdlFuVjmW3LJXoLSh7KG/LPJdQu/wD0VNuFx+S0oooraNAKKKKAKKKKAKKzu098sNBM0D586mAWCg37Ziq5ddbWu3QVKXeXCtos6X74sc2hQXbMLXAsQdbXvpQD2JPayCP7CWaXxOhSP/yPgFH2qx+925Ju02FW5Ny8XDXm0fifu+7pV8m8OEw4CPMD3ZJHksSt17IuWI4aTxkDgF8BWiVgRcG4OoNV1aUakcSLqFedGWqB8/5jcjgRoQRYg8wQeBpwSWr1/eDdXD4rvMuSTlIlg3ryYedeebb3JxsP9mgxC9UsGHiYyfkTXJq2c48bo9Db9RpVNpPD+P5KGSYDiQKYbGFhYcOF7anwApM0YiNpFIfo6kH3MKVBbieJrVccHQUsisECFAvYa+ftHif0qYrAW0qLhSMvHm3+I080ijiR4VGS3EJPCHsxodtKlbP2Ripz9VA5H3iMi/3m/K9bLY30fKLNiXzn/hron9TcW+Aq2la1KnCKK9/Sord7+C3Zkt39hzYsjs9EBs8hHdHgPvNbkPWvSpdlR4bBTRxjTs3JJ1ZjlOrHrVzBCqKFVQqjQACwA8AKi7f/APbT/wDSf/Ca7Fvaxpb8vxPO3l9O4eOI+H5JqcB5UqkpwHlSq2TRCiiigCiiigOGsXtndRcUryJZJxJKAx4MAxsr2+B4jxra1B2T7L/9WX/G1QnCM1pkWU6kqctUXueJ4yGSFzHKpRxxU9OoI0ZfEaUiLEMoNiRfQ+Ne27X2NBiUyTRhhyPBlPVWGqnyrzzbu4E8QZ8O4lUXOVyEcAa+17Letq5daylHeG6O/bdTpzWKmz/b18zJtLzvUrZW8CPHLFbIFFlkIvnOt8o5jhrzvoNKqcRgnXvYiN0U6gMpAbpqdMvz8uPYnDG+lhwHLztWsl7POVubzxUSw9iZsmcxuHtfW+vXwHIcal7ZdRKZVsQ1i4HuuOpAGvUeIqEtutKzAcxaq9WFgscU5as7jYezAg6Pp1F/snw0LD1WpSNYn41HhwcjjLFG7IxABCmyMSMvePdCliCLnQ+B02O7248kyh8RIIxchkTV8ymzKW4LqDwvU4286n6UV1L2lRypvBhX2oolaIggjUnllCGR29LDT+YV6HurtHAYVQzO7SMcjyGNgqNqxiv9khUZyde6ua9rVrsPuxg0XIMNERct3lDEkoYyxZtSShKk9DanBu7hP+Wi/szFqgPca4ZDfiCGYf1Hqa61C0hTxLueeu+oVK2Yr9Prkr8RvphFtcyHNF2q5Yycy8rAai471zYW1vTGA21hu0eeNpOzbs0b6s5DJK8YUBx9q8q3HDvsetriXd/CM2ZsPEWyhL5RfKBYAHlYaCkru1gwCBhoQGUIbIouotZf+1f7o6Cts55X4ffTDtL2feGaQRo1tGJUHUcVGY5fS/CkYXffDuFcLKUdkWMiNiTnjEqsUtcKUOa+tlBJtVnHu3g1ZXXDQhltlIRQRa1tfCwt0pM27GCcKGwsJCjKAY1sFCrHlt93IirbooFAVZ36w5yFVfI18zsLKmWQRMGtc3ubi2h61JO+OGDKjdqrFlUgxtdMzQqpf7oJxENvx+BtOXdzCDLbDRdy5XuDQsQWI8yoPpSotgYVQAMPHYajujSzI49zRxn+hegoCyorlqKAp8fuvhZpe1kjzSad7Mw4BRpY6aItMYTc7CR+wri9w31snfBAWz97viwGh8TzNaCigM3NuPgnUKyObZtTLJchxGrAnNqCsMY14ZdNa0MMQVQo4AADyAsKXRQBXK7WT3oxuNTEIIM3ZBUaSyBhrKEYkZCzgJclVZSOOtAaXEYRJBZ0Vx0ZQ3zqql3QwLccLF6Ll+VqzcG9W0JCt8KYlIk0COz5lfC2W5UqCFlmB4huyYi1rB8b240LEWwRDO8VwEmICSJEzXJXR0MjX0P9mdBraLinyicako/pbRP2TujgSrXwyG0ko1zHQOwHE1dYTYuGj/s4I18Qi399qzMu3Magwf1ZcywkyjsnB7UvCosVUiNgryNZrA5T5iLhN7scI1D4U5hHDncxTjIWEGaV0VNVYySWRLleyOa3eyFCK4Rl1ZvZyf3N+BUHbm0Rh8PLMVzdmpa17A26tyHU8hc1n9k7fxk2JhR8O0CFM0ilHJuYo3BMhUKozs6WvmvGbgcBrWUEWIuDxFSKzKjfERl0mjBdHKsYGV0sEhkZ7uVOgnW6gE6aA3FMYbfCPE5YHhdTPCrlQyXRHJjZmdmAOpSwW7d7hoa0P+wsNeMiCMdlm7MBAFUsVYkIBYNdFN7XFqe/2XB3fqYu62Zfq17ranMNNGuza/zHrQFHtTfWDDyPG0cpKMIwVCENJliYILtcG0yG5AHHWupvvAXVezmGcxLGxVQGklWJ1iAzZg4SUMbgABXN9Kvp8BE4IeJGDXzBkU3uADe410UD0FKGEj07i6EEd0aEDKCPELpfppQFdu9vDHjA7RK4CEC7i2YEXVlsToehsRzAq4pnD4VI82RFTMxZsqhbseLG3EnrT1AFFFFAFQdkew3/AFJf/wBGqdVZgJlSFmYhVEkpJ/8Asf4+FAWEsqqCzEAAXJJsAOpNVyxmchnBWLiqEWL9GkHJeYX1PQdihaUh5QVUEGOM9eTyfzdF4Lx48LICgEvECLEAjoRcfGqrEbr4JzdsLET+AA/C1XFcNYcU+SUZyj+l4M6N0Nnf8vHrw7zeWmvWpuG3dwiG6YaIH8APzrDbP+jnFwgdnio1ZUEcbAMezGeOZrAjX63tj5MvSrufdvGNFHaUqyJl7MYvE5WvJmZWxFu0N00z2uOQqKpwXZE3WqPZyf3ZrmgUqVKjKQQVtoQdCLdKr9mYWSKR19qJgGVr3NxZcrA6k5QuvPLrre+Xm2LjYi0mI2giRZcOjsZZI75JcOXJubRsyJOl1IzdoL2peB3d2ijF/wCM7UWhyAyyZWCFLq3dNtFfvi5btO8NL1MqN1ei9YfBbv46OSDPiJJMzr/EMJGKCKOOMgBWIs7TIdQDdZHueFNYPdrGt3Wx1xHKcwWeYvZpMJIyyOLd4wpiABYAduv4qA3t6L1jt39k46GQSPOuIRo4UN5XOoEatIlwFAADt9osWGo1pnau7GOlknIxV45G0jMsiKUIYAHIt48lwe6bPYhqA296L1g8TuljyzhcYUQrZAssigWgaNBlUdwJLkbunvC9xe1TYd38as+HYYm8UUkrMDLMWZXaQhGDXDizLa+oK6G1gANhRRRQBRRRQBRRRQBRRRQBXK7RQFPt3EYpGi/h41cO+SS/+7BsRKdRdVCuCOJLLwsaz2C2xtYtCZMMoDSlZFCNdVsl+8SBlUl+99oLprodzRQGAj2ttd1CnDhbiUFwjLche7lBJKWubFhqRbUam83axmMaWWPERkIqx9m+Ui5IswJJ7zaXJAtrx5DR0UBy1doooAooooAooooAooooCFtbaKwR52DNqqqqi7M7sERVGguWI1JAHEkCoE+9eGjQtMWiYRtI0Tqe0AXOSCq3BJEbkAE5gpIuBerLaWAjnQpILrdToSpDKQysrLqrBgCCOBFVUm6OEb2kc90qbyyHNcSLna7avaaQZjr3vAWAdw+9OEdiolAYMVswZTddeY4cgeZBA1FQMFtDChO3kxMZjWSUx6lUU2eck5uL9mS1/u8ONzUDFbJu0paY99xIT2+W8VizuvAIhfPmOgLE05iMfsox9m0cjxFhiGOSUi0EcZjkt7RRkiUAgENYg8TQGm/9R4TNl/iI81mPtadzPmu3AW7OQ2vwRjyNL2ftyGZ8kTZu5nuAQLZihBvqGBHAiqnZm7Wz5YYnjjYxdnkRWaTKVyyR3ZCdXyyyLmOve48Kttm7DhgYugbOQQzM7OzXOa7Fjqb8+gAoCzooooAooooCj3v2G2Mg7ESdn3lY3BIOW9gcrKws2VhZhqgBuCQYGI3Wlf8AhC2KJOHWzdzIrkFCGCIwCGyZSOGV2HA2rV0UBh03JnWNYxjWFi2Z8r5yGkhlGU9poV7IoCb2QgcjdqHcKRRGBiVXLOs75YmF8q4dCATISMywOG1N+2bpY72igPPcF9HciQvEcVcNHHGtlkUAIYzYgSar9WbDgO0fiCQdxszDtHFHGzZ2RFUta2YqAC1uV7XtUqigCiiigCiiigCiiigCiiigCiiigCiiigCiiigCiiigCiiigCiiigCiiigCiiigCiiigKKXdLBs4cxDQsxFzlYtl9oHiBlsBw1OmtPru3hALCBRqx0uPaXKQCDouU2A4DlaiigJ2AwMcKCOJAiC9gOpJZj4kkkknUkk1IoooAooooAooooAooooAooooAooooAooooAooooAooooD//2Q==)\n\nAn automated tool for grading severity of diabetic retinopathy would be very useful for accerelating detection and treatment. Recently, there have been a number of attempts to utilize deep learning to diagnose DR and automatically grade diabetic retinopathy. This includes this [competition](https://kaggle.com/c/diabetic-retinopathy-detection) and [work by Google](https://ai.googleblog.com/2016/11/deep-learning-for-detection-of-diabetic.html). Even one deep-learning based system is [FDA approved](https://www.fda.gov/NewsEvents/Newsroom/PressAnnouncements/ucm604357.htm). \n\nClearly, this dataset and deep learning problem is quite well-characterized. \n\n# A look at the data:\n\nData description from the competition:\n\n>You are provided with a large set of high-resolution retina images taken under a variety of imaging conditions. A left and right field is provided for every subject. >Images are labeled with a subject id as well as either left or right (e.g. 1_left.jpeg is the left eye of patient id 1).\n>\n>A clinician has rated the presence of diabetic retinopathy in each image on a scale of 0 to 4, according to the following scale:\n>\n>0 - No DR\n>\n>1 - Mild\n>\n>2 - Moderate\n>\n>3 - Severe\n>\n>4 - Proliferative DR\n>\n>Your task is to create an automated analysis system capable of assigning a score based on this scale.\n\n...\n\n> Like any real-world data set, you will encounter noise in both the images and labels. Images may contain artifacts, be out of focus, underexposed, or overexposed. A major aim of this competition is to develop robust algorithms that can function in the presence of noise and variation.\n\n**A minor problem!**\n\nDue to the large image file size, there are **only 1000 files with labels** and one csv file with the labels of all the images in the directory available in the kernel, as demonstrated below. The actual competition had on the order of 35,000 files so this is clearly **a very small subset of the data**. \nIn addition, this is a highly imbalanced dataset. \n\nOriginally, I had aimed to create a model that would be close to the SOTA, but clearly, with only 1000 images that's not possible.\n\n**AIM OF KERNEL:** I will utilize transfer learning, oversampling, and progressive resizing on this small, imbalanced dataset.\nGiven that many real-world datasets are also small and imbalanced, it will be interesting to see how far these techniques will take us.","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\n# from keras.preprocessing.image import img_to_array\n# from keras.preprocessing.image import array_to_img\n# from sklearn.model_selection import train_test_split\n# from PIL import Image\n# import scipy\n\nimport tensorflow as tf\nfrom tensorflow.keras.applications import *\nfrom tensorflow.keras.optimizers import *\nfrom tensorflow.keras.losses import *\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.models import *\nfrom tensorflow.keras.callbacks import *\nfrom tensorflow.keras.preprocessing.image import *\nfrom tensorflow.keras.utils import *\n# import pydot\nfrom sklearn.metrics import *\nfrom sklearn.model_selection import *\nimport tensorflow.keras.backend as K\n\n# from tqdm import tqdm, tqdm_notebook\n# from colorama import Fore\n# import json\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom glob import glob\nfrom skimage.io import *\n%config Completer.use_jedi = False\n# import time\n# from sklearn.decomposition import PCA\n# from sklearn.svm import LinearSVC\n# from sklearn.linear_model import LogisticRegression\n# from sklearn.metrics import accuracy_score\n# import lightgbm as lgb\n# import xgboost as xgb\n# !pip install livelossplot\n# import livelossplot\n# from livelossplot import PlotLossesKeras\nimport warnings\nwarnings.filterwarnings('ignore')\nprint(\"All modules have been imported\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:10.469147Z","iopub.execute_input":"2024-11-13T08:43:10.469822Z","iopub.status.idle":"2024-11-13T08:43:18.494508Z","shell.execute_reply.started":"2024-11-13T08:43:10.469516Z","shell.execute_reply":"2024-11-13T08:43:18.493387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!unzip ../input/diabetic-retinopathy-detection/trainLabels.csv.zip","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:18.496179Z","iopub.execute_input":"2024-11-13T08:43:18.496438Z","iopub.status.idle":"2024-11-13T08:43:19.593219Z","shell.execute_reply.started":"2024-11-13T08:43:18.496405Z","shell.execute_reply":"2024-11-13T08:43:19.592282Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Importing labels","metadata":{}},{"cell_type":"code","source":"import pandas as pd\n\ntrainLabels = pd.read_csv(\"./trainLabels.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:19.59507Z","iopub.execute_input":"2024-11-13T08:43:19.595714Z","iopub.status.idle":"2024-11-13T08:43:19.631763Z","shell.execute_reply.started":"2024-11-13T08:43:19.595663Z","shell.execute_reply":"2024-11-13T08:43:19.630864Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!apt install p7zip-full -y\n!7z x ../input/diabetic-retinopathy-detection/train.zip.001 \"-i!train/11*.jpeg\" -y # restrict extracted file to about 100 for the disk restriction\n!mkdir data\n!mv train data/train_11\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:19.632945Z","iopub.execute_input":"2024-11-13T08:43:19.633205Z","iopub.status.idle":"2024-11-13T08:43:52.005744Z","shell.execute_reply.started":"2024-11-13T08:43:19.633174Z","shell.execute_reply":"2024-11-13T08:43:52.004383Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\n\nimg = Image.open(\"./data/train_11/1116_right.jpeg\")\n\nimport matplotlib.pyplot as plt\n\nplt.imshow(img)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:52.009359Z","iopub.execute_input":"2024-11-13T08:43:52.009803Z","iopub.status.idle":"2024-11-13T08:43:54.03765Z","shell.execute_reply.started":"2024-11-13T08:43:52.009732Z","shell.execute_reply":"2024-11-13T08:43:54.036548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%reload_ext autoreload\n%autoreload 2\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:54.039202Z","iopub.execute_input":"2024-11-13T08:43:54.039644Z","iopub.status.idle":"2024-11-13T08:43:54.126082Z","shell.execute_reply.started":"2024-11-13T08:43:54.039597Z","shell.execute_reply":"2024-11-13T08:43:54.125237Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Filtering csv file","metadata":{}},{"cell_type":"code","source":"import os\nbase_image_dir = os.path.join('.', 'data/train_11')\ndf = pd.read_csv(os.path.join('./trainLabels.csv'))\ndf['path'] = df['image'].map(lambda x: os.path.join(base_image_dir,'{}.jpeg'.format(x)))\ndf['exists'] = df['path'].map(os.path.exists) #Most of the files do not exist because this is a sample of the original dataset\ndf = df[df['exists']]\ndf = df.drop(columns=['image','exists'])\ndf = df.sample(frac=1).reset_index(drop=True)#shuffle dataframe\ndf['level'] = df['level'].astype(str)\ndf.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:54.127265Z","iopub.execute_input":"2024-11-13T08:43:54.127511Z","iopub.status.idle":"2024-11-13T08:43:54.497707Z","shell.execute_reply.started":"2024-11-13T08:43:54.127467Z","shell.execute_reply":"2024-11-13T08:43:54.496709Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### The dataset is highly imbalanced, with many samples for level 0, and very little for the rest of the levels.\n","metadata":{}},{"cell_type":"code","source":"df['level'].hist(figsize = (10, 5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:54.498904Z","iopub.execute_input":"2024-11-13T08:43:54.49915Z","iopub.status.idle":"2024-11-13T08:43:54.783003Z","shell.execute_reply.started":"2024-11-13T08:43:54.499118Z","shell.execute_reply":"2024-11-13T08:43:54.782021Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def balance_data(class_size,df):\n    train_df = df.groupby(['level']).apply(lambda x: x.sample(class_size, replace = True)).reset_index(drop = True)\n    train_df = train_df.sample(frac=1).reset_index(drop=True)\n    print('New Data Size:', train_df.shape[0], 'Old Size:', df.shape[0])\n    train_df['level'].hist(figsize = (10, 5))\n    return train_df\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:54.784219Z","iopub.execute_input":"2024-11-13T08:43:54.784493Z","iopub.status.idle":"2024-11-13T08:43:54.848935Z","shell.execute_reply.started":"2024-11-13T08:43:54.784448Z","shell.execute_reply":"2024-11-13T08:43:54.847878Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_df, val_df = train_test_split(df,test_size=0.2) # Here we will perform an 80%/20% split of the dataset, with stratification to keep similar distribution in validation set\ntrain_df['level'].hist(figsize = (10, 5))\nlen(val_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:54.850311Z","iopub.execute_input":"2024-11-13T08:43:54.85112Z","iopub.status.idle":"2024-11-13T08:43:55.068669Z","shell.execute_reply.started":"2024-11-13T08:43:54.851056Z","shell.execute_reply":"2024-11-13T08:43:55.067445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = balance_data(train_df.pivot_table(index='level', aggfunc=len).max().max(),train_df) # I will oversample such that all classes have the same number of images as the maximum\ntrain_df['level'].hist(figsize = (10, 5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:55.071759Z","iopub.execute_input":"2024-11-13T08:43:55.07292Z","iopub.status.idle":"2024-11-13T08:43:55.397079Z","shell.execute_reply.started":"2024-11-13T08:43:55.072835Z","shell.execute_reply":"2024-11-13T08:43:55.396106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.concat([train_df,val_df]) #beginning of this dataframe is the oversampled training set, end is the validation set\nlen(df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:55.398528Z","iopub.execute_input":"2024-11-13T08:43:55.39881Z","iopub.status.idle":"2024-11-13T08:43:55.463847Z","shell.execute_reply.started":"2024-11-13T08:43:55.398776Z","shell.execute_reply":"2024-11-13T08:43:55.462801Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n    rescale=1.0/255,\n    horizontal_flip = True,\n    zoom_range=0.2\n)\n\ntest_datagen = ImageDataGenerator(\n    rescale=1.0/255,\n    validation_split = 0.2\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:55.465137Z","iopub.execute_input":"2024-11-13T08:43:55.465429Z","iopub.status.idle":"2024-11-13T08:43:55.532923Z","shell.execute_reply.started":"2024-11-13T08:43:55.465382Z","shell.execute_reply":"2024-11-13T08:43:55.531674Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_train = train_datagen.flow_from_dataframe(\n        train_df,\n        directory=\".\",\n        x_col=\"path\",\n        y_col=\"level\",\n        target_size=(256, 256),\n        batch_size=32,\n        class_mode='categorical')\nx_test = test_datagen.flow_from_dataframe(\n        val_df,\n        x_col=\"path\",\n        y_col=\"level\",\n        directory=\".\",\n        target_size=(256, 256),\n        batch_size=32,\n        class_mode='categorical')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:55.536902Z","iopub.execute_input":"2024-11-13T08:43:55.537418Z","iopub.status.idle":"2024-11-13T08:43:55.644152Z","shell.execute_reply.started":"2024-11-13T08:43:55.537362Z","shell.execute_reply":"2024-11-13T08:43:55.64315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import itertools\ndef plot_confusion_matrix(cm, classes,\n                          normalize=False,\n                          title='Confusion matrix',\n                          cmap=plt.cm.Blues):\n    \"\"\"\n    This function prints and plots the confusion matrix.\n    Normalization can be applied by setting `normalize=True`.\n    \"\"\"\n    plt.figure(figsize = (6,6))\n    plt.imshow(cm, interpolation='nearest', cmap=cmap)\n    plt.title(title)\n    plt.colorbar()\n    tick_marks = np.arange(len(classes))\n    plt.xticks(tick_marks, classes, rotation=90)\n    plt.yticks(tick_marks, classes)\n    if normalize:\n        cm = cm.astype('float') / cm.sum(axis=1)[:, np.newaxis]\n\n    thresh = cm.max() / 2.\n    cm = np.round(cm,2)\n    for i, j in itertools.product(range(cm.shape[0]), range(cm.shape[1])):\n        plt.text(j, i, cm[i, j],\n                 horizontalalignment=\"center\",\n                 color=\"white\" if cm[i, j] > thresh else \"black\")\n    plt.tight_layout()\n    plt.ylabel('True label')\n    plt.xlabel('Predicted label')\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:55.645606Z","iopub.execute_input":"2024-11-13T08:43:55.645936Z","iopub.status.idle":"2024-11-13T08:43:55.715785Z","shell.execute_reply.started":"2024-11-13T08:43:55.645892Z","shell.execute_reply":"2024-11-13T08:43:55.714198Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Image Data Generator","metadata":{}},{"cell_type":"code","source":"train_datagen = ImageDataGenerator(\n        rescale=1./255,\n        shear_range=0.2,\n        zoom_range=0.2,\n        horizontal_flip=True)\ntest_datagen = ImageDataGenerator(rescale=1./255)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:55.717116Z","iopub.execute_input":"2024-11-13T08:43:55.717392Z","iopub.status.idle":"2024-11-13T08:43:55.782288Z","shell.execute_reply.started":"2024-11-13T08:43:55.717359Z","shell.execute_reply":"2024-11-13T08:43:55.781242Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"t_x, t_y = next(x_train)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:43:55.783635Z","iopub.execute_input":"2024-11-13T08:43:55.783961Z","iopub.status.idle":"2024-11-13T08:44:00.516021Z","shell.execute_reply.started":"2024-11-13T08:43:55.783917Z","shell.execute_reply":"2024-11-13T08:44:00.514983Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Showing some pics","metadata":{}},{"cell_type":"code","source":"# plt.figure(figsize=(10, 10))\n# for images, labels in x_train.take(1):\n#   for i in range(6):\n#     ax = plt.subplot(3, 3, i + 1)\n#     plt.imshow(images[i].numpy().astype(\"uint8\"))\n#     plt.title(classnames[labels[i]])\n#     plt.axis(\"off\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:44:00.517455Z","iopub.execute_input":"2024-11-13T08:44:00.517861Z","iopub.status.idle":"2024-11-13T08:44:00.587648Z","shell.execute_reply.started":"2024-11-13T08:44:00.517816Z","shell.execute_reply":"2024-11-13T08:44:00.586401Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## CNN Models","metadata":{}},{"cell_type":"code","source":"def create_CNN_model(model):\n    model.add(Conv2D(32, (3, 3), padding='same', activation='relu', input_shape=(264,264,3)))\n    model.add(MaxPooling2D(pool_size=(2, 2)))\n    model.add(Dropout(0.25))\n\n    model.add(Conv2D(64, (3, 3), padding='same', activation='relu'))\n    model.add(Conv2D(64, (3, 3), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2, 2)))\n    model.add(Dropout(0.25))\n\n    model.add(Conv2D(128, (3, 3), padding='same', activation='relu'))\n    model.add(Conv2D(128, (3, 3), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2, 2)))\n    model.add(Dropout(0.25))\n\n    model.add(Conv2D(256, (3, 3), padding='same', activation='relu'))\n    model.add(Conv2D(256, (3, 3), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2, 2)))\n    model.add(Dropout(0.25))\n\n    model.add(Conv2D(128, (3, 3), padding='same', activation='relu'))\n    model.add(Conv2D(128, (3, 3), padding='same', activation='relu'))\n    model.add(MaxPooling2D(pool_size=(2, 2)))\n    model.add(Dropout(0.25))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:44:00.589449Z","iopub.execute_input":"2024-11-13T08:44:00.58988Z","iopub.status.idle":"2024-11-13T08:44:00.665428Z","shell.execute_reply.started":"2024-11-13T08:44:00.589833Z","shell.execute_reply":"2024-11-13T08:44:00.664003Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# #My Block\n# from tensorflow.keras.applications import VGG16\n# from tensorflow.keras.applications import InceptionResNetV2\n# from tensorflow.keras.applications import InceptionV3\n# from keras.layers import GlobalAveragePooling2D, Dense, Dropout, Flatten, Input, Conv2D, multiply, LocallyConnected2D, Lambda\n# from keras.models import Model\n# from keras.layers import BatchNormalization\n# import numpy as np\n\n# # Define input shape\n# in_lay = Input(t_x.shape[1:])\n\n# # Choose which model you want to use (Uncomment the model you want to use)\n# # For VGG16:\n# # base_pretrained_model = VGG16(input_shape=t_x.shape[1:], include_top=False, weights='imagenet')\n\n# # For InceptionResNetV2:\n# # base_pretrained_model = InceptionResNetV2(input_shape=t_x.shape[1:], include_top=False, weights='imagenet')\n\n# # For InceptionV3:\n# base_pretrained_model = InceptionV3(input_shape=t_x.shape[1:], include_top=False, weights='imagenet')\n\n# # Set the pretrained model layers to be non-trainable\n# base_pretrained_model.trainable = False\n\n# # Extract features from the base model\n# pt_depth = 2048\n# pt_features = base_pretrained_model(in_lay)\n\n# # Apply BatchNormalization to the extracted features\n# bn_features = BatchNormalization()(pt_features)\n\n# # Attention mechanism\n# attn_layer = Conv2D(64, kernel_size=(1,1), padding='same', activation='relu')(Dropout(0.5)(bn_features))\n# attn_layer = Conv2D(16, kernel_size=(1,1), padding='same', activation='relu')(attn_layer)\n# attn_layer = Conv2D(8, kernel_size=(1,1), padding='same', activation='relu')(attn_layer)\n# attn_layer = Conv2D(1, kernel_size=(1,1), padding='valid', activation='sigmoid')(attn_layer)\n\n# # Make sure the number of channels of bn_features matches attn_layer\n# # Conv2D layer to match the number of channels of bn_features to match attn_layer (2048)\n# bn_features_adjusted = Conv2D(2048, kernel_size=(1, 1), padding='same', activation='linear')(bn_features)\n\n# # Fan it out to all of the channels\n# up_c2_w = np.ones((1, 1, 1, pt_depth))\n# up_c2 = Conv2D(pt_depth, kernel_size=(1, 1), padding='same', activation='linear', use_bias=False, weights=[up_c2_w])\n# up_c2.trainable = False\n# attn_layer = up_c2(attn_layer)\n\n# # Apply element-wise multiplication with adjusted bn_features\n# mask_features = multiply([attn_layer, bn_features_adjusted])\n\n# # Global Average Pooling\n# gap_features = GlobalAveragePooling2D()(mask_features)\n# gap_mask = GlobalAveragePooling2D()(attn_layer)\n\n# # To account for missing values from the attention model\n# gap = Lambda(lambda x: x[0] / x[1], name='RescaleGAP')([gap_features, gap_mask])\n# gap_dr = Dropout(0.25)(gap)\n\n# # Dense layers with Dropout\n# dr_steps = Dropout(0.25)(Dense(128, activation='relu')(gap_dr))\n\n# # Output layer\n# out_layer = Dense(t_y.shape[-1], activation='softmax')(dr_steps)\n\n# # Define the model\n# model = Model(inputs=[in_lay], outputs=[out_layer])\n\n# # Custom top-2 accuracy function\n# from keras.metrics import top_k_categorical_accuracy\n\n# def top_2_accuracy(in_gt, in_pred):\n#     return top_k_categorical_accuracy(in_gt, in_pred, k=2)\n\n# # Compile the model\n# model.compile(optimizer='adam', loss='categorical_crossentropy',\n#               metrics=['categorical_accuracy', top_2_accuracy])\n\n# # Print model summary\n# model.summary()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:44:00.666923Z","iopub.execute_input":"2024-11-13T08:44:00.667213Z","iopub.status.idle":"2024-11-13T08:44:00.735031Z","shell.execute_reply.started":"2024-11-13T08:44:00.66717Z","shell.execute_reply":"2024-11-13T08:44:00.734031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def create_models(model_type):\n#     model = Sequential()\n#     if (model_type==\"cnn\"):\n#         create_CNN_model(model)\n#     elif(model_type == \"resnet\"):\n#         pretrained_model= tf.keras.applications.ResNet50(include_top=False,\n#                    input_shape=(256,256,3),\n#                    pooling='avg',classes=5,\n#                    weights='imagenet')\n#         for layer in pretrained_model.layers:\n#                 layer.trainable=False\n\n#         model.add(pretrained_model)\n#     elif(model_type == \"googlenet\"):\n#         pretrained_model = tf.keras.applications.InceptionV3(\n#                     include_top=True,\n#                     weights=\"imagenet\",\n#                     input_shape=(299,299,3),\n#                     pooling='avg',\n#                     classes=1000)\n#         for layer in pretrained_model.layers:\n#                 layer.trainable=False\n#         model.add(pretrained_model)\n        \n#     model.add(Flatten())\n#     model.add(Dense(512, activation='relu'))\n#     model.add(Dropout(0.5))\n#     model.add(Dense(256, activation='relu'))\n#     model.add(Dropout(0.5))\n#     model.add(Dense(128, activation='relu'))\n#     model.add(Dropout(0.5))\n#     model.add(Dense(5, activation='softmax'))\n\n#     model.compile(loss='categorical_crossentropy',\n#                   optimizer=Adam(learning_rate=0.002),\n#                   metrics=['accuracy',\"AUC\"])\n#     return model;\n\nfrom keras.applications.vgg16 import VGG16 as PTModel\nfrom keras.applications.inception_resnet_v2 import InceptionResNetV2 as PTModel\nfrom keras.applications.inception_v3 import InceptionV3 as PTModel\nfrom tensorflow.keras.applications import VGG16\nfrom tensorflow.keras.applications import InceptionResNetV2\nfrom tensorflow.keras.applications import InceptionV3\n\nfrom keras.layers import GlobalAveragePooling2D, Dense, Dropout, Flatten, Input, Conv2D, multiply, LocallyConnected2D, Lambda\nfrom keras.models import Model\nin_lay = Input(t_x.shape[1:])\nbase_pretrained_model = PTModel(input_shape =  t_x.shape[1:], include_top = False, weights = 'imagenet')\nbase_pretrained_model.trainable = False\npt_depth = 2048\npt_features = base_pretrained_model(in_lay)\nfrom keras.layers import BatchNormalization\nbn_features = BatchNormalization()(pt_features)\n\n\n# here we do an attention mechanism to turn pixels in the GAP on an off\n\nattn_layer = Conv2D(64, kernel_size = (1,1), padding = 'same', activation = 'relu')(Dropout(0.5)(bn_features))\nattn_layer = Conv2D(16, kernel_size = (1,1), padding = 'same', activation = 'relu')(attn_layer)\nattn_layer = Conv2D(8, kernel_size = (1,1), padding = 'same', activation = 'relu')(attn_layer)\nattn_layer = Conv2D(1, \n                    kernel_size = (1,1), \n                    padding = 'valid', \n                    activation = 'sigmoid')(attn_layer)\n# fan it out to all of the channels\nup_c2_w = np.ones((1, 1, 1, pt_depth))\nup_c2 = Conv2D(pt_depth, kernel_size = (1,1), padding = 'same', \n               activation = 'linear', use_bias = False, weights = [up_c2_w])\nup_c2.trainable = False\nattn_layer = up_c2(attn_layer)\n\nmask_features = multiply([attn_layer, bn_features])\ngap_features = GlobalAveragePooling2D()(mask_features)\ngap_mask = GlobalAveragePooling2D()(attn_layer)\n# to account for missing values from the attention model\ngap = Lambda(lambda x: x[0]/x[1], name = 'RescaleGAP')([gap_features, gap_mask])\ngap_dr = Dropout(0.25)(gap)\ndr_steps = Dropout(0.25)(Dense(128, activation = 'relu')(gap_dr))\nout_layer = Dense(t_y.shape[-1], activation = 'softmax')(dr_steps)\nmodel = Model(inputs = [in_lay], outputs = [out_layer])\nfrom keras.metrics import top_k_categorical_accuracy\ndef top_2_accuracy(in_gt, in_pred):\n    return top_k_categorical_accuracy(in_gt, in_pred, k=2)\n\nmodel.compile(optimizer = 'adam', loss = 'categorical_crossentropy',\n                           metrics = ['categorical_accuracy', top_2_accuracy])\nmodel.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:44:00.7378Z","iopub.execute_input":"2024-11-13T08:44:00.738115Z","iopub.status.idle":"2024-11-13T08:44:08.423719Z","shell.execute_reply.started":"2024-11-13T08:44:00.738078Z","shell.execute_reply":"2024-11-13T08:44:08.422138Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model = create_models(\"googlenet\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:44:08.425037Z","iopub.execute_input":"2024-11-13T08:44:08.425371Z","iopub.status.idle":"2024-11-13T08:44:08.489124Z","shell.execute_reply.started":"2024-11-13T08:44:08.425324Z","shell.execute_reply":"2024-11-13T08:44:08.488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:44:08.490872Z","iopub.execute_input":"2024-11-13T08:44:08.491281Z","iopub.status.idle":"2024-11-13T08:44:08.577639Z","shell.execute_reply.started":"2024-11-13T08:44:08.491195Z","shell.execute_reply":"2024-11-13T08:44:08.57488Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Callbacks for Parameter Tuning","metadata":{}},{"cell_type":"code","source":"filepath = \"dr-detector.hdf5\"\ncheckpoint = ModelCheckpoint(filepath,\n                             monitor=\"val_top2_accuracy\",\n                             verbose=1,\n                             save_best_only=True,\n                             mode=\"max\")\n\nearlystop = EarlyStopping(monitor='val_categorical_accuracy',\n                          verbose=1, \n                          min_delta=0, \n                          patience=15, \n                          restore_best_weights=True)\n\nreduce_lr = ReduceLROnPlateau(monitor='val_loss', \n                              verbose=1,\n                              factor=0.2, \n                              patience=6, \n                              min_delta=0.0001,\n                              cooldown=0,\n                              min_lr=0.001)\n\n\ncallbacks = [checkpoint, earlystop, reduce_lr]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:44:08.579138Z","iopub.execute_input":"2024-11-13T08:44:08.579666Z","iopub.status.idle":"2024-11-13T08:44:08.643979Z","shell.execute_reply.started":"2024-11-13T08:44:08.579627Z","shell.execute_reply":"2024-11-13T08:44:08.643044Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history = model.fit(\n        x_train,\n        steps_per_epoch=x_train.samples // 64,\n        epochs=2,\n        validation_data=x_test,\n        validation_steps=x_test.samples // 64,\n        callbacks=callbacks)\nmodel.save_weights(\"dr_messidor.h5\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:44:08.645405Z","iopub.execute_input":"2024-11-13T08:44:08.645741Z","iopub.status.idle":"2024-11-13T08:53:03.119331Z","shell.execute_reply.started":"2024-11-13T08:44:08.645697Z","shell.execute_reply":"2024-11-13T08:53:03.118269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Evaluate the model on the training set\ntrain_results = model.evaluate(x_train)\ntrain_loss = train_results[0]  # Loss value\ntrain_accu = train_results[1]  # Accuracy value\n\n# Evaluate the model on the test set\ntest_results = model.evaluate(x_test)\ntest_loss = test_results[0]  # Loss value\ntest_accu = test_results[1]  # Accuracy value\n\n# Print the results\nprint(\"Final training accuracy = {:.2f} , validation accuracy = {:.2f}\".format(train_accu * 100, test_accu * 100))\nprint(\"Final training loss = {:.2f} , validation loss = {:.2f}\".format(train_loss, test_loss))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T08:53:03.120992Z","iopub.execute_input":"2024-11-13T08:53:03.121349Z","iopub.status.idle":"2024-11-13T09:01:33.307441Z","shell.execute_reply.started":"2024-11-13T08:53:03.121303Z","shell.execute_reply":"2024-11-13T09:01:33.306505Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"history.history","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T09:01:33.308939Z","iopub.execute_input":"2024-11-13T09:01:33.30921Z","iopub.status.idle":"2024-11-13T09:01:33.373443Z","shell.execute_reply.started":"2024-11-13T09:01:33.309168Z","shell.execute_reply":"2024-11-13T09:01:33.372393Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check available keys in history.history\nprint(\"Available keys in history:\", history.history.keys())\n\n# Plot the accuracy and loss graphs\nplt.figure(figsize=(14, 5))\n\n# Accuracy Plot\nplt.subplot(1, 2, 1)\nplt.plot(history.history['categorical_accuracy'], label='train accuracy')\nplt.plot(history.history['val_categorical_accuracy'], label='test accuracy')\nplt.title('Model Accuracy')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend(loc='upper left')\n\n# Loss Plot\nplt.subplot(1, 2, 2)\nplt.plot(history.history['loss'], label='train loss')\nplt.plot(history.history['val_loss'], label='test loss')\nplt.title('Model Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend(loc='upper left')\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T09:01:33.375188Z","iopub.execute_input":"2024-11-13T09:01:33.375682Z","iopub.status.idle":"2024-11-13T09:01:33.796028Z","shell.execute_reply.started":"2024-11-13T09:01:33.375632Z","shell.execute_reply":"2024-11-13T09:01:33.795137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"y_pred = model.predict(x_train)\ny_pred = np.argmax(y_pred, axis=1)\nclass_labels = x_test.class_indices\nclass_labels = {v:k for k,v in class_labels.items()}\n\nfrom sklearn.metrics import classification_report, confusion_matrix\ncm_train = confusion_matrix(x_train.classes, y_pred)\nprint('Confusion Matrix')\nprint(cm_train)\nprint('Classification Report')\ntarget_names = list(class_labels.values())\nprint(classification_report(x_train.classes, y_pred, target_names=target_names))\n\nplt.figure(figsize=(8,8))\nplt.imshow(cm_train, interpolation='nearest')\nplt.colorbar()\ntick_mark = np.arange(len(target_names))\n_ = plt.xticks(tick_mark, target_names, rotation=90)\n_ = plt.yticks(tick_mark, target_names)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T09:01:33.797596Z","iopub.execute_input":"2024-11-13T09:01:33.798241Z","iopub.status.idle":"2024-11-13T09:09:49.073463Z","shell.execute_reply.started":"2024-11-13T09:01:33.798191Z","shell.execute_reply":"2024-11-13T09:09:49.072421Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Print all layer names with their indices and types to identify the correct last conv layer\nfor i, layer in enumerate(model.layers):\n    print(f\"Layer {i}: Name = {layer.name}, Type = {type(layer)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T09:09:49.074894Z","iopub.execute_input":"2024-11-13T09:09:49.075157Z","iopub.status.idle":"2024-11-13T09:09:49.141993Z","shell.execute_reply.started":"2024-11-13T09:09:49.075126Z","shell.execute_reply":"2024-11-13T09:09:49.140937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras import models\nimport cv2\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom PIL import Image\n\n# Path to your image\nimg_path = \"./data/train_11/1116_right.jpeg\"\n\n# Open image using PIL and preprocess it\nimg = Image.open(img_path)\nimg = img.resize((256, 256))  # Resize to match model's expected input size\nimg_array = np.array(img)  # Convert to array\nimg_array = np.expand_dims(img_array, axis=0)  # Add batch dimension\nimg_array = img_array / 255.0  # Normalize if necessary\n\n# Define the Grad-CAM function\ndef get_gradcam_heatmap(model, img_array, class_idx, last_conv_layer_name):\n    # Get the last convolutional layer\n    grad_model = models.Model(\n        [model.inputs], [model.get_layer(last_conv_layer_name).output, model.output]\n    )\n\n    # Calculate the gradients of the top predicted class for the input image\n    with tf.GradientTape() as tape:\n        conv_outputs, predictions = grad_model(img_array)\n        loss = predictions[:, class_idx]\n\n    # Get the gradients of the loss with respect to the convolutional output\n    grads = tape.gradient(loss, conv_outputs)\n    \n    # Pool the gradients over all the filters\n    pooled_grads = tf.reduce_mean(grads, axis=(0, 1, 2))\n    \n    # Multiply each channel in the feature map array by the corresponding gradient\n    conv_outputs = conv_outputs[0]\n    heatmap = conv_outputs @ pooled_grads[..., tf.newaxis]\n    heatmap = tf.squeeze(heatmap)\n\n    # Apply ReLU to discard negative values and normalize between 0 and 1\n    heatmap = np.maximum(heatmap, 0) / np.max(heatmap)\n    return heatmap\n\n# Assuming the model has already been loaded and defined\n# Replace `model` with your model instance\n# Example: model = tf.keras.models.load_model('path/to/your/model')\n\n# Predict the class of the image and generate heatmap if infected\npreds = model.predict(img_array)\npred_class_idx = np.argmax(preds[0])  # Class with the highest prediction probability\n\n# Define the dictionary to map class names to indices (adjust to your model's classes)\nclass_labels = {'infected': 1, 'healthy': 0}  # Example class labels\n\n# Use 'conv2d_98' as the last conv layer\nlast_conv_layer_name = 'conv2d_98'\n\n# Only show heatmap if the prediction is for an infected class\nif pred_class_idx == class_labels.get('infected', None):\n    # Obtain the heatmap for the infected class\n    heatmap = get_gradcam_heatmap(model, img_array, pred_class_idx, last_conv_layer_name)\n\n    # Overlay the heatmap on the original image\n    original_img = cv2.imread(img_path)\n    heatmap = cv2.resize(heatmap, (original_img.shape[1], original_img.shape[0]))\n    heatmap = np.uint8(255 * heatmap)\n    heatmap = cv2.applyColorMap(heatmap, cv2.COLORMAP_JET)\n\n    # Overlay the heatmap on the original image with transparency\n    superimposed_img = cv2.addWeighted(heatmap, 0.4, original_img, 0.6, 0)\n    plt.figure(figsize=(8, 8))\n    plt.imshow(superimposed_img[..., ::-1])  # Convert BGR to RGB for matplotlib\n    plt.title(\"Infected Area Highlighted\")\n    plt.axis(\"off\")\n    plt.show()\nelse:\n    print(\"The model did not detect an infected area in this sample image.\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-13T09:09:49.143371Z","iopub.execute_input":"2024-11-13T09:09:49.14366Z","iopub.status.idle":"2024-11-13T09:09:54.372231Z","shell.execute_reply.started":"2024-11-13T09:09:49.143626Z","shell.execute_reply":"2024-11-13T09:09:54.371322Z"}},"outputs":[],"execution_count":null}]}