{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"},{"sourceId":6634555,"sourceType":"datasetVersion","datasetId":3830103}],"dockerImageVersionId":30699,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-06-18T15:10:29.784865Z","iopub.execute_input":"2024-06-18T15:10:29.785357Z","iopub.status.idle":"2024-06-18T15:10:40.212120Z","shell.execute_reply.started":"2024-06-18T15:10:29.785317Z","shell.execute_reply":"2024-06-18T15:10:40.211060Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install datasets transformers accelerate torch scikit-learn matplotlib wandb","metadata":{"execution":{"iopub.status.busy":"2024-06-18T15:10:40.213884Z","iopub.execute_input":"2024-06-18T15:10:40.214694Z","iopub.status.idle":"2024-06-18T15:11:13.759392Z","shell.execute_reply.started":"2024-06-18T15:10:40.214659Z","shell.execute_reply":"2024-06-18T15:11:13.758093Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install ipywidgets\n!jupyter nbextension enable --py widgetsnbextension","metadata":{"execution":{"iopub.status.busy":"2024-06-18T15:11:13.761129Z","iopub.execute_input":"2024-06-18T15:11:13.761526Z","iopub.status.idle":"2024-06-18T15:11:47.955128Z","shell.execute_reply.started":"2024-06-18T15:11:13.761490Z","shell.execute_reply":"2024-06-18T15:11:47.953870Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nfrom fastai.vision.all import *\nfrom fastcore.all import *","metadata":{"execution":{"iopub.status.busy":"2024-06-18T15:11:47.958226Z","iopub.execute_input":"2024-06-18T15:11:47.958602Z","iopub.status.idle":"2024-06-18T15:11:47.986374Z","shell.execute_reply.started":"2024-06-18T15:11:47.958568Z","shell.execute_reply":"2024-06-18T15:11:47.985584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import fastai\nprint(fastai.__version__)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T15:11:47.987501Z","iopub.execute_input":"2024-06-18T15:11:47.987817Z","iopub.status.idle":"2024-06-18T15:11:47.992820Z","shell.execute_reply.started":"2024-06-18T15:11:47.987789Z","shell.execute_reply":"2024-06-18T15:11:47.991822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os","metadata":{"execution":{"iopub.status.busy":"2024-06-18T15:11:47.993939Z","iopub.execute_input":"2024-06-18T15:11:47.994268Z","iopub.status.idle":"2024-06-18T15:11:48.000996Z","shell.execute_reply.started":"2024-06-18T15:11:47.994240Z","shell.execute_reply":"2024-06-18T15:11:48.000062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_path = Path(\"/kaggle/input/aptos2019-blindness-detection/train_images\")\ntest_path = Path(\"/kaggle/input/aptos2019-blindness-detection/test_images\")\nprint(len(get_image_files(train_path)))","metadata":{"execution":{"iopub.status.busy":"2024-06-18T15:11:48.002085Z","iopub.execute_input":"2024-06-18T15:11:48.002745Z","iopub.status.idle":"2024-06-18T15:11:48.825320Z","shell.execute_reply.started":"2024-06-18T15:11:48.002714Z","shell.execute_reply":"2024-06-18T15:11:48.824416Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = pd.read_csv(\"/kaggle/input/aptos2019-blindness-detection/train.csv\")\ntrain_df.head(100)\n\n#0 - No DR\n\n#1 - Mild\n\n#2 - Moderate\n\n#3 - Severe\n\n#4 - Proliferative DR","metadata":{"execution":{"iopub.status.busy":"2024-06-18T15:11:48.826669Z","iopub.execute_input":"2024-06-18T15:11:48.827331Z","iopub.status.idle":"2024-06-18T15:11:48.864447Z","shell.execute_reply.started":"2024-06-18T15:11:48.827304Z","shell.execute_reply":"2024-06-18T15:11:48.863420Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Up until now, we've just created variables for the multiple image folders and datasets.  From here, we'll follow https://docs.fast.ai/tutorial.medical_imaging.html in an attempt to train the model.","metadata":{}},{"cell_type":"code","source":"if not os.path.exists('/root/.cache/torch/hub/checkpoints/'):\n        os.makedirs('/root/.cache/torch/hub/checkpoints/')\n!cp '/kaggle/input/restnet50-for-fastai-offline-use/resnet50-0676ba61.pth' '/root/.cache/torch/hub/checkpoints/resnet50-11ad3fa6.pth'","metadata":{"execution":{"iopub.status.busy":"2024-06-18T15:11:48.865901Z","iopub.execute_input":"2024-06-18T15:11:48.866547Z","iopub.status.idle":"2024-06-18T15:11:50.625095Z","shell.execute_reply.started":"2024-06-18T15:11:48.866513Z","shell.execute_reply":"2024-06-18T15:11:50.623654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_x(r):\n    filename = f\"{r['id_code']}.png\"\n    #folder = str(r['folder']) # only used when training with small tile set\n    file_path = train_path / filename\n    #file_path = path / folder / filename\n    if os.path.exists(file_path):\n        return str(file_path)\n    else:\n        return \"/kaggle/input/aptos2019-blindness-detection/train_images/000c1434d8d7.png\"\n        \ndef get_y(r): return r['diagnosis']\n\nRetinopathyDataBlock = DataBlock( \n    blocks=(ImageBlock, CategoryBlock), \n    splitter=RandomSplitter(valid_pct=0.2, seed=42), #Random Splitter works well most of the time, this is the default splitter to use \n    get_x = get_x, \n    get_y = get_y,\n    item_tfms=Resize(460),\n    batch_tfms=[*aug_transforms(size=224, min_scale=0.75), Normalize.from_stats(*imagenet_stats)]\n    )\n#batch size (bs) helps prevent out of memory problems when training - try to lower the setting if we run into problems\ndls = RetinopathyDataBlock.dataloaders(train_df, num_workers=4, bs=36)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T15:11:50.629764Z","iopub.execute_input":"2024-06-18T15:11:50.630101Z","iopub.status.idle":"2024-06-18T15:11:52.113683Z","shell.execute_reply.started":"2024-06-18T15:11:50.630071Z","shell.execute_reply":"2024-06-18T15:11:52.112652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls.show_batch(max_n=6)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T15:11:52.114959Z","iopub.execute_input":"2024-06-18T15:11:52.115294Z","iopub.status.idle":"2024-06-18T15:11:59.772055Z","shell.execute_reply.started":"2024-06-18T15:11:52.115268Z","shell.execute_reply":"2024-06-18T15:11:59.771030Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\ntorch.cuda.empty_cache()\n\nfrom fastai.callback.mixup import *\n# Use gradient accumulation\n\n#.to_fp16 reduces the floating point precision to reduce memory use\nlearn = vision_learner(dls, resnet50,metrics=[accuracy, error_rate], loss_func=CrossEntropyLossFlat()).to_fp16()\n#learn = vision_learner(dls, resnet50,metrics=[accuracy, error_rate], loss_func=CrossEntropyLossFlat(), cbs=[MixUp(), GradientAccumulation(n_acc=16)]).to_fp16()\n#learn = vision_learner(dls, resnet18, metrics=[accuracy, error_rate], loss_func=CrossEntropyLossFlat(), cbs=MixUp()).to_fp16()","metadata":{"execution":{"iopub.status.busy":"2024-06-18T15:11:59.773265Z","iopub.execute_input":"2024-06-18T15:11:59.773554Z","iopub.status.idle":"2024-06-18T15:12:00.508120Z","shell.execute_reply.started":"2024-06-18T15:11:59.773530Z","shell.execute_reply":"2024-06-18T15:12:00.507325Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#find the learning rate\nlrs = learn.lr_find(suggest_funcs=(minimum, steep, valley, slide))","metadata":{"execution":{"iopub.status.busy":"2024-06-18T15:12:00.509272Z","iopub.execute_input":"2024-06-18T15:12:00.509616Z","iopub.status.idle":"2024-06-18T15:15:19.397659Z","shell.execute_reply.started":"2024-06-18T15:12:00.509585Z","shell.execute_reply":"2024-06-18T15:15:19.396683Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#force gpu\nlearn.model.cuda()\n#use the minimum learning rate to start\nlearn.fine_tune(50,freeze_epochs=3, base_lr=lrs.minimum)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T15:15:19.399037Z","iopub.execute_input":"2024-06-18T15:15:19.399338Z","iopub.status.idle":"2024-06-18T18:21:10.934610Z","shell.execute_reply.started":"2024-06-18T15:15:19.399310Z","shell.execute_reply":"2024-06-18T18:21:10.933491Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#we can also use fine_tune by itself and it will use the appropiate learning rate for most cases\n#learn.fine_tune(10)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T18:21:10.936125Z","iopub.execute_input":"2024-06-18T18:21:10.936445Z","iopub.status.idle":"2024-06-18T18:21:10.940771Z","shell.execute_reply.started":"2024-06-18T18:21:10.936417Z","shell.execute_reply":"2024-06-18T18:21:10.939862Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"interp = ClassificationInterpretation.from_learner(learn)\ninterp.plot_confusion_matrix()","metadata":{"execution":{"iopub.status.busy":"2024-06-18T18:21:10.941846Z","iopub.execute_input":"2024-06-18T18:21:10.942146Z","iopub.status.idle":"2024-06-18T18:22:31.576080Z","shell.execute_reply.started":"2024-06-18T18:21:10.942121Z","shell.execute_reply":"2024-06-18T18:22:31.574536Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#where are we most confused?\ninterp.most_confused()","metadata":{"execution":{"iopub.status.busy":"2024-06-18T18:22:31.578652Z","iopub.execute_input":"2024-06-18T18:22:31.579427Z","iopub.status.idle":"2024-06-18T18:23:11.357828Z","shell.execute_reply.started":"2024-06-18T18:22:31.579372Z","shell.execute_reply":"2024-06-18T18:23:11.356818Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#try to infer the Retinopathy level for one of the test images\nlearn.predict('/kaggle/input/aptos2019-blindness-detection/test_images/0005cfc8afb6.png')","metadata":{"execution":{"iopub.status.busy":"2024-06-18T18:23:11.359387Z","iopub.execute_input":"2024-06-18T18:23:11.360061Z","iopub.status.idle":"2024-06-18T18:23:11.645721Z","shell.execute_reply.started":"2024-06-18T18:23:11.359999Z","shell.execute_reply":"2024-06-18T18:23:11.644911Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#export the model\nlearn.export('Retinopathy-June-2024.pkl')","metadata":{"execution":{"iopub.status.busy":"2024-06-18T18:23:11.646879Z","iopub.execute_input":"2024-06-18T18:23:11.647181Z","iopub.status.idle":"2024-06-18T18:23:11.933926Z","shell.execute_reply.started":"2024-06-18T18:23:11.647157Z","shell.execute_reply":"2024-06-18T18:23:11.933175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from PIL import Image\nimport gc\n\n# Load the learner\nlearn_inf = load_learner('/kaggle/working/Retinopathy-June-2024.pkl')\n\n# Get a list of all image files in the folder\nimage_files = [f for f in os.listdir(test_path) if f.endswith('.png')]\n\n# Image size\ntarget_size = (224, 224)\n\n# Make predictions for each image and store the results in a list\npredictions = []\nfor image_file in image_files:\n    # Get the image ID from the filename\n    image_id = os.path.splitext(image_file)[0]\n    print(f\"Processing image: {image_id}\")\n\n    # Open the image file\n    image = Image.open(test_path/image_file)\n\n    # Resize the image to the desired size\n    resized_image = image.resize(target_size)\n\n    # Get the prediction for the image\n    pred, _, _ = learn_inf.predict(resized_image)\n\n    # Append the result to the predictions list\n    result = {'id_code': image_id, 'diagnosis': str(pred)}\n    predictions.append(result)\n    \n    del image\n    del resized_image\n    gc.collect() \n\n# Create a DataFrame from the predictions list and save it to a CSV file\ndf = pd.DataFrame(predictions)\ndf.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-06-18T18:23:11.935029Z","iopub.execute_input":"2024-06-18T18:23:11.935310Z","iopub.status.idle":"2024-06-18T18:35:48.909437Z","shell.execute_reply.started":"2024-06-18T18:23:11.935285Z","shell.execute_reply":"2024-06-18T18:35:48.908620Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]}]}