{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%reload_ext autoreload\n%autoreload 2\n%matplotlib inline","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:29:41.708055Z","iopub.execute_input":"2024-11-21T08:29:41.708420Z","iopub.status.idle":"2024-11-21T08:29:41.727546Z","shell.execute_reply.started":"2024-11-21T08:29:41.708389Z","shell.execute_reply":"2024-11-21T08:29:41.726737Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install git+https://github.com/rwightman/pytorch-image-models.git","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:49:37.407937Z","iopub.execute_input":"2024-11-21T08:49:37.408340Z","iopub.status.idle":"2024-11-21T08:49:40.392890Z","shell.execute_reply.started":"2024-11-21T08:49:37.408306Z","shell.execute_reply":"2024-11-21T08:49:40.391970Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport os\nimport random\nimport numpy as np\nimport torch\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\nimport cv2\nfrom sklearn.model_selection import train_test_split\nimport time\nimport torchvision\nimport torch.nn as nn\nfrom torch.cuda.amp import autocast, GradScaler\nfrom tqdm import tqdm\nfrom PIL import Image, ImageFile\nfrom torch.utils.data import Dataset\nimport torch.optim as optim\nfrom torchvision import transforms\nfrom torch.optim import lr_scheduler\nimport os\nimport timm\nimport ipywidgets as widgets\nimport os\nfrom collections import Counter\nfrom sklearn.utils import resample\n\ndevice = torch.device(\"cuda:0\")\nImageFile.LOAD_TRUNCATED_IMAGES = True","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:44.602673Z","iopub.execute_input":"2024-11-21T08:30:44.603479Z","iopub.status.idle":"2024-11-21T08:30:44.653236Z","shell.execute_reply.started":"2024-11-21T08:30:44.603429Z","shell.execute_reply":"2024-11-21T08:30:44.652147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.listdir('../input')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:44.769150Z","iopub.execute_input":"2024-11-21T08:30:44.769436Z","iopub.status.idle":"2024-11-21T08:30:44.814315Z","shell.execute_reply.started":"2024-11-21T08:30:44.769410Z","shell.execute_reply":"2024-11-21T08:30:44.813445Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print('Make sure cudnn is enabled:', torch.backends.cudnn.enabled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:44.913702Z","iopub.execute_input":"2024-11-21T08:30:44.914490Z","iopub.status.idle":"2024-11-21T08:30:44.964452Z","shell.execute_reply.started":"2024-11-21T08:30:44.914442Z","shell.execute_reply":"2024-11-21T08:30:44.963508Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n\nSEED = 999\nseed_everything(SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:45.055713Z","iopub.execute_input":"2024-11-21T08:30:45.056534Z","iopub.status.idle":"2024-11-21T08:30:45.103829Z","shell.execute_reply.started":"2024-11-21T08:30:45.056488Z","shell.execute_reply":"2024-11-21T08:30:45.102783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base_image_dir = os.path.join('..', 'input/aptos2019-blindness-detection/')\ntrain_dir = os.path.join(base_image_dir,'train_images/')\ndf = pd.read_csv(os.path.join(base_image_dir, 'train.csv'))\ndf['path'] = df['id_code'].map(lambda x: os.path.join(train_dir,'{}.png'.format(x)))\ndf = df.drop(columns=['id_code'])\ndf = df.sample(frac=1).reset_index(drop=True) #shuffle dataframe\ndf.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:45.233633Z","iopub.execute_input":"2024-11-21T08:30:45.234614Z","iopub.status.idle":"2024-11-21T08:30:45.302002Z","shell.execute_reply.started":"2024-11-21T08:30:45.234558Z","shell.execute_reply":"2024-11-21T08:30:45.301137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len_df = len(df)\nprint(f\"There are {len_df} images\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:45.383920Z","iopub.execute_input":"2024-11-21T08:30:45.384202Z","iopub.status.idle":"2024-11-21T08:30:45.428469Z","shell.execute_reply.started":"2024-11-21T08:30:45.384177Z","shell.execute_reply":"2024-11-21T08:30:45.427734Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df['diagnosis'].hist(figsize = (10, 5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:45.540412Z","iopub.execute_input":"2024-11-21T08:30:45.541063Z","iopub.status.idle":"2024-11-21T08:30:45.771550Z","shell.execute_reply.started":"2024-11-21T08:30:45.541030Z","shell.execute_reply":"2024-11-21T08:30:45.770711Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_labels = df['diagnosis']\ndf_labels","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:45.773231Z","iopub.execute_input":"2024-11-21T08:30:45.773849Z","iopub.status.idle":"2024-11-21T08:30:45.820945Z","shell.execute_reply.started":"2024-11-21T08:30:45.773807Z","shell.execute_reply":"2024-11-21T08:30:45.820145Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_path = df['path']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:45.869261Z","iopub.execute_input":"2024-11-21T08:30:45.869500Z","iopub.status.idle":"2024-11-21T08:30:45.912204Z","shell.execute_reply.started":"2024-11-21T08:30:45.869477Z","shell.execute_reply":"2024-11-21T08:30:45.911433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"label_map = {0: \"No DR\", 1:\"Mild\", 2:\"Moderate\", 3:\"Severe\", 4:\"Proliferative DR\"}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:46.060771Z","iopub.execute_input":"2024-11-21T08:30:46.061255Z","iopub.status.idle":"2024-11-21T08:30:46.104916Z","shell.execute_reply.started":"2024-11-21T08:30:46.061227Z","shell.execute_reply":"2024-11-21T08:30:46.104041Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def plot_unique_classes_from_df(df, image_col, label_col, label_map):\n    \"\"\"\n    Plot one sample image for each unique class in ascending order of class labels.\n\n    Args:\n    - df (pd.DataFrame): DataFrame containing image paths and labels.\n    - image_col (str): Column name for image paths.\n    - label_col (str): Column name for labels.\n    - label_map (dict): Mapping of label IDs to class names.\n    \"\"\"\n    # Get unique labels and sort them in ascending order\n    unique_labels = sorted(df[label_col].unique())\n    fig, axes = plt.subplots(1, len(unique_labels), figsize=(15, 5))\n\n    for i, label in enumerate(unique_labels):\n        # Get the first image of the class\n        sample_row = df[df[label_col] == label].iloc[0]\n        sample_image_path = sample_row[image_col]\n        \n        # Load and display the image\n        image = cv2.imread(sample_image_path)\n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)  # Convert BGR to RGB\n\n         # Resize the image\n        image = cv2.resize(image, (300, 300))\n        axes[i].imshow(image)\n        axes[i].axis(\"off\")\n        axes[i].set_title(label_map[label])\n\n    plt.tight_layout()\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:46.267316Z","iopub.execute_input":"2024-11-21T08:30:46.267570Z","iopub.status.idle":"2024-11-21T08:30:46.312988Z","shell.execute_reply.started":"2024-11-21T08:30:46.267546Z","shell.execute_reply":"2024-11-21T08:30:46.312282Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot unique class images\nplot_unique_classes_from_df(df, 'path', 'diagnosis', label_map)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:46.422228Z","iopub.execute_input":"2024-11-21T08:30:46.422811Z","iopub.status.idle":"2024-11-21T08:30:47.654950Z","shell.execute_reply.started":"2024-11-21T08:30:46.422777Z","shell.execute_reply":"2024-11-21T08:30:47.654070Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"im = Image.open(df['path'][1])\nwidth, height = im.size\nprint(width,height) \nplt.imshow(np.asarray(im))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:47.656629Z","iopub.execute_input":"2024-11-21T08:30:47.656902Z","iopub.status.idle":"2024-11-21T08:30:48.813942Z","shell.execute_reply.started":"2024-11-21T08:30:47.656873Z","shell.execute_reply":"2024-11-21T08:30:48.813137Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"IMG_SIZE=224","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:48.815155Z","iopub.execute_input":"2024-11-21T08:30:48.815518Z","iopub.status.idle":"2024-11-21T08:30:48.862266Z","shell.execute_reply.started":"2024-11-21T08:30:48.815479Z","shell.execute_reply":"2024-11-21T08:30:48.861301Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load and display the image\nimage = cv2.imread(df['path'][1])\nimage = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)  # Convert BGR to RGB\n\n# Resize the image\nimage = cv2.resize(image, (IMG_SIZE, IMG_SIZE))\nwidth, height, _ = image.shape\nprint(width,height) \nplt.imshow(image)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:48.864593Z","iopub.execute_input":"2024-11-21T08:30:48.864853Z","iopub.status.idle":"2024-11-21T08:30:49.267984Z","shell.execute_reply.started":"2024-11-21T08:30:48.864828Z","shell.execute_reply":"2024-11-21T08:30:49.267143Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')\ndf_test = pd.read_csv('../input/aptos2019-blindness-detection/test.csv')\n\nx = df_train['id_code']\ny = df_train['diagnosis']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:49.269203Z","iopub.execute_input":"2024-11-21T08:30:49.269532Z","iopub.status.idle":"2024-11-21T08:30:49.324246Z","shell.execute_reply.started":"2024-11-21T08:30:49.269495Z","shell.execute_reply":"2024-11-21T08:30:49.323458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_x, valid_x, train_y, valid_y = train_test_split(x, y, test_size=0.15,\n                                                      stratify=y,\n                                                      shuffle=True,\n                                                      random_state=SEED)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:49.325538Z","iopub.execute_input":"2024-11-21T08:30:49.325986Z","iopub.status.idle":"2024-11-21T08:30:49.376016Z","shell.execute_reply.started":"2024-11-21T08:30:49.325933Z","shell.execute_reply":"2024-11-21T08:30:49.375159Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def crop_image_from_gray(img,tol=7):\n    if img.ndim ==2:\n        mask = img>tol\n        return img[np.ix_(mask.any(1),mask.any(0))]\n    elif img.ndim==3:\n        gray_img = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n        mask = gray_img>tol\n        \n        check_shape = img[:,:,0][np.ix_(mask.any(1),mask.any(0))].shape[0]\n        if (check_shape == 0): # image is too dark so that we crop out everything,\n            return img # return original image\n        else:\n            img1=img[:,:,0][np.ix_(mask.any(1),mask.any(0))]\n            img2=img[:,:,1][np.ix_(mask.any(1),mask.any(0))]\n            img3=img[:,:,2][np.ix_(mask.any(1),mask.any(0))]\n    #         print(img1.shape,img2.shape,img3.shape)\n            img = np.stack([img1,img2,img3],axis=-1)\n    #         print(img.shape)\n        return img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:49.377200Z","iopub.execute_input":"2024-11-21T08:30:49.377523Z","iopub.status.idle":"2024-11-21T08:30:49.432745Z","shell.execute_reply.started":"2024-11-21T08:30:49.377487Z","shell.execute_reply":"2024-11-21T08:30:49.432130Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_color(path, sigmaX=10):\n    image = cv2.imread(path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    image = crop_image_from_gray(image)\n    image = cv2.resize(image, (IMG_SIZE, IMG_SIZE))\n    image=cv2.addWeighted ( image,4, cv2.GaussianBlur( image , (0,0) , sigmaX) ,-4 ,128)\n        \n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:49.433643Z","iopub.execute_input":"2024-11-21T08:30:49.433956Z","iopub.status.idle":"2024-11-21T08:30:49.491311Z","shell.execute_reply.started":"2024-11-21T08:30:49.433920Z","shell.execute_reply":"2024-11-21T08:30:49.490486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"%%time\n\nNUM_SAMP=7\nfig = plt.figure(figsize=(25, 16))\nfor class_id in sorted(train_y.unique()):\n    for i, (idx, row) in enumerate(df_train.loc[df_train['diagnosis'] == class_id].sample(NUM_SAMP, random_state=SEED).iterrows()):\n        ax = fig.add_subplot(5, NUM_SAMP, class_id * NUM_SAMP + i + 1, xticks=[], yticks=[])\n        path=f\"../input/aptos2019-blindness-detection/train_images/{row['id_code']}.png\"\n        image = load_color(path,sigmaX=30)\n\n        plt.imshow(image)\n        ax.set_title('%d-%d-%s' % (class_id, idx, row['id_code']) )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:30:49.492512Z","iopub.execute_input":"2024-11-21T08:30:49.492798Z","iopub.status.idle":"2024-11-21T08:31:02.076068Z","shell.execute_reply.started":"2024-11-21T08:30:49.492773Z","shell.execute_reply":"2024-11-21T08:31:02.074613Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_super_basic_albumentations(path):\n    \"\"\"Load an image and apply preprocessing pipeline with Albumentations.\"\"\"\n    # Read image\n    image = cv2.imread(path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n    transform = A.Compose([\n        A.Resize(IMG_SIZE, IMG_SIZE),  # Resize to (IMG_SIZE, IMG_SIZE)\n        ToTensorV2()  # Convert to tensor for PyTorch\n    ])\n\n    # Apply transformations\n    augmented = transform(image=image)\n    return augmented['image']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:31:02.079379Z","iopub.execute_input":"2024-11-21T08:31:02.079718Z","iopub.status.idle":"2024-11-21T08:31:02.126112Z","shell.execute_reply.started":"2024-11-21T08:31:02.079684Z","shell.execute_reply":"2024-11-21T08:31:02.125325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_basic_albumentations(path):\n    \"\"\"Load an image and apply preprocessing pipeline with Albumentations.\"\"\"\n    # Read image\n    image = cv2.imread(path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2GRAY)\n    dst = cv2.equalizeHist(image)\n    value = [0, 0, 0]\n    top = 476  # shape[0] = rows\n    bottom = top\n    left = 0  # shape[1] = cols\n    right = 0\n    dst = cv2.copyMakeBorder(dst, top, bottom, left, right, cv2.BORDER_CONSTANT, None, value)\n    # Define Albumentations transformations\n    transform = A.Compose([\n        A.Resize(IMG_SIZE, IMG_SIZE),  # Resize to (IMG_SIZE, IMG_SIZE)\n        # A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),  # Normalize for pre-trained models\n        ToTensorV2()  # Convert to tensor for PyTorch\n    ])\n\n    # Apply transformations\n    augmented = transform(image=dst)\n    return augmented['image']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:42:26.319032Z","iopub.execute_input":"2024-11-21T08:42:26.319719Z","iopub.status.idle":"2024-11-21T08:42:26.367080Z","shell.execute_reply.started":"2024-11-21T08:42:26.319684Z","shell.execute_reply":"2024-11-21T08:42:26.366249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# def load_basic_albumentations(path):\n#     \"\"\"Load an image and apply preprocessing pipeline with Albumentations.\"\"\"\n#     # Read image\n#     image = cv2.imread(path)\n#     image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n#     # Define Albumentations transformations\n#     transform = A.Compose([\n#         A.Resize(IMG_SIZE, IMG_SIZE),  # Resize to (IMG_SIZE, IMG_SIZE)\n#         A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),  # Normalize for pre-trained models\n#         ToTensorV2()  # Convert to tensor for PyTorch\n#     ])\n\n#     # Apply transformations\n#     augmented = transform(image=dst)\n#     return augmented['image']","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def load_color_albumentations(path, sigmaX=25):\n    \"\"\"Load an image and apply preprocessing pipeline with Albumentations.\"\"\"\n    # Read image\n    image = cv2.imread(path)\n    image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n    # Crop image using a custom transformation\n    image = crop_image_from_gray(image)\n\n    # Define Albumentations transformations\n    transform = A.Compose([\n        A.RGBShift(r_shift_limit=0, g_shift_limit=0, b_shift_limit=0, p=1.0),  # Equivalent to cv2.COLOR_BGR2RGB\n        A.Resize(IMG_SIZE, IMG_SIZE),  # Resize to (IMG_SIZE, IMG_SIZE)\n        A.GaussianBlur(blur_limit=(sigmaX, sigmaX), p=1.0),  # Gaussian blur\n        # A.Normalize(mean=(0.485, 0.456, 0.406), std=(0.229, 0.224, 0.225)),\n        ToTensorV2()  # Convert to tensor for PyTorch\n    ])\n\n    # Apply transformations\n    augmented = transform(image=image)\n    return augmented['image']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:31:35.090204Z","iopub.execute_input":"2024-11-21T08:31:35.090517Z","iopub.status.idle":"2024-11-21T08:31:35.137638Z","shell.execute_reply.started":"2024-11-21T08:31:35.090492Z","shell.execute_reply":"2024-11-21T08:31:35.136707Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img = load_basic_albumentations('/kaggle/input/aptos2019-blindness-detection/test_images/0005cfc8afb6.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:42:29.575899Z","iopub.execute_input":"2024-11-21T08:42:29.576299Z","iopub.status.idle":"2024-11-21T08:42:29.629477Z","shell.execute_reply.started":"2024-11-21T08:42:29.576265Z","shell.execute_reply":"2024-11-21T08:42:29.628566Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img.dtype","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:42:30.103343Z","iopub.execute_input":"2024-11-21T08:42:30.103646Z","iopub.status.idle":"2024-11-21T08:42:30.150307Z","shell.execute_reply.started":"2024-11-21T08:42:30.103622Z","shell.execute_reply":"2024-11-21T08:42:30.149461Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img.ndim","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:42:30.297487Z","iopub.execute_input":"2024-11-21T08:42:30.297767Z","iopub.status.idle":"2024-11-21T08:42:30.345610Z","shell.execute_reply.started":"2024-11-21T08:42:30.297742Z","shell.execute_reply":"2024-11-21T08:42:30.344527Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(dst)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:42:37.358339Z","iopub.execute_input":"2024-11-21T08:42:37.358924Z","iopub.status.idle":"2024-11-21T08:42:38.056324Z","shell.execute_reply.started":"2024-11-21T08:42:37.358889Z","shell.execute_reply":"2024-11-21T08:42:38.055477Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(img.permute(1, 2, 0))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:32:17.804850Z","iopub.execute_input":"2024-11-21T08:32:17.805215Z","iopub.status.idle":"2024-11-21T08:32:18.083370Z","shell.execute_reply.started":"2024-11-21T08:32:17.805183Z","shell.execute_reply":"2024-11-21T08:32:18.082563Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RetinopathyDatasetTrain(Dataset):\n\n    def __init__(self, csv_file):\n        self.data = csv_file\n        self.data = self.data.reset_index(drop=True)\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        img_name = os.path.join('../input/aptos2019-blindness-detection/train_images', self.data.loc[idx, 'id_code'] + '.png')\n        image = load_basic_albumentations(img_name)\n        label = torch.tensor(self.data.loc[idx, 'diagnosis'])\n        return {'image': image,\n                'labels': label\n                }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:41.335659Z","iopub.execute_input":"2024-11-21T08:43:41.336654Z","iopub.status.idle":"2024-11-21T08:43:41.382764Z","shell.execute_reply.started":"2024-11-21T08:43:41.336616Z","shell.execute_reply":"2024-11-21T08:43:41.381729Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_df = pd.read_csv('../input/aptos2019-blindness-detection/train.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:41.541592Z","iopub.execute_input":"2024-11-21T08:43:41.541866Z","iopub.status.idle":"2024-11-21T08:43:41.592843Z","shell.execute_reply.started":"2024-11-21T08:43:41.541841Z","shell.execute_reply":"2024-11-21T08:43:41.592150Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"file_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:41.708235Z","iopub.execute_input":"2024-11-21T08:43:41.708522Z","iopub.status.idle":"2024-11-21T08:43:41.757725Z","shell.execute_reply.started":"2024-11-21T08:43:41.708497Z","shell.execute_reply":"2024-11-21T08:43:41.756824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df, val_df = train_test_split(file_df, test_size=0.1,\n                                    random_state=69, stratify=file_df['diagnosis'])\nlen(train_df), len(val_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:41.812951Z","iopub.execute_input":"2024-11-21T08:43:41.813448Z","iopub.status.idle":"2024-11-21T08:43:41.863182Z","shell.execute_reply.started":"2024-11-21T08:43:41.813421Z","shell.execute_reply":"2024-11-21T08:43:41.862326Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:41.963545Z","iopub.execute_input":"2024-11-21T08:43:41.964030Z","iopub.status.idle":"2024-11-21T08:43:42.013082Z","shell.execute_reply.started":"2024-11-21T08:43:41.964000Z","shell.execute_reply":"2024-11-21T08:43:42.012315Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:42.107760Z","iopub.execute_input":"2024-11-21T08:43:42.108007Z","iopub.status.idle":"2024-11-21T08:43:42.155602Z","shell.execute_reply.started":"2024-11-21T08:43:42.107983Z","shell.execute_reply":"2024-11-21T08:43:42.154887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Group by diagnosis and sample 5 images from each class\nsampled_images = val_df.groupby('diagnosis').apply(lambda x: x.sample(5))\n\n# Display the sampled images\nprint(sampled_images)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:42.244630Z","iopub.execute_input":"2024-11-21T08:43:42.244914Z","iopub.status.idle":"2024-11-21T08:43:42.294586Z","shell.execute_reply.started":"2024-11-21T08:43:42.244887Z","shell.execute_reply":"2024-11-21T08:43:42.293653Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_paths = []\nfor idx, row in sampled_images.iterrows():\n    # Construct the image path (assuming images are in a directory called 'images')\n    img_path = f\"/kaggle/input/aptos2019-blindness-detection/train_images/{row['id_code']}.png\"\n    image_paths.append(img_path)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:42.491618Z","iopub.execute_input":"2024-11-21T08:43:42.491941Z","iopub.status.idle":"2024-11-21T08:43:42.538783Z","shell.execute_reply.started":"2024-11-21T08:43:42.491911Z","shell.execute_reply":"2024-11-21T08:43:42.537962Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\nimport os\nimport shutil\n\n# Define the zip file name\nzip_file_name = \"test_images.zip\"\ndestination_dir = \"/kaggle/working/\"  # Modify this path as needed\n\n# Create a zip file and add images to it\nwith zipfile.ZipFile(zip_file_name, 'w', zipfile.ZIP_DEFLATED) as zipf:\n    for img_path in image_paths:\n        try:\n            # Add each image to the zip file\n            zipf.write(img_path, os.path.basename(img_path))  # Save with original filename\n            print(f\"Added {img_path} to {zip_file_name}\")\n        except FileNotFoundError:\n            print(f\"File {img_path} not found.\")\n        except Exception as e:\n            print(f\"Error adding {img_path}: {e}\")\n\nprint(f\"Zip file created: {zip_file_name}\")\n\n# Move the zip file to the destination directory\nif not os.path.exists(destination_dir):\n    os.makedirs(destination_dir)  # Create the directory if it doesn't exist\n\n# Full path for the new location\nnew_zip_path = os.path.join(destination_dir, zip_file_name)\n\n# Move the zip file\nshutil.move(zip_file_name, new_zip_path)\nprint(f\"Zip file moved to: {new_zip_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:42.853977Z","iopub.execute_input":"2024-11-21T08:43:42.854613Z","iopub.status.idle":"2024-11-21T08:43:46.515440Z","shell.execute_reply.started":"2024-11-21T08:43:42.854581Z","shell.execute_reply":"2024-11-21T08:43:46.514543Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = RetinopathyDatasetTrain(csv_file=train_df)\nval_dataset = RetinopathyDatasetTrain(csv_file=val_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:46.516810Z","iopub.execute_input":"2024-11-21T08:43:46.517122Z","iopub.status.idle":"2024-11-21T08:43:46.561311Z","shell.execute_reply.started":"2024-11-21T08:43:46.517074Z","shell.execute_reply":"2024-11-21T08:43:46.560471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset.data","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:46.562341Z","iopub.execute_input":"2024-11-21T08:43:46.562606Z","iopub.status.idle":"2024-11-21T08:43:46.610804Z","shell.execute_reply.started":"2024-11-21T08:43:46.562565Z","shell.execute_reply":"2024-11-21T08:43:46.609861Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"a=train_dataset.__getitem__(386)\n# plt.imshow(a[0][0])\n# a[1].shape\nplt.imshow(a['image'].cpu().permute(1,2,0))\nprint(a['labels'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:46.612553Z","iopub.execute_input":"2024-11-21T08:43:46.612802Z","iopub.status.idle":"2024-11-21T08:43:46.898757Z","shell.execute_reply.started":"2024-11-21T08:43:46.612778Z","shell.execute_reply":"2024-11-21T08:43:46.897958Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_data_loader = torch.utils.data.DataLoader(train_dataset, batch_size=16, shuffle=True, num_workers=4)\nval_data_loader = torch.utils.data.DataLoader(val_dataset, batch_size=16, shuffle=False, num_workers=4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:46.899973Z","iopub.execute_input":"2024-11-21T08:43:46.900343Z","iopub.status.idle":"2024-11-21T08:43:46.946325Z","shell.execute_reply.started":"2024-11-21T08:43:46.900304Z","shell.execute_reply":"2024-11-21T08:43:46.945601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"timm.list_models(pretrained=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:46.947272Z","iopub.execute_input":"2024-11-21T08:43:46.947531Z","iopub.status.idle":"2024-11-21T08:43:47.008805Z","shell.execute_reply.started":"2024-11-21T08:43:46.947507Z","shell.execute_reply":"2024-11-21T08:43:47.007969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model_name = 'tf_efficientnetv2_m_in21k'\nmodel = timm.create_model(model_name, pretrained=True, in_chans=3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:47.009873Z","iopub.execute_input":"2024-11-21T08:43:47.010139Z","iopub.status.idle":"2024-11-21T08:43:50.313574Z","shell.execute_reply.started":"2024-11-21T08:43:47.010080Z","shell.execute_reply":"2024-11-21T08:43:50.312544Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.conv_stem = nn","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:44:39.037371Z","iopub.execute_input":"2024-11-21T08:44:39.037708Z","iopub.status.idle":"2024-11-21T08:44:39.084948Z","shell.execute_reply.started":"2024-11-21T08:44:39.037678Z","shell.execute_reply":"2024-11-21T08:44:39.084149Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.eval()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.conv_stem = timm.models.layers.Conv2dSame(1, 24, kernel_size=(3, 3), stride=(2, 2), bias=False)\nmodel.classifier = nn.Linear(model.classifier.in_features, 1)\nmodel = model.to(device)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:48:02.225695Z","iopub.execute_input":"2024-11-21T08:48:02.226608Z","iopub.status.idle":"2024-11-21T08:48:02.295967Z","shell.execute_reply.started":"2024-11-21T08:48:02.226570Z","shell.execute_reply":"2024-11-21T08:48:02.294740Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"## Best Curve\n# optimizer = optim.AdamW(model.parameters(), lr=0.000001) #, weight_decay=0.5\n# criterion = nn.SmoothL1Loss()\n# lr_scheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=3, verbose=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:50.680594Z","iopub.execute_input":"2024-11-21T08:43:50.681132Z","iopub.status.idle":"2024-11-21T08:43:50.725421Z","shell.execute_reply.started":"2024-11-21T08:43:50.681073Z","shell.execute_reply":"2024-11-21T08:43:50.724573Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# optimizer = optim.AdamW(model.parameters(), lr=0.00005) #, weight_decay=0.5\n# criterion = nn.SmoothL1Loss()\n# lr_scheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=3, verbose=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:50.728431Z","iopub.execute_input":"2024-11-21T08:43:50.728747Z","iopub.status.idle":"2024-11-21T08:43:50.779194Z","shell.execute_reply.started":"2024-11-21T08:43:50.728699Z","shell.execute_reply":"2024-11-21T08:43:50.778103Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optimizer = optim.AdamW(model.parameters(), lr=0.00001) #, weight_decay=0.5\ncriterion = nn.SmoothL1Loss()\nlr_scheduler = optim.lr_scheduler.ReduceLROnPlateau(optimizer, mode='min', factor=0.5, patience=3, verbose=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:50.780458Z","iopub.execute_input":"2024-11-21T08:43:50.781119Z","iopub.status.idle":"2024-11-21T08:43:50.834753Z","shell.execute_reply.started":"2024-11-21T08:43:50.781059Z","shell.execute_reply":"2024-11-21T08:43:50.833866Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def train_one_epoch():\n    model.train()  # Set model to training mode\n    train_loss = 0\n    scaler = GradScaler()  # Initialize GradScaler once before the loop\n    pbar = tqdm(train_data_loader)  # Track progress with tqdm\n\n    for batch_idx, data in enumerate(pbar):\n        inputs = data[\"image\"]\n        labels = data[\"labels\"].view(-1, 1)  # Reshape if necessary, ensure it's 2D for regression\n        inputs = inputs.to(device, dtype=torch.float)\n        labels = labels.to(device, dtype=torch.float)\n\n        optimizer.zero_grad()\n        with autocast():  # Mixed precision training\n            outputs = model(inputs)\n            loss = criterion(outputs, labels)\n        \n        scaler.scale(loss).backward()  # Backpropagate loss with scaling\n        scaler.step(optimizer)  # Update optimizer\n        scaler.update()  # Update scaler\n\n        train_loss += loss.item()\n\n        # Update progress bar description with loss\n        pbar.set_description(f\"Train loss: {train_loss / len(train_data_loader):.5f}\")\n    \n    return train_loss / len(train_data_loader)  # Return average loss for the epoch","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:50.835863Z","iopub.execute_input":"2024-11-21T08:43:50.836303Z","iopub.status.idle":"2024-11-21T08:43:50.883504Z","shell.execute_reply.started":"2024-11-21T08:43:50.836262Z","shell.execute_reply":"2024-11-21T08:43:50.882646Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def evaluate():\n    model.eval()  # Set model to evaluation mode\n    val_loss = 0\n    pbar = tqdm(val_data_loader)  # Track progress with tqdm for validation data\n    with torch.no_grad():  # Disable gradient calculation for evaluation\n        for batch_idx, data in enumerate(pbar):\n            inputs = data[\"image\"]\n            labels = data[\"labels\"].view(-1, 1)  # Reshape if necessary, ensure it's 2D for regression\n            inputs = inputs.to(device, dtype=torch.float)\n            labels = labels.to(device, dtype=torch.float)\n\n            with autocast():  # Mixed precision evaluation\n                outputs = model(inputs)\n                loss = criterion(outputs, labels)\n            \n            val_loss += loss.item()  # Accumulate the loss\n            \n            # Update progress bar description with average loss\n            pbar.set_description(f\"Val loss: {val_loss / len(val_data_loader):.5f}\")\n    \n    # Return the average validation loss for the entire dataset\n    return val_loss / len(val_data_loader)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:50.884411Z","iopub.execute_input":"2024-11-21T08:43:50.884683Z","iopub.status.idle":"2024-11-21T08:43:50.930357Z","shell.execute_reply.started":"2024-11-21T08:43:50.884658Z","shell.execute_reply":"2024-11-21T08:43:50.929558Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"save_path = r'/kaggle/working/model/'\nos.makedirs(save_path, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:50.955347Z","iopub.execute_input":"2024-11-21T08:43:50.955577Z","iopub.status.idle":"2024-11-21T08:43:50.998124Z","shell.execute_reply.started":"2024-11-21T08:43:50.955553Z","shell.execute_reply":"2024-11-21T08:43:50.997255Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# model = model.load_state_dict(torch.load(r'D:\\SuperAISS3\\Mini-Hack\\thai-handwritten-characters-recognition\\train\\efficientnetv2_s.pth'))\nsince = time.time()\nnum_epochs = 15\nbest_loss = 100\n# Initialize lists to store losses over num_epochs\ntrain_losses = []\nval_losses = []\nfor epoch in range(num_epochs):\n    print(f'Epoch {epoch + 1}/{num_epochs}')\n    train_loss = train_one_epoch()\n    val_loss = evaluate()\n\n    train_losses.append(train_loss)  # Store average training loss for the epoch\n    val_losses.append(val_loss)\n    \n    lr_scheduler.step(val_loss)\n    if val_loss < best_loss:\n        best_loss = val_loss\n        torch.save(model.state_dict(), os.path.join(save_path, model_name + 'epoch_' + str(epoch+1) + '_' + str(round(val_loss, ndigits=2)) +'.pth'))\n        # torch.save({'epoch': epoch,\n        #             'model_state_dict': model.state_dict(),\n        #             'optimizer_state_dict': optimizer.state_dict(),\n        #             'loss': val_loss,\n        #             'acc': val_acc}, os.path.join(save_path, model_name + str(epoch) + str(val_loss) +'.pth'))\n        print('save model')\n    # if val_acc > best_acc:\n    #     best_acc = val_acc\n    #     torch.save(model.state_dict(), os.path.join(save_path, model_name + epoch, val_acc +'.pth'))\n    #     print('save model')\n    \ntime_elapsed = time.time() - since\nprint('Training complete in {:.0f}m {:.0f}s'.format(time_elapsed // 60, time_elapsed % 60))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:51.975821Z","iopub.execute_input":"2024-11-21T08:43:51.976615Z","iopub.status.idle":"2024-11-21T08:43:57.418650Z","shell.execute_reply.started":"2024-11-21T08:43:51.976580Z","shell.execute_reply":"2024-11-21T08:43:57.415825Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(len(train_losses), len(val_losses))  # Should both be the same","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.419315Z","iopub.status.idle":"2024-11-21T08:43:57.419587Z","shell.execute_reply.started":"2024-11-21T08:43:57.419452Z","shell.execute_reply":"2024-11-21T08:43:57.419466Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Train Losses:\", train_losses)\nprint(\"Val Losses:\", val_losses)\nprint(\"Length of Train Losses:\", len(train_losses))\nprint(\"Length of Validation Losses:\", len(val_losses))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.420751Z","iopub.status.idle":"2024-11-21T08:43:57.421201Z","shell.execute_reply.started":"2024-11-21T08:43:57.420954Z","shell.execute_reply":"2024-11-21T08:43:57.420975Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plotting\nplt.figure(figsize=(10, 6))\nplt.plot(range(1, len(train_losses)+1), train_losses, label='Training Loss', color='blue')\nplt.plot(range(1, len(val_losses)+1), val_losses, label='Validation Loss', color='red')\nplt.title('Training and Validation Loss')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid(True)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.422275Z","iopub.status.idle":"2024-11-21T08:43:57.422688Z","shell.execute_reply.started":"2024-11-21T08:43:57.422471Z","shell.execute_reply":"2024-11-21T08:43:57.422493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RetinopathyDatasetTest(Dataset):\n\n    def __init__(self, csv_file):\n        self.data = csv_file\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        img_name = os.path.join('../input/aptos2019-blindness-detection/test_images', self.data.loc[idx, 'id_code'] + '.png')\n        image = load_basic_albumentations(img_name)\n        return {'image': image}","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.423859Z","iopub.status.idle":"2024-11-21T08:43:57.424318Z","shell.execute_reply.started":"2024-11-21T08:43:57.424076Z","shell.execute_reply":"2024-11-21T08:43:57.424117Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df = pd.read_csv('/kaggle/input/aptos2019-blindness-detection/test.csv')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.426201Z","iopub.status.idle":"2024-11-21T08:43:57.426636Z","shell.execute_reply.started":"2024-11-21T08:43:57.426407Z","shell.execute_reply":"2024-11-21T08:43:57.426430Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.427677Z","iopub.status.idle":"2024-11-21T08:43:57.428163Z","shell.execute_reply.started":"2024-11-21T08:43:57.427886Z","shell.execute_reply":"2024-11-21T08:43:57.427909Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"load_basic_albumentations('../input/aptos2019-blindness-detection/test_images/0005cfc8afb6' + '.png')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.429215Z","iopub.status.idle":"2024-11-21T08:43:57.429637Z","shell.execute_reply.started":"2024-11-21T08:43:57.429416Z","shell.execute_reply":"2024-11-21T08:43:57.429439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_dataset = RetinopathyDatasetTest(csv_file=test_df)\ntest_loader = torch.utils.data.DataLoader(test_dataset, batch_size=16, shuffle=False, num_workers=4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.430947Z","iopub.status.idle":"2024-11-21T08:43:57.431406Z","shell.execute_reply.started":"2024-11-21T08:43:57.431183Z","shell.execute_reply":"2024-11-21T08:43:57.431206Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.load_state_dict(torch.load(r'/kaggle/working/model/tf_efficientnetv2_m_in21kepoch_2_0.24.pth'))\nmodel.eval()\n# preds = []\n# final_preds = []\n\ntest_preds = np.zeros((len(test_dataset), 1))  # Initialize predictions array with the correct size\nfor batch_idx, data in enumerate(tqdm(test_loader)):\n    with torch.no_grad():\n        x_batch = data[\"image\"].to(device, dtype=torch.float32)\n        outputs = model(x_batch)\n        \n        # Get the number of samples in the current batch (it could be less than 16 for the last batch)\n        batch_size = outputs.size(0)\n        \n        # Calculate start and end indices\n        start_idx = batch_idx * 16\n        end_idx = start_idx + batch_size\n        \n        # Assign the predictions for this batch to the corresponding slice of test_preds\n        test_preds[start_idx:end_idx] = outputs.detach().cpu().squeeze().numpy().ravel().reshape(-1, 1)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.432398Z","iopub.status.idle":"2024-11-21T08:43:57.432814Z","shell.execute_reply.started":"2024-11-21T08:43:57.432596Z","shell.execute_reply":"2024-11-21T08:43:57.432617Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_preds[:10]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.434320Z","iopub.status.idle":"2024-11-21T08:43:57.434589Z","shell.execute_reply.started":"2024-11-21T08:43:57.434456Z","shell.execute_reply":"2024-11-21T08:43:57.434470Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn import metrics\nimport scipy as sp\nfrom functools import partial","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.436265Z","iopub.status.idle":"2024-11-21T08:43:57.436692Z","shell.execute_reply.started":"2024-11-21T08:43:57.436469Z","shell.execute_reply":"2024-11-21T08:43:57.436493Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class OptimizedRounder(object):\n    def __init__(self):\n        self.coef_ = 0\n\n    def _smooth_l1_loss(self, coef, X, y):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n\n        # Calculate SmoothL1Loss\n        loss_fn = nn.SmoothL1Loss()\n        y_tensor = torch.tensor(y, dtype=torch.float32)\n        X_p_tensor = torch.tensor(X_p, dtype=torch.float32)\n        loss = loss_fn(X_p_tensor, y_tensor)\n        \n        return loss.item()  # Return as a scalar value\n\n    def fit(self, X, y):\n        # Using the SmoothL1Loss function as the loss metric\n        loss_partial = partial(self._smooth_l1_loss, X=X, y=y)\n        initial_coef = [0.5, 1.5, 2.5, 3.5]\n        # Minimize the SmoothL1Loss\n        self.coef_ = sp.optimize.minimize(loss_partial, initial_coef, method='nelder-mead')\n\n    def predict(self, X, coef):\n        X_p = np.copy(X)\n        for i, pred in enumerate(X_p):\n            if pred < coef[0]:\n                X_p[i] = 0\n            elif pred >= coef[0] and pred < coef[1]:\n                X_p[i] = 1\n            elif pred >= coef[1] and pred < coef[2]:\n                X_p[i] = 2\n            elif pred >= coef[2] and pred < coef[3]:\n                X_p[i] = 3\n            else:\n                X_p[i] = 4\n        return X_p\n\n    def coefficients(self):\n        return self.coef_['x']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.437825Z","iopub.status.idle":"2024-11-21T08:43:57.438309Z","shell.execute_reply.started":"2024-11-21T08:43:57.438035Z","shell.execute_reply":"2024-11-21T08:43:57.438057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# class OptimizedRounder(object):\n#     def __init__(self):\n#         self.coef_ = 0\n\n#     def _kappa_loss(self, coef, X, y):\n#         X_p = np.copy(X)\n#         for i, pred in enumerate(X_p):\n#             if pred < coef[0]:\n#                 X_p[i] = 0\n#             elif pred >= coef[0] and pred < coef[1]:\n#                 X_p[i] = 1\n#             elif pred >= coef[1] and pred < coef[2]:\n#                 X_p[i] = 2\n#             elif pred >= coef[2] and pred < coef[3]:\n#                 X_p[i] = 3\n#             else:\n#                 X_p[i] = 4\n\n#         ll = metrics.cohen_kappa_score(y, X_p, weights='quadratic')\n#         return -ll\n\n#     def fit(self, X, y):\n#         loss_partial = partial(self._kappa_loss, X=X, y=y)\n#         initial_coef = [0.5, 1.5, 2.5, 3.5]\n#         self.coef_ = sp.optimize.minimize(loss_partial, initial_coef, method='nelder-mead')\n\n#     def predict(self, X, coef):\n#         X_p = np.copy(X)\n#         for i, pred in enumerate(X_p):\n#             if pred < coef[0]:\n#                 X_p[i] = 0\n#             elif pred >= coef[0] and pred < coef[1]:\n#                 X_p[i] = 1\n#             elif pred >= coef[1] and pred < coef[2]:\n#                 X_p[i] = 2\n#             elif pred >= coef[2] and pred < coef[3]:\n#                 X_p[i] = 3\n#             else:\n#                 X_p[i] = 4\n#         return X_p\n\n#     def coefficients(self):\n#         return self.coef_['x']","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.439524Z","iopub.status.idle":"2024-11-21T08:43:57.439944Z","shell.execute_reply.started":"2024-11-21T08:43:57.439727Z","shell.execute_reply":"2024-11-21T08:43:57.439749Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class RetinopathyDatasetVal(Dataset):\n\n    def __init__(self, csv_file):\n        self.data = csv_file\n        self.data = self.data.reset_index(drop=True)\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        img_name = os.path.join('../input/aptos2019-blindness-detection/train_images', self.data.loc[idx, 'id_code'] + '.png')\n        id = self.data.loc[idx, 'id_code']\n        image = load_basic_albumentations(img_name)\n        label = torch.tensor(self.data.loc[idx, 'diagnosis'])\n        return {'id_code': id,\n            'image': image,\n                'labels': label\n                }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.441461Z","iopub.status.idle":"2024-11-21T08:43:57.441757Z","shell.execute_reply.started":"2024-11-21T08:43:57.441617Z","shell.execute_reply":"2024-11-21T08:43:57.441632Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_data_set = RetinopathyDatasetVal(val_df)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.442888Z","iopub.status.idle":"2024-11-21T08:43:57.443221Z","shell.execute_reply.started":"2024-11-21T08:43:57.443044Z","shell.execute_reply":"2024-11-21T08:43:57.443060Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_dataloader = test_loader = torch.utils.data.DataLoader(val_data_set, batch_size=16, shuffle=False, num_workers=4)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.569366Z","iopub.execute_input":"2024-11-21T08:43:57.570174Z","iopub.status.idle":"2024-11-21T08:43:57.637662Z","shell.execute_reply.started":"2024-11-21T08:43:57.570128Z","shell.execute_reply":"2024-11-21T08:43:57.636537Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_preds = np.zeros((len(val_data_set), 1))  # Initialize predictions array with the correct size","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:57.795605Z","iopub.execute_input":"2024-11-21T08:43:57.796381Z","iopub.status.idle":"2024-11-21T08:43:57.860855Z","shell.execute_reply.started":"2024-11-21T08:43:57.796347Z","shell.execute_reply":"2024-11-21T08:43:57.859805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred_ids = []\nreal_labels = []","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:58.382798Z","iopub.execute_input":"2024-11-21T08:43:58.383436Z","iopub.status.idle":"2024-11-21T08:43:58.435447Z","shell.execute_reply.started":"2024-11-21T08:43:58.383401Z","shell.execute_reply":"2024-11-21T08:43:58.434432Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Iterate through the DataLoader\nfor batch_idx, data in enumerate(tqdm(val_dataloader)):\n    with torch.no_grad():\n        x_batch = data[\"image\"].to(device, dtype=torch.float32)\n        outputs = model(x_batch)\n        \n        # Get the number of samples in the current batch (it could be less than 16 for the last batch)\n        batch_size = outputs.size(0)\n        \n        # Calculate start and end indices\n        start_idx = batch_idx * 16\n        end_idx = start_idx + batch_size\n        \n        # Assign the predictions for this batch to the corresponding slice of test_preds\n        val_preds[start_idx:end_idx] = outputs.detach().cpu().squeeze().numpy().ravel().reshape(-1, 1)\n        \n        # Collect ids or any identifiers for each sample in the batch (assuming 'data' contains 'id')\n        pred_ids.extend(data[\"id_code\"])  # Or use the correct key that holds the image identifiers\n        real_labels.extend(data['labels'])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:58.837475Z","iopub.execute_input":"2024-11-21T08:43:58.837824Z","iopub.status.idle":"2024-11-21T08:43:59.396274Z","shell.execute_reply.started":"2024-11-21T08:43:58.837794Z","shell.execute_reply":"2024-11-21T08:43:59.395057Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a DataFrame from the predictions and ids\ndf_predictions = pd.DataFrame({\n    'id_code': pred_ids,  # Image identifiers\n    'real': real_labels,\n    'predictions': val_preds.flatten().squeeze()  # Flatten the predictions array\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:59.396894Z","iopub.status.idle":"2024-11-21T08:43:59.397285Z","shell.execute_reply.started":"2024-11-21T08:43:59.397055Z","shell.execute_reply":"2024-11-21T08:43:59.397071Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"targets = val_df['diagnosis']\ntargets = targets.to_numpy()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:59.567927Z","iopub.execute_input":"2024-11-21T08:43:59.568344Z","iopub.status.idle":"2024-11-21T08:43:59.615479Z","shell.execute_reply.started":"2024-11-21T08:43:59.568312Z","shell.execute_reply":"2024-11-21T08:43:59.614731Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_preds.max()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:59.757138Z","iopub.execute_input":"2024-11-21T08:43:59.757759Z","iopub.status.idle":"2024-11-21T08:43:59.821273Z","shell.execute_reply.started":"2024-11-21T08:43:59.757730Z","shell.execute_reply":"2024-11-21T08:43:59.820164Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict(X):\n    initial_coef = [0.5, 1.5, 2.5, 3.5]\n    X_p = np.copy(X)\n    for i, pred in enumerate(X_p):\n        if pred < initial_coef[0]:\n            X_p[i] = 0\n        elif pred >= initial_coef[0] and pred < initial_coef[1]:\n            X_p[i] = 1\n        elif pred >= initial_coef[1] and pred < initial_coef[2]:\n            X_p[i] = 2\n        elif pred >= initial_coef[2] and pred < initial_coef[3]:\n            X_p[i] = 3\n        else:\n            X_p[i] = 4\n    return X_p","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-21T08:43:59.951140Z","iopub.execute_input":"2024-11-21T08:43:59.951470Z","iopub.status.idle":"2024-11-21T08:43:59.996903Z","shell.execute_reply.started":"2024-11-21T08:43:59.951442Z","shell.execute_reply":"2024-11-21T08:43:59.996028Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"result = predict(val_preds)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:33.712952Z","iopub.execute_input":"2024-11-20T15:49:33.713395Z","iopub.status.idle":"2024-11-20T15:49:33.761094Z","shell.execute_reply.started":"2024-11-20T15:49:33.713356Z","shell.execute_reply":"2024-11-20T15:49:33.760127Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"optR = OptimizedRounder()\noptR.fit(val_preds, targets)\ncoefficients = optR.coefficients()\nvalid_predictions = optR.predict(val_preds, coefficients)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:33.762170Z","iopub.execute_input":"2024-11-20T15:49:33.762468Z","iopub.status.idle":"2024-11-20T15:49:34.061978Z","shell.execute_reply.started":"2024-11-20T15:49:33.762420Z","shell.execute_reply":"2024-11-20T15:49:34.061033Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"coefficients","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:34.063150Z","iopub.execute_input":"2024-11-20T15:49:34.063434Z","iopub.status.idle":"2024-11-20T15:49:34.111558Z","shell.execute_reply.started":"2024-11-20T15:49:34.063407Z","shell.execute_reply":"2024-11-20T15:49:34.110577Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_predictions['predictions'] = result.astype(int)\n# Map the tensors to integers (you can use rounding or other methods)\ndf_predictions['real'] = df_predictions['real'].apply(lambda x: int(x.item()))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:34.112779Z","iopub.execute_input":"2024-11-20T15:49:34.113206Z","iopub.status.idle":"2024-11-20T15:49:34.163727Z","shell.execute_reply.started":"2024-11-20T15:49:34.113163Z","shell.execute_reply":"2024-11-20T15:49:34.162762Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_predictions","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:34.165039Z","iopub.execute_input":"2024-11-20T15:49:34.165381Z","iopub.status.idle":"2024-11-20T15:49:34.218299Z","shell.execute_reply.started":"2024-11-20T15:49:34.165352Z","shell.execute_reply":"2024-11-20T15:49:34.217208Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_predictions['predictions'].unique()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:34.219544Z","iopub.execute_input":"2024-11-20T15:49:34.220442Z","iopub.status.idle":"2024-11-20T15:49:34.269997Z","shell.execute_reply.started":"2024-11-20T15:49:34.220404Z","shell.execute_reply":"2024-11-20T15:49:34.269147Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Compare the two columns\ndf_predictions['difference'] = df_predictions['real'] != df_predictions['predictions']  # This will be True where values differ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:34.271164Z","iopub.execute_input":"2024-11-20T15:49:34.272014Z","iopub.status.idle":"2024-11-20T15:49:34.319605Z","shell.execute_reply.started":"2024-11-20T15:49:34.271964Z","shell.execute_reply":"2024-11-20T15:49:34.318722Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Alternatively, you can count the number of differences\nnum_differences = df_predictions['difference'].sum()\nprint(f'Number of differences: {num_differences}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:34.320757Z","iopub.execute_input":"2024-11-20T15:49:34.321022Z","iopub.status.idle":"2024-11-20T15:49:34.369950Z","shell.execute_reply.started":"2024-11-20T15:49:34.320997Z","shell.execute_reply":"2024-11-20T15:49:34.368897Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Rows where 'real' and 'predicted' differ\ndifference_rows = df_predictions[df_predictions['difference']]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:34.371329Z","iopub.execute_input":"2024-11-20T15:49:34.372138Z","iopub.status.idle":"2024-11-20T15:49:34.418203Z","shell.execute_reply.started":"2024-11-20T15:49:34.372096Z","shell.execute_reply":"2024-11-20T15:49:34.417136Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"difference_rows","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:34.419474Z","iopub.execute_input":"2024-11-20T15:49:34.419801Z","iopub.status.idle":"2024-11-20T15:49:34.472912Z","shell.execute_reply.started":"2024-11-20T15:49:34.419770Z","shell.execute_reply":"2024-11-20T15:49:34.471957Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the DataFrame to a CSV file\ndf_predictions.to_csv('predictions.csv', index=False)\n\nprint(\"Predictions saved to 'predictions.csv'.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:34.474215Z","iopub.execute_input":"2024-11-20T15:49:34.474486Z","iopub.status.idle":"2024-11-20T15:49:34.529975Z","shell.execute_reply.started":"2024-11-20T15:49:34.474461Z","shell.execute_reply":"2024-11-20T15:49:34.529011Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nfrom sklearn.metrics import confusion_matrix, accuracy_score, classification_report\nimport seaborn as sns\nimport matplotlib.pyplot as plt\n\ndef evaluate_predictions(df, true_col='real', pred_col='predictions'):\n    \"\"\"\n    Evaluate model predictions by calculating confusion matrix, accuracy, and classification report.\n    \n    Parameters:\n    - df: pandas DataFrame containing the true and predicted values\n    - true_col: name of the column with true values (default 'real')\n    - pred_col: name of the column with predicted values (default 'predicted')\n    \n    Returns:\n    - A tuple containing the confusion matrix, accuracy, and classification report\n    \"\"\"\n    # Extract true and predicted values\n    y_true = df[true_col]\n    y_pred = df[pred_col]\n    \n    # Calculate confusion matrix\n    cm = confusion_matrix(y_true, y_pred)\n    \n    # Calculate accuracy\n    accuracy = accuracy_score(y_true, y_pred)\n    \n    # Generate classification report\n    class_report = classification_report(y_true, y_pred)\n    \n    # Print confusion matrix, accuracy, and classification report\n    print(f\"Confusion Matrix:\\n{cm}\")\n    print(f\"Accuracy: {accuracy:.4f}\")\n    print(f\"Classification Report:\\n{class_report}\")\n    \n    # Plot the confusion matrix using seaborn heatmap\n    plt.figure(figsize=(8, 6))\n    sns.heatmap(cm, annot=True, fmt='d', cmap='Blues', xticklabels=True, yticklabels=True)\n    plt.title('Confusion Matrix')\n    plt.xlabel('Predicted Labels')\n    plt.ylabel('True Labels')\n    plt.show()\n\n    return cm, accuracy, class_report\n\n# Example usage\n# Assuming df_predictions is the DataFrame with 'real' and 'predicted' columns\n# cm, accuracy, class_report = evaluate_predictions(df_predictions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:34.531298Z","iopub.execute_input":"2024-11-20T15:49:34.531580Z","iopub.status.idle":"2024-11-20T15:49:34.796453Z","shell.execute_reply.started":"2024-11-20T15:49:34.531553Z","shell.execute_reply":"2024-11-20T15:49:34.795759Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"cm, accuracy, class_report = evaluate_predictions(df_predictions)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:34.799668Z","iopub.execute_input":"2024-11-20T15:49:34.800097Z","iopub.status.idle":"2024-11-20T15:49:35.174110Z","shell.execute_reply.started":"2024-11-20T15:49:34.800050Z","shell.execute_reply":"2024-11-20T15:49:35.173215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Map the 'real' and 'predictions' columns to class names\ndf_predictions['real_class'] = df_predictions['real'].map(label_map)\ndf_predictions['predicted_class'] = df_predictions['predictions'].map(label_map)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:35.175198Z","iopub.execute_input":"2024-11-20T15:49:35.175473Z","iopub.status.idle":"2024-11-20T15:49:35.224218Z","shell.execute_reply.started":"2024-11-20T15:49:35.175447Z","shell.execute_reply":"2024-11-20T15:49:35.223455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_predictions['real'].hist(figsize = (10, 5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:35.225400Z","iopub.execute_input":"2024-11-20T15:49:35.225690Z","iopub.status.idle":"2024-11-20T15:49:35.534462Z","shell.execute_reply.started":"2024-11-20T15:49:35.225655Z","shell.execute_reply":"2024-11-20T15:49:35.533663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_predictions['predictions'].hist(figsize = (10, 5))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-20T15:49:35.535859Z","iopub.execute_input":"2024-11-20T15:49:35.536304Z","iopub.status.idle":"2024-11-20T15:49:35.841974Z","shell.execute_reply.started":"2024-11-20T15:49:35.536263Z","shell.execute_reply":"2024-11-20T15:49:35.841024Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### ","metadata":{}}]}