{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":63056,"databundleVersionId":9094797,"sourceType":"competition"},{"sourceId":1225697,"sourceType":"datasetVersion","datasetId":701123},{"sourceId":1253590,"sourceType":"datasetVersion","datasetId":720563},{"sourceId":1322494,"sourceType":"datasetVersion","datasetId":688574},{"sourceId":1339680,"sourceType":"datasetVersion","datasetId":756214},{"sourceId":1339694,"sourceType":"datasetVersion","datasetId":756315},{"sourceId":1353811,"sourceType":"datasetVersion","datasetId":762203},{"sourceId":1444814,"sourceType":"datasetVersion","datasetId":846815}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#### Versions:\n* v9: ColorJitter transformation added **[0.896]**\n* v10: Changed the dataset to [this one](https://www.kaggle.com/shonenkov/melanoma-merged-external-data-512x512-jpeg) with external data. **[0.894]**\n* v11: Switched to [another dataset](https://www.kaggle.com/nroman/melanoma-external-malignant-256/) which I've created by myself. Also switched from StratifiedKFold to GroupKFold **[0.916]**\n* v12: Switched to efficientnet-b1 **[0.919]**\n* v13: Using meta featues: sex and age **[0.918]**\n* v14: anatom_site_general_challenge meta feature added as one-hot encoded matrix **[0.923]**\n* v16: Fixed OOF - now it contains only data from original training dataset, without extarnal data. Also switched back to StratifiedKFold. Added DrawHair augmentation. **[0.909]**\n* v18: Too many things were changed at the same time. All experiments should have only one small change each, so it would be easy to understand how changes affect the result. Said that I rolled back everything, keeping only OOF fix, to make sure it work.\n* v19: Added 'Hair' augmentation. OOF rework posponed untill the best time, since there is some bug in my code for it. **[0.925]**\n* v20: Advanced Hair Augmentation technique used. Read more about it here: https://www.kaggle.com/c/siim-isic-melanoma-classification/discussion/159176 **[0.923]**\n* v21: Microscope augmentation added instead of Cutout. Read more here: https://www.kaggle.com/c/siim-isic-melanoma-classification/discussion/159476 **[0.914]**\n* v22: Changed the dataset to [this one](https://www.kaggle.com/cdeotte/jpeg-melanoma-256x256) by Chris Deotte. More info [here](https://www.kaggle.com/c/siim-isic-melanoma-classification/discussion/165526) **[0.900]**\n* v23: All the same as v22 but effnet-b0 instead of b1 and more epochs per fold. **[0.895]**\n* v24: effnet-b01 and more epochs. **[0.9092]**\n* v25: Fixed a mistake in a way of filling preds. See [this comment](https://www.kaggle.com/nroman/melanoma-pytorch-starter-efficientnet/comments?scriptVersionId=39125585#913846). **[0.9016]**\n* v26: Fix for another mistake. This time with a way of averaging TTA. See [this comment](https://www.kaggle.com/nroman/melanoma-pytorch-starter-efficientnet/comments#955916) **[0.915]**\n* v27: Back to [my dataset](https://www.kaggle.com/nroman/melanoma-external-malignant-256/)","metadata":{}},{"cell_type":"code","source":"!pip install -q efficientnet_pytorch geffnet # torchvision==0.1.6\n# !pip install -q --upgrade torch","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2024-07-27T20:29:13.198662Z","iopub.execute_input":"2024-07-27T20:29:13.199287Z","iopub.status.idle":"2024-07-27T20:29:25.473473Z","shell.execute_reply.started":"2024-07-27T20:29:13.199251Z","shell.execute_reply":"2024-07-27T20:29:25.472384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport cv2\nimport math\nimport copy\nimport time\nimport random\nimport glob\nfrom pathlib import Path\n\n# For data manipulation\nimport numpy as np\nimport pandas as pd\n\n# Pytorch Imports\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nimport torch.nn.functional as F\nfrom torch.optim import lr_scheduler\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.cuda import amp\nimport torchvision\n# from torcheval.metrics.functional import binary_auroc\n# from torchmetrics import AUROC\nfrom sklearn.metrics import f1_score, roc_auc_score, average_precision_score\n\n# Utils\nimport joblib\nfrom tqdm import tqdm\nfrom collections import defaultdict\n\n# Sklearn Imports\nfrom sklearn.preprocessing import LabelEncoder\n# from sklearn.model_selection import StratifiedKFold, StratifiedGroupKFold \n\n# For Image Models\n# import timm\n\n# Albumentations for augmentations\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\n# For colored terminal text\n# from colorama import Fore, Back, Style\n# b_ = Fore.BLUE\n# sr_ = Style.RESET_ALL\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# For descriptive error messages\nos.environ['CUDA_LAUNCH_BLOCKING'] = \"1\"\n\n# Plot package\nimport PIL\nfrom matplotlib import pyplot as plt\nfrom torch.utils.data import DataLoader, WeightedRandomSampler\n","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:25.475843Z","iopub.execute_input":"2024-07-27T20:29:25.476520Z","iopub.status.idle":"2024-07-27T20:29:29.275850Z","shell.execute_reply.started":"2024-07-27T20:29:25.476478Z","shell.execute_reply":"2024-07-27T20:29:29.274836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nimport torchvision\nimport torch.nn.functional as F\nimport torch.nn as nn\n# import torchtoolbox.transform as transforms\nfrom torchvision import transforms\nfrom torch.utils.data import Dataset, DataLoader, Subset\nfrom torch.optim.lr_scheduler import ReduceLROnPlateau\nfrom sklearn.metrics import accuracy_score, roc_auc_score\nfrom sklearn.model_selection import StratifiedKFold, GroupKFold, KFold\nimport pandas as pd\nimport numpy as np\nimport gc\nimport os\nimport cv2\nimport time\nimport datetime\nimport warnings\nimport random\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom efficientnet_pytorch import EfficientNet\nimport albumentations as A\nimport geffnet\nimport glob\nfrom pathlib import Path\n# Utils\nimport joblib\nfrom tqdm import tqdm\nfrom collections import defaultdict\n%matplotlib inline","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-07-27T20:29:29.277023Z","iopub.execute_input":"2024-07-27T20:29:29.277449Z","iopub.status.idle":"2024-07-27T20:29:29.357172Z","shell.execute_reply.started":"2024-07-27T20:29:29.277424Z","shell.execute_reply":"2024-07-27T20:29:29.356417Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pdb\nfrom torch.cuda.amp import autocast, GradScaler","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:29.359559Z","iopub.execute_input":"2024-07-27T20:29:29.360032Z","iopub.status.idle":"2024-07-27T20:29:29.364367Z","shell.execute_reply.started":"2024-07-27T20:29:29.359998Z","shell.execute_reply":"2024-07-27T20:29:29.363402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CONFIG = {\n    \"seed\": 42,\n    \"epochs\": 12, # 42, ~MAX 20 hours of training\n    \"img_size\": 512,\n    \"model_name\": \"efficientnet_b3\",\n    # \"model_name\": \"densenet121.tv_in1k\",\n#     \"checkpoint_path\" : \"/kaggle/input/tf-efficientnet/pytorch/tf-efficientnet-b3/1/tf_efficientnet_b3_aa-84b4657e.pth\",\n    \"train_batch_size\": 16,\n    \"valid_batch_size\": 64,\n    \"learning_rate\": 5e-5,\n    \"scheduler\": 'CosineAnnealingLR',\n    \"min_lr\": 5e-7,\n    \"T_max\": 12,\n    \"weight_decay\": 1e-6,\n    \"fold\" : 0,\n    \"n_fold\": 5,\n    \"n_accumulate\": 1,\n    \"device\": torch.device(\"cuda:0\" if torch.cuda.is_available() else \"cpu\"),\n    'criterion': 'BCEWithLogitsLoss'\n}","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:29.365352Z","iopub.execute_input":"2024-07-27T20:29:29.365628Z","iopub.status.idle":"2024-07-27T20:29:29.407654Z","shell.execute_reply.started":"2024-07-27T20:29:29.365580Z","shell.execute_reply":"2024-07-27T20:29:29.406588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"warnings.simplefilter('ignore')\ndef seed_everything(seed):\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\ndef flush():\n    gc.collect()\n    if torch.cuda.is_available():\n        torch.cuda.empty_cache()\n        torch.cuda.reset_peak_memory_stats()\nseed_everything(CONFIG['seed'])\n# seed_everything(42)\nflush()","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:29.408886Z","iopub.execute_input":"2024-07-27T20:29:29.409181Z","iopub.status.idle":"2024-07-27T20:29:29.587318Z","shell.execute_reply.started":"2024-07-27T20:29:29.409146Z","shell.execute_reply":"2024-07-27T20:29:29.586524Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nROOT_DIR = \"/kaggle/input/isic-2024-challenge\"\nTRAIN_DIR = f'{ROOT_DIR}/train-image/image'\ndef get_train_file_path(image_id):\n    return f\"{TRAIN_DIR}/{image_id}.jpg\"\n\ntrain_images_2024 = sorted(glob.glob(f\"{TRAIN_DIR}/*.jpg\"))","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:29.588414Z","iopub.execute_input":"2024-07-27T20:29:29.588712Z","iopub.status.idle":"2024-07-27T20:29:31.361343Z","shell.execute_reply.started":"2024-07-27T20:29:29.588687Z","shell.execute_reply":"2024-07-27T20:29:31.360313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Data augmentation","metadata":{}},{"cell_type":"code","source":"# Read past isic competition data\ncheck_path = lambda p: os.path.exists(p)\n\n# 2020\nDATA_PATH_2020 = Path(\"/kaggle/input/jpeg-melanoma-512x512/\")\nget_image_path_2020 = lambda p: os.path.join(f'{str(DATA_PATH_2020/\"train\")}/{p}.jpg')\ndf_train_2020 = pd.read_csv(DATA_PATH_2020/\"train.csv\")\ndf_positive_2020 = df_train_2020[df_train_2020[\"target\"] == 1].reset_index(drop=True)\ndf_positive_2020['file_path'] = df_positive_2020[\"image_name\"].apply(get_image_path_2020)\ndf_positive_2020['exists'] = df_positive_2020['file_path'].apply(check_path)\ndf_positive_2020 = df_positive_2020[df_positive_2020['exists'] == True].reset_index()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:31.362643Z","iopub.execute_input":"2024-07-27T20:29:31.363008Z","iopub.status.idle":"2024-07-27T20:29:31.839364Z","shell.execute_reply.started":"2024-07-27T20:29:31.362976Z","shell.execute_reply":"2024-07-27T20:29:31.838585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_2020.shape # (33126, 11)\ndf_positive_2020.shape # (584, 14)","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:31.840447Z","iopub.execute_input":"2024-07-27T20:29:31.840771Z","iopub.status.idle":"2024-07-27T20:29:31.847492Z","shell.execute_reply.started":"2024-07-27T20:29:31.840745Z","shell.execute_reply":"2024-07-27T20:29:31.846648Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 2019\nDATA_PATH_2019 = Path(\"/kaggle/input/jpeg-isic2019-512x512\")\nget_image_path_2019 = lambda p: os.path.join(f'{str(DATA_PATH_2019/\"train\")}/{p}.jpg')\ndf_train_2019 = pd.read_csv(DATA_PATH_2019/\"train.csv\")\ndf_positive_2019 = df_train_2019[df_train_2019[\"target\"] == 1].reset_index(drop=True)\ndf_positive_2019['file_path'] = df_positive_2019[\"image_name\"].apply(get_image_path_2019)\ndf_positive_2019['exists'] = df_positive_2019['file_path'].apply(check_path)\ndf_positive_2019 = df_positive_2019[df_positive_2019['exists'] == True].reset_index()","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:31.852090Z","iopub.execute_input":"2024-07-27T20:29:31.852467Z","iopub.status.idle":"2024-07-27T20:29:34.662186Z","shell.execute_reply.started":"2024-07-27T20:29:31.852437Z","shell.execute_reply":"2024-07-27T20:29:34.660975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_2019.shape # (25331, 11)\ndf_positive_2019.shape # (4522, 14)","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:34.663652Z","iopub.execute_input":"2024-07-27T20:29:34.664054Z","iopub.status.idle":"2024-07-27T20:29:34.670481Z","shell.execute_reply.started":"2024-07-27T20:29:34.664017Z","shell.execute_reply":"2024-07-27T20:29:34.669632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"past_df_positive = pd.concat([df_positive_2019, df_positive_2020], axis=0)\npast_df_positive.head(3)\nprint(f'Number of positive cases: {past_df_positive.shape[0]}')","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:34.671642Z","iopub.execute_input":"2024-07-27T20:29:34.671927Z","iopub.status.idle":"2024-07-27T20:29:34.681879Z","shell.execute_reply.started":"2024-07-27T20:29:34.671904Z","shell.execute_reply":"2024-07-27T20:29:34.680972Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"past_df_positive.head(2)","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:34.683034Z","iopub.execute_input":"2024-07-27T20:29:34.683786Z","iopub.status.idle":"2024-07-27T20:29:34.701767Z","shell.execute_reply.started":"2024-07-27T20:29:34.683746Z","shell.execute_reply":"2024-07-27T20:29:34.700923Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"past_df_positive.rename(columns={'image_name': 'isic_id'}, inplace=True)","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:34.702748Z","iopub.execute_input":"2024-07-27T20:29:34.703040Z","iopub.status.idle":"2024-07-27T20:29:34.708578Z","shell.execute_reply.started":"2024-07-27T20:29:34.703015Z","shell.execute_reply":"2024-07-27T20:29:34.707694Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(f\"{ROOT_DIR}/train-metadata.csv\")\nprint(\"        df.size, # of positive cases\")\nprint(\"original>\", df.shape[0], df.target.sum())\ndf_2024 = df.copy()\nselected_cols = ['isic_id', 'target', 'file_path', 'patient_id']\n# Positive cases\ndf_positive = df[df[\"target\"] == 1].reset_index(drop=True)\ndf_positive['file_path'] = df_positive['isic_id'].apply(get_train_file_path)\ndf_positive = pd.concat([df_positive[selected_cols], past_df_positive[selected_cols]], axis = 0).reset_index(drop=True)\n# Negative cases\ndf_negative = df[df[\"target\"] == 0].reset_index(drop=True)\ndf_negative = df_negative.iloc[:df_positive.shape[0]*5, :] # positive:negative = 1:5\ndf_negative['file_path'] = df_negative['isic_id'].apply(get_train_file_path)\ndf_negative = df_negative[selected_cols]\ndf_negative = df_negative[ df_negative[\"file_path\"].isin(train_images_2024) ]\ndf = pd.concat([df_positive, df_negative]).reset_index(drop=True)\nprint(\"filtered>\", df.shape[0], df.target.sum())\n\ndf.head(3)","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:34.709759Z","iopub.execute_input":"2024-07-27T20:29:34.710179Z","iopub.status.idle":"2024-07-27T20:29:40.288027Z","shell.execute_reply.started":"2024-07-27T20:29:34.710143Z","shell.execute_reply":"2024-07-27T20:29:40.287056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_negative = df.shape[0] - df.target.sum()\nnum_positive = df.target.sum()\nCONFIG['pos_weights'] = torch.tensor([num_negative / num_positive]).to(CONFIG['device'])","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:40.289370Z","iopub.execute_input":"2024-07-27T20:29:40.289772Z","iopub.status.idle":"2024-07-27T20:29:40.373571Z","shell.execute_reply.started":"2024-07-27T20:29:40.289738Z","shell.execute_reply":"2024-07-27T20:29:40.372796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"CONFIG['pos_weights'] ","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:40.374695Z","iopub.execute_input":"2024-07-27T20:29:40.375007Z","iopub.status.idle":"2024-07-27T20:29:40.448096Z","shell.execute_reply.started":"2024-07-27T20:29:40.374981Z","shell.execute_reply":"2024-07-27T20:29:40.447155Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.subplot(141)\nplt.title(\"Total Data - 2024\")\ndf_2024['target'].value_counts().plot(kind='bar', figsize=(20,4))\nplt.subplot(142)\nplt.title(\"Data combined past positive cases\")\ndf['target'].value_counts().plot(kind='bar', figsize=(20,4))","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:40.449389Z","iopub.execute_input":"2024-07-27T20:29:40.449774Z","iopub.status.idle":"2024-07-27T20:29:40.914109Z","shell.execute_reply.started":"2024-07-27T20:29:40.449741Z","shell.execute_reply":"2024-07-27T20:29:40.913169Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import KFold\nkf = KFold(CONFIG['n_fold'], shuffle=True, random_state=42)\nfor fold, ( _, val_) in enumerate(kf.split(df)):\n    df.loc[val_ , \"kfold\"] = int(fold)","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:40.915449Z","iopub.execute_input":"2024-07-27T20:29:40.915758Z","iopub.status.idle":"2024-07-27T20:29:40.931261Z","shell.execute_reply.started":"2024-07-27T20:29:40.915732Z","shell.execute_reply":"2024-07-27T20:29:40.930529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class AdvancedHairAugmentation:\n    \"\"\"\n    Impose an image of a hair to the target image\n\n    Args:\n        hairs (int): maximum number of hairs to impose\n        hairs_folder (str): path to the folder with hairs images\n    \"\"\"\n\n    def __init__(self, hairs: int = 5, hairs_folder: str = \"\"):\n        self.hairs = hairs\n        self.hairs_folder = hairs_folder\n\n    def __call__(self, **kwargs):\n        \"\"\"\n        Args:\n            kwargs (dict): Dictionary containing the image.\n\n        Returns:\n            dict: Dictionary containing the augmented image.\n        \"\"\"\n        img = kwargs['image']  # Access the image from the keyword arguments\n        n_hairs = random.randint(0, self.hairs)\n        \n        if not n_hairs:\n            result = {'image': img}       \n            return result\n        \n        height, width, _ = img.shape  # target image width and height\n        hair_images = [im for im in os.listdir(self.hairs_folder) if 'png' in im]\n        \n        for _ in range(n_hairs):\n            hair = cv2.imread(os.path.join(self.hairs_folder, random.choice(hair_images)))\n            hair = cv2.flip(hair, random.choice([-1, 0, 1]))\n            hair = cv2.rotate(hair, random.choice([0, 1, 2]))\n\n            h_height, h_width, _ = hair.shape  # hair image width and height\n            roi_ho = random.randint(0, img.shape[0] - hair.shape[0])\n            roi_wo = random.randint(0, img.shape[1] - hair.shape[1])\n            roi = img[roi_ho:roi_ho + h_height, roi_wo:roi_wo + h_width]\n\n            # Creating a mask and inverse mask\n            img2gray = cv2.cvtColor(hair, cv2.COLOR_BGR2GRAY)\n            ret, mask = cv2.threshold(img2gray, 10, 255, cv2.THRESH_BINARY)\n            mask_inv = cv2.bitwise_not(mask)\n\n            # Now black-out the area of hair in ROI\n            img_bg = cv2.bitwise_and(roi, roi, mask=mask_inv)\n\n            # Take only region of hair from hair image.\n            hair_fg = cv2.bitwise_and(hair, hair, mask=mask)\n\n            # Put hair in ROI and modify the target image\n            dst = cv2.add(img_bg, hair_fg)\n\n            img[roi_ho:roi_ho + h_height, roi_wo:roi_wo + h_width] = dst\n        result = {'image': img}       \n        return result\n\n\n    def __repr__(self):\n        return f'{self.__class__.__name__}(hairs={self.hairs}, hairs_folder=\"{self.hairs_folder}\")'","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:40.932573Z","iopub.execute_input":"2024-07-27T20:29:40.933088Z","iopub.status.idle":"2024-07-27T20:29:40.947438Z","shell.execute_reply.started":"2024-07-27T20:29:40.933054Z","shell.execute_reply":"2024-07-27T20:29:40.946474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class DrawHair:\n    \"\"\"\n    Draw a random number of pseudo hairs\n\n    Args:\n        hairs (int): maximum number of hairs to draw\n        width (tuple): possible width of the hair in pixels\n    \"\"\"\n\n    def __init__(self, hairs:int = 4, width:tuple = (1, 2)):\n        self.hairs = hairs\n        self.width = width\n\n    def __call__(self, **kwargs):\n        \"\"\"\n        Args:\n            kwargs (dict): Dictionary containing the image.\n\n        Returns:\n            dict: Dictionary containing the augmented image.\n        \"\"\"\n        img = kwargs['image']  # Access the image from the keyword arguments\n        \n        if not self.hairs:\n            return {'image': img} \n        \n        width, height, _ = img.shape\n        \n        for _ in range(random.randint(0, self.hairs)):\n            # The origin point of the line will always be at the top half of the image\n            origin = (random.randint(0, width), random.randint(0, height // 2))\n            # The end of the line \n            end = (random.randint(0, width), random.randint(0, height))\n            color = (0, 0, 0)  # color of the hair. Black.\n            cv2.line(img, origin, end, color, random.randint(self.width[0], self.width[1]))\n        \n        result = {'image': img}       \n        return result\n\n    def __repr__(self):\n        return f'{self.__class__.__name__}(hairs={self.hairs}, width={self.width})'","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:40.948564Z","iopub.execute_input":"2024-07-27T20:29:40.948886Z","iopub.status.idle":"2024-07-27T20:29:40.959879Z","shell.execute_reply.started":"2024-07-27T20:29:40.948862Z","shell.execute_reply":"2024-07-27T20:29:40.958988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Microscope:\n    \"\"\"\n    Cutting out the edges around the center circle of the image\n    Imitating a picture, taken through the microscope\n\n    Args:\n        p (float): probability of applying an augmentation\n    \"\"\"\n\n    def __init__(self, p: float = 0.5):\n        self.p = p\n\n    def __call__(self, **kwargs):\n        \"\"\"\n        Args:\n            kwargs (dict): Dictionary containing the image.\n\n        Returns:\n            dict: Dictionary containing the augmented image.\n        \"\"\"\n        img = kwargs['image']  # Access the image from the keyword arguments\n        \n        if random.random() < self.p:\n            circle = cv2.circle((np.ones(img.shape) * 255).astype(np.uint8), # image placeholder\n                        (img.shape[0]//2, img.shape[1]//2), # center point of circle\n                        random.randint(img.shape[0]//2 - 3, img.shape[0]//2 + 15), # radius\n                        (0, 0, 0), # color\n                        -1)\n\n            mask = circle - 255\n            img = np.multiply(img, mask)\n        \n        result = {'image': img}       \n        return result\n\n    def __repr__(self):\n        return f'{self.__class__.__name__}(p={self.p})'","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:40.960978Z","iopub.execute_input":"2024-07-27T20:29:40.961321Z","iopub.status.idle":"2024-07-27T20:29:40.972067Z","shell.execute_reply.started":"2024-07-27T20:29:40.961296Z","shell.execute_reply":"2024-07-27T20:29:40.971065Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_transform = A.Compose([\n    A.Resize(512, 512, always_apply=True),\n    AdvancedHairAugmentation(hairs_folder='/kaggle/input/melanoma-hairs'),\n#     A.RandomResizedCrop(height=256, width=256, scale=(0.8, 1.0)),\n    A.HorizontalFlip(),\n    A.VerticalFlip(),\n    A.Transpose(p=0.5),\n    A.VerticalFlip(p=0.5),\n    A.HorizontalFlip(p=0.5),\n    A.RandomBrightnessContrast(brightness_limit=0.2, contrast_limit=0.2, p=0.75),\n    # A.RandomContrast(limit=0.2, p=0.75),\n    A.OneOf([\n        A.MotionBlur(blur_limit=3),\n        A.MedianBlur(blur_limit=3),\n        A.GaussianBlur(blur_limit=3),\n        A.GaussNoise(var_limit=(2.0, 10.0)),\n    ], p=0.7),\n\n    A.OneOf([\n        A.OpticalDistortion(distort_limit=0.5),\n        A.GridDistortion(num_steps=4, distort_limit=0.5),\n        A.ElasticTransform(alpha=1, sigma=50, alpha_affine=50),\n    ], p=0.7),\n\n    A.CLAHE(clip_limit=2.0, p=0.5),\n    A.HueSaturationValue(hue_shift_limit=5, sat_shift_limit=10, val_shift_limit=5, p=0.5),\n    A.ShiftScaleRotate(shift_limit=0.1, scale_limit=0.1, rotate_limit=10, border_mode=0, p=0.5),\n    # A.Resize(image_size, image_size),\n    A.CoarseDropout(max_holes=4, max_height=8, max_width=8, p=0.5),    \n    Microscope(p=0.5),\n    \n    A.Normalize(mean=[0.485, 0.456, 0.406],std=[0.229, 0.224, 0.225]),\n    A.pytorch.ToTensorV2(),\n#     A.Normalize()\n])\ntest_transform = A.Compose([\n    A.Resize(512, 512, always_apply=True),\n    A.Normalize(mean=[0.485, 0.456, 0.406],std=[0.229, 0.224, 0.225]),\n    A.pytorch.ToTensorV2(),\n])\n\ndata_transforms = {\n    'train' : train_transform,\n    'valid' : test_transform\n}","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:40.973272Z","iopub.execute_input":"2024-07-27T20:29:40.973793Z","iopub.status.idle":"2024-07-27T20:29:40.986668Z","shell.execute_reply.started":"2024-07-27T20:29:40.973766Z","shell.execute_reply":"2024-07-27T20:29:40.985787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def debug_transforms(transforms, image_path):\n    # Load a sample image\n    img = cv2.imread(image_path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)  # Convert to RGB for display\n\n    # Apply the transformations\n    transformed = transforms(image=img)  # Assuming transforms is an Albumentations pipeline\n\n    # Display the original and transformed images\n    plt.figure(figsize=(12, 6))\n    plt.subplot(1, 2, 1)\n    plt.title(\"Original Image\")\n    plt.imshow(img)\n    plt.axis('off')\n\n    plt.subplot(1, 2, 2)\n    plt.title(\"Transformed Image\")\n    plt.imshow(transformed['image'].permute(1, 2, 0).numpy() )\n    plt.axis('off')\n\n    plt.show()\n\n# Example usage\ndebug_transforms(train_transform, '/kaggle/input/jpeg-isic2019-512x512/train/ISIC_0000001.jpg')","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:40.987782Z","iopub.execute_input":"2024-07-27T20:29:40.988980Z","iopub.status.idle":"2024-07-27T20:29:41.751282Z","shell.execute_reply.started":"2024-07-27T20:29:40.988949Z","shell.execute_reply":"2024-07-27T20:29:41.750316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ISICDataset_for_Train(Dataset):\n    def __init__(self, df, transforms=train_transform):\n        self.df_positive = df[df[\"target\"] == 1].reset_index()\n        self.df_negative = df[df[\"target\"] == 0].reset_index()\n        self.file_names_positive = self.df_positive['file_path'].values\n        self.file_names_negative = self.df_negative['file_path'].values\n        self.targets_positive = self.df_positive['target'].values\n        self.targets_negative = self.df_negative['target'].values\n        self.transforms = transforms\n        \n    def __len__(self):\n        return len(self.df_positive) * 4\n    \n    def __getitem__(self, index):\n        if random.random() <= 0.25:\n            df = self.df_positive\n            file_names = self.file_names_positive\n            targets = self.targets_positive\n        else:\n            df = self.df_negative\n            file_names = self.file_names_negative\n            targets = self.targets_negative\n        index = index % df.shape[0]\n        \n        img_path = file_names[index]\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        target = targets[index]\n        \n        if self.transforms:\n            img = self.transforms(image=img)[\"image\"]\n            \n        return {\n            'image': img,\n            'target': target\n        }\n    \nclass ISICDataset(Dataset):\n    def __init__(self, df, transforms=None):\n        self.df = df\n        self.file_names = df['file_path'].values\n        self.targets = df['target'].values\n        self.transforms = transforms\n        \n    def __len__(self):\n        return len(self.df)\n    \n    def __getitem__(self, index):\n        img_path = self.file_names[index]\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        target = self.targets[index]\n        \n        if self.transforms:\n#             pdb.set_trace()\n            img = self.transforms(image=img)[\"image\"]\n            \n        return {\n            'image': img,\n            'target': target\n        }","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:41.752678Z","iopub.execute_input":"2024-07-27T20:29:41.753160Z","iopub.status.idle":"2024-07-27T20:29:41.769622Z","shell.execute_reply.started":"2024-07-27T20:29:41.753103Z","shell.execute_reply":"2024-07-27T20:29:41.768652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Swish(torch.autograd.Function):\n    @staticmethod\n    def forward(ctx, i):\n        result = i * sigmoid(i)\n        ctx.save_for_backward(i)\n        return result\n    @staticmethod\n    def backward(ctx, grad_output):\n        i = ctx.saved_variables[0]\n        sigmoid_i = sigmoid(i)\n        return grad_output * (sigmoid_i * (1 + i * (1 - sigmoid_i)))\n\n\nclass Swish_Module(nn.Module):\n    def forward(self, x):\n        return Swish.apply(x)\n\n\nclass Effnet_Melanoma(nn.Module):\n    def __init__(self, enet_type='efficientnet_b3', out_dim=1, n_meta_features=0, n_meta_dim=[512, 128], pretrained=False):\n        super(Effnet_Melanoma, self).__init__()\n        self.n_meta_features = n_meta_features\n        self.enet = geffnet.create_model(enet_type, pretrained=pretrained)\n#         self.dropouts = nn.ModuleList([\n#             nn.Dropout(0.5) for _ in range(5)\n#         ])\n        self.dropout = nn.Dropout(0.5)\n        in_ch = self.enet.classifier.in_features\n        if n_meta_features > 0:\n            self.meta = nn.Sequential(\n                nn.Linear(n_meta_features, n_meta_dim[0]),\n                nn.BatchNorm1d(n_meta_dim[0]),\n                Swish_Module(),\n                nn.Dropout(p=0.3),\n                nn.Linear(n_meta_dim[0], n_meta_dim[1]),\n                nn.BatchNorm1d(n_meta_dim[1]),\n                Swish_Module(),\n            )\n            in_ch += n_meta_dim[1]\n        self.myfc = nn.Linear(in_ch, out_dim)\n        self.enet.classifier = nn.Identity()\n\n    def extract(self, x):\n        x = self.enet(x)\n        return x\n\n    def forward(self, x, x_meta=None):\n        x = self.extract(x).squeeze(-1).squeeze(-1)\n#         pdb.set_trace()\n#         if self.n_meta_features > 0:\n#             x_meta = self.meta(x_meta)\n#             x = torch.cat((x, x_meta), dim=1)\n#         for i, dropout in enumerate(self.dropouts):\n#             if i == 0:\n#                 out = self.myfc(dropout(x))\n#             else:\n#                 out += self.myfc(dropout(x))\n#         out /= len(self.dropouts)\n        x = self.myfc(self.dropout(x))\n        return x","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:41.770939Z","iopub.execute_input":"2024-07-27T20:29:41.771571Z","iopub.status.idle":"2024-07-27T20:29:41.785229Z","shell.execute_reply.started":"2024-07-27T20:29:41.771535Z","shell.execute_reply":"2024-07-27T20:29:41.784458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# kernel_type = '/kaggle/input/melanoma-winning-models/9c_meta_b3_768_512_ext_18ep_best_fold0.pth'\n# image_size = 512\n\n# enet_type = 'efficientnet-b3'","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:41.786365Z","iopub.execute_input":"2024-07-27T20:29:41.786833Z","iopub.status.idle":"2024-07-27T20:29:41.797061Z","shell.execute_reply.started":"2024-07-27T20:29:41.786801Z","shell.execute_reply":"2024-07-27T20:29:41.796100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_models = []\nfor i in range(5):\n    model = Effnet_Melanoma()\n    state_dict = torch.load(f'/kaggle/input/melanoma-winning-models/9c_meta_b3_768_512_ext_18ep_best_fold{i}.pth')\n    state_dict = {k.replace('module.', ''): state_dict[k] for k in state_dict.keys()}\n    state_dict.pop('myfc.weight', None)\n    state_dict.pop('myfc.bias', None)\n    model.load_state_dict(state_dict, strict=False)\n    cv_models.append(model.to(CONFIG['device']))","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:41.802748Z","iopub.execute_input":"2024-07-27T20:29:41.803007Z","iopub.status.idle":"2024-07-27T20:29:43.983617Z","shell.execute_reply.started":"2024-07-27T20:29:41.802985Z","shell.execute_reply":"2024-07-27T20:29:43.982558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cv_1= cv_models[0]\nrand_image = torch.rand((1, 3, 512, 512)).to('cuda')\ntemp_y = cv_1(rand_image) # use BCEWithLogitsLoss\ntemp_y\nflush()","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:43.984885Z","iopub.execute_input":"2024-07-27T20:29:43.985193Z","iopub.status.idle":"2024-07-27T20:29:45.025824Z","shell.execute_reply.started":"2024-07-27T20:29:43.985168Z","shell.execute_reply":"2024-07-27T20:29:45.025009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def criterion(outputs, targets):\n    return nn.BCEWithLogitsLoss(pos_weight=CONFIG['pos_weights'])(outputs, targets)","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.026902Z","iopub.execute_input":"2024-07-27T20:29:45.027194Z","iopub.status.idle":"2024-07-27T20:29:45.031636Z","shell.execute_reply.started":"2024-07-27T20:29:45.027170Z","shell.execute_reply":"2024-07-27T20:29:45.030653Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"criterion(temp_y, torch.tensor(1)[None, None].float().to('cuda'))","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.032811Z","iopub.execute_input":"2024-07-27T20:29:45.033222Z","iopub.status.idle":"2024-07-27T20:29:45.087780Z","shell.execute_reply.started":"2024-07-27T20:29:45.033189Z","shell.execute_reply":"2024-07-27T20:29:45.086901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_one_epoch(model, optimizer, scheduler, dataloader, device, epoch):\n    model.train()\n    \n    scaler = GradScaler()\n    dataset_size = 0\n    running_loss = 0.0\n    running_auroc  = 0.0\n    running_f1 = 0.0\n    bar = tqdm(enumerate(dataloader), total=len(dataloader))\n    for step, data in bar:\n        images = data['image'].to(device, dtype=torch.float)\n        targets = data['target'].to(device, dtype=torch.float)\n        \n        batch_size = images.size(0)\n        with autocast():\n            outputs = model(images).squeeze()\n            loss = criterion(outputs, targets)\n            loss = loss / CONFIG['n_accumulate']\n            \n        # Backward pass with scaling\n        scaler.scale(loss).backward()\n        \n        \n    \n        if (step + 1) % CONFIG['n_accumulate'] == 0:\n            # Step the optimizer\n            scaler.step(optimizer)\n\n            # Update the scale for next iteration\n            scaler.update()\n            # optimizer.step()\n\n            # zero the parameter gradients\n            optimizer.zero_grad()\n\n            if scheduler is not None:\n                scheduler.step()\n                \n        probabilities = torch.sigmoid(outputs).detach().cpu().numpy()\n        preds = (torch.sigmoid(outputs) > 0.5).float()\n#         pdb.set_trace()\n        auroc = average_precision_score(targets.cpu().numpy(), probabilities)\n        f1 = f1_score(targets.cpu().numpy(), preds.cpu().numpy(), average='binary')\n        \n        running_loss += (loss.item() * batch_size)\n        running_auroc  += (auroc * batch_size)\n        running_f1 += (f1 * batch_size)\n        dataset_size += batch_size\n        \n        epoch_loss = running_loss / dataset_size\n        epoch_auroc = running_auroc / dataset_size\n        epoch_f1 = running_f1 / dataset_size\n        \n        bar.set_postfix(Epoch=epoch, Train_Loss=epoch_loss, Train_Auroc=epoch_auroc, Train_F1=epoch_f1,\n                        LR=optimizer.param_groups[0]['lr'])\n    gc.collect()\n    \n    return epoch_loss, epoch_auroc","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.088810Z","iopub.execute_input":"2024-07-27T20:29:45.089080Z","iopub.status.idle":"2024-07-27T20:29:45.101497Z","shell.execute_reply.started":"2024-07-27T20:29:45.089056Z","shell.execute_reply":"2024-07-27T20:29:45.100383Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"flush()\ngc.collect()\ntorch.cuda.empty_cache()","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.102556Z","iopub.execute_input":"2024-07-27T20:29:45.102888Z","iopub.status.idle":"2024-07-27T20:29:45.462529Z","shell.execute_reply.started":"2024-07-27T20:29:45.102865Z","shell.execute_reply":"2024-07-27T20:29:45.461661Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_one_epoch(model, optimizer, scheduler, train_loader, 'cuda', 1)","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.463708Z","iopub.execute_input":"2024-07-27T20:29:45.464003Z","iopub.status.idle":"2024-07-27T20:29:45.470053Z","shell.execute_reply.started":"2024-07-27T20:29:45.463968Z","shell.execute_reply":"2024-07-27T20:29:45.469272Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"@torch.no_grad()\ndef valid_one_epoch(model, dataloader, device, epoch):\n    model.eval()\n    \n    dataset_size = 0\n    running_loss = 0.0\n    running_auroc = 0.0\n    running_f1 = 0.0 \n    \n    bar = tqdm(enumerate(dataloader), total=len(dataloader))\n    for step, data in bar:  \n        images = data['image'].to(device, dtype=torch.float)\n        targets = data['target'].to(device, dtype=torch.float)\n        \n        batch_size = images.size(0)\n        outputs = model(images).squeeze()\n        loss = criterion(outputs, targets)\n        \n#         auroc = binary_auroc(input=outputs.squeeze(), target=targets).item()\n#         running_loss += (loss.item() * batch_size)\n#         running_auroc  += (auroc * batch_size)\n#         dataset_size += batch_size\n        \n#         epoch_loss = running_loss / dataset_size\n#         epoch_auroc = running_auroc / dataset_size\n        probabilities = torch.sigmoid(outputs).detach().cpu().numpy()\n        preds = (torch.sigmoid(outputs) > 0.5).float()\n        \n        auroc = average_precision_score(targets.cpu().numpy(), probabilities)\n        f1 = f1_score(targets.cpu().numpy(), preds.cpu().numpy(), average='binary')\n        \n        running_loss += (loss.item() * batch_size)\n        running_auroc  += (auroc * batch_size)\n        running_f1 += (f1 * batch_size)\n        dataset_size += batch_size\n        \n        epoch_loss = running_loss / dataset_size\n        epoch_auroc = running_auroc / dataset_size\n        epoch_f1 = running_f1 / dataset_size\n        \n        bar.set_postfix(Epoch=epoch, Valid_Loss=epoch_loss, Valid_Auroc=epoch_auroc, \n                        Valid_F1=epoch_f1,\n                        LR=optimizer.param_groups[0]['lr'])   \n    gc.collect()\n    \n    return epoch_loss, epoch_auroc, epoch_f1","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.471169Z","iopub.execute_input":"2024-07-27T20:29:45.471468Z","iopub.status.idle":"2024-07-27T20:29:45.483331Z","shell.execute_reply.started":"2024-07-27T20:29:45.471435Z","shell.execute_reply":"2024-07-27T20:29:45.482644Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# valid_one_epoch(model, valid_loader, 'cuda', 1)","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.484316Z","iopub.execute_input":"2024-07-27T20:29:45.484559Z","iopub.status.idle":"2024-07-27T20:29:45.492749Z","shell.execute_reply.started":"2024-07-27T20:29:45.484539Z","shell.execute_reply":"2024-07-27T20:29:45.491878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def run_training(model, optimizer, scheduler, device, num_epochs, train_loader, valid_loader, fold):\n    if torch.cuda.is_available():\n        print(\"[INFO] Using GPU: {}\\n\".format(torch.cuda.get_device_name()))\n    \n    start = time.time()\n    best_model_wts = copy.deepcopy(model.state_dict())\n    best_epoch_f1 = -np.inf\n    best_valid_loss = np.inf\n    history = defaultdict(list)\n    \n    for epoch in range(1, num_epochs + 1): \n        gc.collect()\n        train_epoch_loss, train_epoch_auroc = train_one_epoch(model, optimizer, scheduler, \n                                           dataloader=train_loader, \n                                           device=CONFIG['device'], epoch=epoch)\n        \n        val_epoch_loss, val_epoch_auroc, val_epoch_f1 = valid_one_epoch(model, valid_loader, device=CONFIG['device'], \n                                         epoch=epoch)\n    \n        history['Train Loss'].append(train_epoch_loss)\n        history['Valid Loss'].append(val_epoch_loss)\n        history['Train AUROC'].append(train_epoch_auroc)\n        history['Valid AUROC'].append(val_epoch_auroc)\n        history['Valid F1'].append(val_epoch_f1)\n        history['lr'].append( scheduler.get_lr()[0] )\n        if val_epoch_loss <= best_valid_loss:\n            print(f\"Validation Loss Improved ({best_valid_loss} ---> {val_epoch_loss})\")\n            best_valid_loss = val_epoch_loss\n            best_model_wts = copy.deepcopy(model.state_dict())\n            PATH = f\"best_VAL_LOSS_model_{fold}.bin\"\n            torch.save(model.state_dict(), PATH)\n            # Save a model file from the current directory\n            print(f\"Model Saved\")\n        \n        if best_epoch_f1 <= val_epoch_f1:\n            print(f\"Validation F1 Improved ({best_epoch_f1} ---> {val_epoch_f1})\")\n            best_epoch_f1 = val_epoch_f1\n            best_model_wts = copy.deepcopy(model.state_dict())\n            PATH = f\"best_F1_model{fold}.bin\"\n            torch.save(model.state_dict(), PATH)\n            # Save a model file from the current directory\n            print(f\"Model Saved\")\n            \n\n            \n        print()\n    \n    end = time.time()\n    time_elapsed = end - start\n    print('Training complete in {:.0f}h {:.0f}m {:.0f}s'.format(\n        time_elapsed // 3600, (time_elapsed % 3600) // 60, (time_elapsed % 3600) % 60))\n    print(\"Best F1: {:.4f}\".format(best_epoch_f1))\n    print(\"Best Loss: {:.4f}\".format(best_valid_loss))\n    \n    # load best model weights\n    model.load_state_dict(best_model_wts)\n    \n    return model, history","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.493837Z","iopub.execute_input":"2024-07-27T20:29:45.494096Z","iopub.status.idle":"2024-07-27T20:29:45.507554Z","shell.execute_reply.started":"2024-07-27T20:29:45.494074Z","shell.execute_reply":"2024-07-27T20:29:45.506717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def fetch_scheduler(optimizer):\n    if CONFIG['scheduler'] == 'CosineAnnealingLR':\n        scheduler = lr_scheduler.CosineAnnealingLR(optimizer,T_max=CONFIG['T_max'], \n                                                   eta_min=CONFIG['min_lr'])\n    elif CONFIG['scheduler'] == 'CosineAnnealingWarmRestarts':\n        scheduler = lr_scheduler.CosineAnnealingWarmRestarts(optimizer,T_0=CONFIG['T_0'], \n                                                             eta_min=CONFIG['min_lr'])\n    elif CONFIG['scheduler'] == None:\n        return None\n        \n    return scheduler","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.508744Z","iopub.execute_input":"2024-07-27T20:29:45.509040Z","iopub.status.idle":"2024-07-27T20:29:45.518996Z","shell.execute_reply.started":"2024-07-27T20:29:45.509014Z","shell.execute_reply":"2024-07-27T20:29:45.518190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def prepare_loaders(df, fold):\n    \n    df_train = df[df.kfold != fold].reset_index(drop=True)\n    df_valid = df[df.kfold == fold].reset_index(drop=True)\n    \n#     train_dataset = ISICDataset_for_Train(df_train, transforms=data_transforms[\"train\"])\n    train_dataset = ISICDataset(df_train, transforms=data_transforms[\"train\"])\n    valid_dataset = ISICDataset(df_valid, transforms=data_transforms[\"valid\"])\n    \n    train_targets = df_train['target']\n    valid_targets = df_valid['target']\n    # Assuming `targets` is a pandas Series, convert it to a Tensor\n    train_targets = torch.tensor(train_targets.values)\n    valid_targets = torch.tensor(valid_targets.values)\n    # Calculate class counts and weights\n    train_class_counts = torch.bincount(train_targets)\n    train_class_weights = 1. / train_class_counts.float()\n    train_sample_weights = train_class_weights[train_targets]\n    # Calculate class counts and weights\n    valid_class_counts = torch.bincount(valid_targets)\n    valid_class_weights = 1. / valid_class_counts.float()\n    valid_sample_weights = valid_class_weights[valid_targets]\n\n    # Create a WeightedRandomSampler\n    train_sampler = WeightedRandomSampler(weights=train_sample_weights, num_samples=len(train_sample_weights), replacement=True)\n    valid_sampler = WeightedRandomSampler(weights=valid_sample_weights, num_samples=len(valid_sample_weights), replacement=True)\n    \n    train_loader = DataLoader(train_dataset, batch_size=CONFIG['train_batch_size'], \n                              num_workers=0, shuffle=True, drop_last=False)\n    valid_loader = DataLoader(valid_dataset, batch_size=CONFIG['valid_batch_size'], \n                              num_workers=0, shuffle=False)\n    \n    return train_loader, valid_loader","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.520445Z","iopub.execute_input":"2024-07-27T20:29:45.521241Z","iopub.status.idle":"2024-07-27T20:29:45.530838Z","shell.execute_reply.started":"2024-07-27T20:29:45.521199Z","shell.execute_reply":"2024-07-27T20:29:45.530041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.531940Z","iopub.execute_input":"2024-07-27T20:29:45.532251Z","iopub.status.idle":"2024-07-27T20:29:45.542038Z","shell.execute_reply.started":"2024-07-27T20:29:45.532216Z","shell.execute_reply":"2024-07-27T20:29:45.541198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loader, valid_loader = prepare_loaders(df, fold=CONFIG[\"fold\"])","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.543156Z","iopub.execute_input":"2024-07-27T20:29:45.543853Z","iopub.status.idle":"2024-07-27T20:29:45.556504Z","shell.execute_reply.started":"2024-07-27T20:29:45.543820Z","shell.execute_reply":"2024-07-27T20:29:45.555783Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = optim.AdamW(model.parameters(), lr=CONFIG['learning_rate'], \n                       weight_decay=CONFIG['weight_decay'])\nscheduler = fetch_scheduler(optimizer)","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.557550Z","iopub.execute_input":"2024-07-27T20:29:45.557831Z","iopub.status.idle":"2024-07-27T20:29:45.565686Z","shell.execute_reply.started":"2024-07-27T20:29:45.557808Z","shell.execute_reply":"2024-07-27T20:29:45.564602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# temp['image']","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.566962Z","iopub.execute_input":"2024-07-27T20:29:45.567215Z","iopub.status.idle":"2024-07-27T20:29:45.572048Z","shell.execute_reply.started":"2024-07-27T20:29:45.567194Z","shell.execute_reply":"2024-07-27T20:29:45.571122Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# temp = next(iter(train_loader))\n# with torch.no_grad():\n#     flush()\n#     y = model(temp['image'].to('cuda'))\n# flush()","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.573199Z","iopub.execute_input":"2024-07-27T20:29:45.573529Z","iopub.status.idle":"2024-07-27T20:29:45.581390Z","shell.execute_reply.started":"2024-07-27T20:29:45.573498Z","shell.execute_reply":"2024-07-27T20:29:45.580566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# for fold in range(5):\nfold = 4 # 0, 1, 2, 3 done\nmodel = cv_models[fold]\ntrain_loader, valid_loader = prepare_loaders(df, fold=fold)\noptimizer = optim.AdamW(model.parameters(), lr=CONFIG['learning_rate'], \n           weight_decay=CONFIG['weight_decay'])\nscheduler = fetch_scheduler(optimizer)\nmodel, history = run_training(model, optimizer, scheduler,\n                  device=CONFIG['device'],\n                  num_epochs=CONFIG['epochs'], train_loader=train_loader,\n                 valid_loader=valid_loader,fold=fold)","metadata":{"execution":{"iopub.status.busy":"2024-07-27T20:29:45.582381Z","iopub.execute_input":"2024-07-27T20:29:45.582930Z","iopub.status.idle":"2024-07-27T21:01:03.477077Z","shell.execute_reply.started":"2024-07-27T20:29:45.582904Z","shell.execute_reply":"2024-07-27T21:01:03.475007Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model, history = run_training(model, optimizer, scheduler,\n#                               device=CONFIG['device'],\n#                               num_epochs=1, train_loader=train_loader,\n#                              valid_loader=valid_loader,fold=0)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# sub = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/sample_submission.csv')\n# sub['target'] = preds.cpu().numpy().reshape(-1,)\n# sub.to_csv('submission.csv', index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}