{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"colab":{"machine_shape":"hm","gpuType":"T4"},"accelerator":"GPU","kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":25383,"databundleVersionId":2684322,"sourceType":"competition"},{"sourceId":9797223,"sourceType":"datasetVersion","datasetId":982170},{"sourceId":80230206,"sourceType":"kernelVersion"},{"sourceId":266643867,"sourceType":"kernelVersion"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import sys\nsys.path.append(\"../input/tez-lib/\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T18:03:51.375659Z","iopub.execute_input":"2025-10-11T18:03:51.375962Z","iopub.status.idle":"2025-10-11T18:03:51.380171Z","shell.execute_reply.started":"2025-10-11T18:03:51.375937Z","shell.execute_reply":"2025-10-11T18:03:51.379332Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch.nn as nn\nimport tez\nimport albumentations\nimport pandas as pd\nimport cv2\nimport numpy as np\nimport timm\nimport torch.nn as nn\nimport torch\nimport random\nimport numpy as np\nimport albumentations as A\nimport torch\nimport os","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T18:03:51.381355Z","iopub.execute_input":"2025-10-11T18:03:51.381584Z","iopub.status.idle":"2025-10-11T18:03:51.403694Z","shell.execute_reply.started":"2025-10-11T18:03:51.381562Z","shell.execute_reply":"2025-10-11T18:03:51.402974Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tqdm import tqdm\nfrom tez.callbacks import EarlyStopping\nfrom tez import Tez, TezConfig\nfrom sklearn import metrics","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T18:03:51.404828Z","iopub.execute_input":"2025-10-11T18:03:51.405118Z","iopub.status.idle":"2025-10-11T18:03:51.419649Z","shell.execute_reply.started":"2025-10-11T18:03:51.405095Z","shell.execute_reply":"2025-10-11T18:03:51.418904Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def set_global_seed(seed_value):\n    random.seed(seed_value)\n    os.environ[\"PYTHONHASHSEED\"] = str(seed_value)\n    np.random.seed(seed_value)\n    torch.manual_seed(seed_value)\n    torch.cuda.manual_seed(seed_value)\n    torch.backends.cudnn.deterministic = True\n\nset_global_seed(42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T18:03:51.420351Z","iopub.execute_input":"2025-10-11T18:03:51.420543Z","iopub.status.idle":"2025-10-11T18:03:51.437175Z","shell.execute_reply.started":"2025-10-11T18:03:51.420527Z","shell.execute_reply":"2025-10-11T18:03:51.436439Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ProcessArguments:\n    sample_batch_size = 8\n    sample_image_size = 384\n    sample_epochs = 1 #10\n    sample_fold = 0","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T18:03:51.438752Z","iopub.execute_input":"2025-10-11T18:03:51.439175Z","iopub.status.idle":"2025-10-11T18:03:51.452760Z","shell.execute_reply.started":"2025-10-11T18:03:51.439157Z","shell.execute_reply":"2025-10-11T18:03:51.451949Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class CustomDataset:\n    def __init__(self, image_paths, dense_features, targets, augmentations=None):\n        self.img_paths = image_paths\n        self.tabular_features = dense_features\n        self.labels = targets\n        self.transforms = augmentations\n\n    def __len__(self):\n        return len(self.img_paths)\n\n    def __getitem__(self, index):\n        # Read image\n        image_file = self.img_paths[index]\n        image_data = cv2.imread(image_file)\n        if image_data is None:\n            raise FileNotFoundError(f\"Could not load image: {image_file}\")\n\n        # Convert to RGB\n        image_data = cv2.cvtColor(image_data, cv2.COLOR_BGR2RGB)\n\n        # Apply augmentations (if available)\n        if self.transforms:\n            augmented_result = self.transforms(image=image_data)\n            image_data = augmented_result[\"image\"]\n\n        # Convert (H, W, C) → (C, H, W)\n        image_data = np.moveaxis(image_data, -1, 0).astype(np.float32)\n\n        # Collect tabular features and labels\n        features_vec = self.tabular_features[index, :]\n        label_value = self.labels[index]\n\n        return {\n            \"image\": torch.tensor(image_data, dtype=torch.float32),\n            \"features\": torch.tensor(features_vec, dtype=torch.float32),\n            \"targets\": torch.tensor(label_value, dtype=torch.float32),\n        }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T18:03:51.453444Z","iopub.execute_input":"2025-10-11T18:03:51.453651Z","iopub.status.idle":"2025-10-11T18:03:51.469314Z","shell.execute_reply.started":"2025-10-11T18:03:51.453637Z","shell.execute_reply":"2025-10-11T18:03:51.468690Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport timm\n\nclass CustomModel(nn.Module):\n    def __init__(self):\n        super().__init__()\n\n        # Base model backbone\n        self.backbone = timm.create_model(\n            \"resnet50\", pretrained=True, in_chans=3\n        )\n\n        self.dropout_layer = nn.Dropout(0.5)\n        self.output_layer = nn.Linear(1000, 1)\n\n        # Used by Tez\n        self.step_scheduler_after = \"epoch\"\n\n    def monitor_metrics(self, outputs, targets, loss_value):\n        metric_val = loss_value\n        if str(metric_val) == \"nan\":\n            metric_val = float(\"inf\")\n        return {\"rmse\": metric_val}\n\n    def optimizer_scheduler(self):\n        optimizer = torch.optim.AdamW(\n            self.parameters(), lr=2.5e-5, weight_decay=0.01\n        )\n        scheduler = torch.optim.lr_scheduler.CosineAnnealingWarmRestarts(\n            optimizer, T_0=10, T_mult=1, eta_min=1e-6, last_epoch=-1\n        )\n        return optimizer, scheduler\n\n    def forward(self, image, features, targets=None):\n        # match dataset keys exactly: \"image\", \"features\", \"targets\"\n        x = self.backbone(image)\n        x = self.dropout_layer(x)\n\n        # if you decide to use dense features, uncomment this:\n        # x = torch.cat([x, features], dim=1)\n\n        preds = self.output_layer(x)\n\n        if targets is not None:\n            loss_fn = nn.BCEWithLogitsLoss()\n            loss = loss_fn(preds, targets.view(-1, 1).type_as(preds))\n            metrics = self.monitor_metrics(preds, targets, loss)\n            return preds, loss, metrics\n\n        return preds, 0, {}\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T18:05:04.242290Z","iopub.execute_input":"2025-10-11T18:05:04.243102Z","iopub.status.idle":"2025-10-11T18:05:04.250486Z","shell.execute_reply.started":"2025-10-11T18:05:04.243072Z","shell.execute_reply":"2025-10-11T18:05:04.249673Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# --- Training augmentations ---\ntrain_augmentations = A.Compose(\n    [\n        A.LongestMaxSize(max_size=ProcessArguments.sample_image_size, p=1.0),\n        A.PadIfNeeded(\n            min_height=ProcessArguments.sample_image_size,\n            min_width=ProcessArguments.sample_image_size,\n            border_mode=0,\n            p=1.0\n        ),\n\n        A.HorizontalFlip(p=0.5),\n        A.VerticalFlip(p=0.1),\n        A.Rotate(limit=180, p=0.5),\n        A.ShiftScaleRotate(\n            shift_limit=0.1,\n            scale_limit=0.1,\n            rotate_limit=45,\n            p=0.5\n        ),\n\n        A.HueSaturationValue(\n            hue_shift_limit=0.2,\n            sat_shift_limit=0.2,\n            val_shift_limit=0.2,\n            p=0.5\n        ),\n        A.RandomBrightnessContrast(\n            brightness_limit=(-0.1, 0.1),\n            contrast_limit=(-0.1, 0.1),\n            p=0.5\n        ),\n\n        A.Normalize(\n            mean=[0.485, 0.456, 0.406],\n            std=[0.229, 0.224, 0.225],\n            max_pixel_value=255.0,\n            p=1.0\n        ),\n    ],\n    p=1.0,\n)\n\n# --- Validation augmentations ---\nvalid_augmentations = A.Compose(\n    [\n        A.LongestMaxSize(max_size=ProcessArguments.sample_image_size, p=1.0),\n        A.PadIfNeeded(\n            min_height=ProcessArguments.sample_image_size,\n            min_width=ProcessArguments.sample_image_size,\n            border_mode=0,\n            p=1.0\n        ),\n        A.Normalize(\n            mean=[0.485, 0.456, 0.406],\n            std=[0.229, 0.224, 0.225],\n            max_pixel_value=255.0,\n            p=1.0\n        ),\n    ],\n    p=1.0,\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T18:05:06.038248Z","iopub.execute_input":"2025-10-11T18:05:06.038722Z","iopub.status.idle":"2025-10-11T18:05:06.053544Z","shell.execute_reply.started":"2025-10-11T18:05:06.038697Z","shell.execute_reply":"2025-10-11T18:05:06.052855Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/shreyas-10155-same-old-creating-folds/train_5folds.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T18:05:06.200164Z","iopub.execute_input":"2025-10-11T18:05:06.200419Z","iopub.status.idle":"2025-10-11T18:05:06.252614Z","shell.execute_reply.started":"2025-10-11T18:05:06.200400Z","shell.execute_reply":"2025-10-11T18:05:06.251927Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from tez.callbacks import EarlyStopping\nfrom tez import Tez, TezConfig\n\n# --- Training Setup ---\nfold_idx = 0\nprint(f\"Training fold: {fold_idx} start\")\n\n# Assign current fold\nProcessArguments.sample_fold = fold_idx\n\n# --- Split data ---\ndf_train = df[df.kfold != ProcessArguments.sample_fold].reset_index(drop=True)\ndf_valid = df[df.kfold == ProcessArguments.sample_fold].reset_index(drop=True)\n\ndense_features = []  # No dense features used for now\n\n# --- Image paths ---\ntrain_img_paths = [\n    f\"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/{img_name}.jpg\"\n    for img_name in df_train[\"image_name\"].values\n]\nvalid_img_paths = [\n    f\"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/{img_name}.jpg\"\n    for img_name in df_valid[\"image_name\"].values\n]\n\n# --- Datasets ---\ntrain_dataset = CustomDataset(\n    image_paths=train_img_paths,\n    dense_features=df_train[dense_features].values,\n    targets=df_train.target.values,\n    augmentations=train_augmentations,\n)\n\nvalid_dataset = CustomDataset(\n    image_paths=valid_img_paths,\n    dense_features=df_valid[dense_features].values,\n    targets=df_valid.target.values,\n    augmentations=valid_augmentations,\n)\n\n# --- Model ---\nmodel = CustomModel()\ntez_model = Tez(model)\n\n# --- Configuration ---\nconfig = TezConfig(\n    training_batch_size=ProcessArguments.sample_batch_size,\n    validation_batch_size=2 * ProcessArguments.sample_batch_size,\n    epochs=ProcessArguments.sample_epochs,\n    step_scheduler_after=\"epoch\",\n    step_scheduler_metric=\"valid_rmse\",\n    fp16=True,\n    val_strategy=\"batch\",\n    val_steps=900,\n)\n\n# --- Early stopping ---\nearly_stopper = EarlyStopping(\n    monitor=\"valid_rmse\",\n    model_path=f\"model_f{ProcessArguments.sample_fold}.bin\",\n    patience=4,\n    mode=\"min\",\n    save_weights_only=True,\n)\n\n# --- Train ---\ntez_model.fit(\n    train_dataset,\n    valid_dataset=valid_dataset,\n    callbacks=[early_stopper],\n    config=config,\n)\n\nprint(f\"Training fold: {fold_idx} complete\")\n\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-11T18:05:07.949094Z","iopub.execute_input":"2025-10-11T18:05:07.949798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}