{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"},{"sourceId":13205039,"sourceType":"datasetVersion","datasetId":8369086}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfrom kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\n\n# Retrieve the Hugging Face token from Kaggle Secrets and set it as an environment variable\nhf_token = user_secrets.get_secret(\"HUGGING_FACE_TOKEN\")\nos.environ[\"HF_TOKEN\"] = hf_token","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-15T07:57:45.309338Z","iopub.execute_input":"2025-10-15T07:57:45.309693Z","iopub.status.idle":"2025-10-15T07:57:45.348068Z","shell.execute_reply.started":"2025-10-15T07:57:45.309663Z","shell.execute_reply":"2025-10-15T07:57:45.347153Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import requests\nfrom concurrent.futures import ThreadPoolExecutor, as_completed\n\nfrom typing import Any\nfrom PIL import Image\nimport numpy as np\nimport math\nimport re\nimport base64\nimport random\nfrom google.cloud import storage\nimport pandas as pd\nimport io\nimport json\nimport ast\n\nimport timm\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.utils.data import Dataset, DataLoader\nfrom torch.optim.lr_scheduler import CosineAnnealingWarmRestarts\nfrom sklearn.metrics import f1_score, roc_auc_score, roc_curve, confusion_matrix\nfrom sklearn.preprocessing import label_binarize\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\nimport albumentations as A\nfrom albumentations.pytorch import ToTensorV2\n\ndef seed_everything(seed: int):\n    import random, os\n    import numpy as np\n    import torch\n    \n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    np.random.seed(seed)\n    torch.manual_seed(seed)\n    torch.cuda.manual_seed(seed)\n    torch.backends.cudnn.deterministic = True\n    torch.backends.cudnn.benchmark = True\n    \nseed_everything(42)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-15T07:57:45.349678Z","iopub.execute_input":"2025-10-15T07:57:45.349959Z","iopub.status.idle":"2025-10-15T07:58:03.960155Z","shell.execute_reply.started":"2025-10-15T07:57:45.349937Z","shell.execute_reply":"2025-10-15T07:58:03.958539Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class Globals:\n  gcp_project = 'dx-scin-public' #@param\n  gcs_bucket_name = 'dx-scin-public-data' #@param\n  cases_csv = 'dataset/scin_cases.csv' #@param\n  labels_csv = 'dataset/scin_labels.csv' #@param\n  gcs_images_dir = 'dataset/images/' #@param\n\n  ### Key column names\n  image_path_columns = ['image_1_path', 'image_2_path', 'image_3_path']\n  weighted_skin_condition_label = \"weighted_skin_condition_label\"\n  skin_condition_label = \"dermatologist_skin_condition_on_label_name\"\n\n  gcs_storage_client = None\n  gcs_bucket = None\n  cases_df = None\n  cases_and_labels_df = None","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-15T07:58:03.960740Z","iopub.status.idle":"2025-10-15T07:58:03.960977Z","shell.execute_reply.started":"2025-10-15T07:58:03.960860Z","shell.execute_reply":"2025-10-15T07:58:03.960870Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"os.environ[\"GOOGLE_APPLICATION_CREDENTIALS\"] = \"/kaggle/input/gcs-json/gen-lang-client-0182740461-c4c482d65f84.json\"\n\ndef list_blobs(storage_client, bucket_name):\n  \"\"\"Helper to list blobs in a bucket (useful for debugging).\"\"\"\n  blobs = storage_client.list_blobs(bucket_name)\n  for blob in blobs:\n    print(blob)\n\ndef initialize_df_with_metadata(bucket, csv_path):\n  \"\"\"Loads the given CSV into a pd.DataFrame.\"\"\"\n  df = pd.read_csv(io.BytesIO(bucket.blob(csv_path).download_as_string()), dtype={'case_id': str})\n  df['case_id'] = df['case_id'].astype(str)\n  return df\n\ndef augment_metadata_with_labels(df, bucket, csv_path):\n  \"\"\"Loads the given CSV into a pd.DataFrame.\"\"\"\n  labels_df = pd.read_csv(io.BytesIO(bucket.blob(csv_path).download_as_string()), dtype={'case_id': str})\n  labels_df['case_id'] = labels_df['case_id'].astype(str)\n  merged_df = pd.merge(df, labels_df, on='case_id')\n  return merged_df\n\nGlobals.gcs_storage_client = storage.Client(Globals.gcp_project)\nGlobals.gcs_bucket = Globals.gcs_storage_client.bucket(\n    Globals.gcs_bucket_name\n)\nGlobals.cases_df = initialize_df_with_metadata(Globals.gcs_bucket, Globals.cases_csv)\nGlobals.cases_and_labels_df = augment_metadata_with_labels(Globals.cases_df, Globals.gcs_bucket, Globals.labels_csv)\nprint(len(Globals.cases_and_labels_df))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-15T07:58:03.962283Z","iopub.status.idle":"2025-10-15T07:58:03.962537Z","shell.execute_reply.started":"2025-10-15T07:58:03.962424Z","shell.execute_reply":"2025-10-15T07:58:03.962435Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_top_condition_and_prob(label_str):\n    try:\n        label_dict = ast.literal_eval(label_str)\n        if not isinstance(label_dict, dict) or len(label_dict) == 0:\n            return None, None\n        cond, prob = max(label_dict.items(), key=lambda x: x[1])\n        return cond, prob\n    except Exception:\n        return None, None\n\nGlobals.cases_and_labels_df[[\"top_condition\", \"top_prob\"]] = (\n    Globals.cases_and_labels_df[\"weighted_skin_condition_label\"]\n    .apply(lambda s: pd.Series(get_top_condition_and_prob(s)))\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-15T07:58:03.963420Z","iopub.status.idle":"2025-10-15T07:58:03.963713Z","shell.execute_reply.started":"2025-10-15T07:58:03.963547Z","shell.execute_reply":"2025-10-15T07:58:03.963562Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = Globals.cases_and_labels_df\ndf[\"image_paths\"] = df[Globals.image_path_columns].values.tolist()\ndf[\"weighted_skin_condition_label\"] = df[\"weighted_skin_condition_label\"].apply(\n    lambda x: ast.literal_eval(x) if isinstance(x, str) else x\n)\n# Drop rows with empty dicts\ndf = df[df[\"weighted_skin_condition_label\"].astype(str) != \"{}\"]\n# checking dict type\ndf = df[df[\"weighted_skin_condition_label\"].apply(lambda x: bool(x))]\nprint(df.shape)\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-15T07:58:03.965102Z","iopub.status.idle":"2025-10-15T07:58:03.965413Z","shell.execute_reply.started":"2025-10-15T07:58:03.965239Z","shell.execute_reply":"2025-10-15T07:58:03.965254Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df[\"main_label\"] = df[\"weighted_skin_condition_label\"].apply(lambda d: max(d, key=d.get))\n# Count frequencies\nlabel_counts = df[\"main_label\"].value_counts()\n\n# Labels with less than 10 members\nrare_labels = label_counts[label_counts < 10].index.tolist()\ndf = df[~df[\"main_label\"].isin(rare_labels)].reset_index(drop=True)\ndf = df[[\"case_id\",\n        \"age_group\",\n        \"sex_at_birth\",\n        \"fitzpatrick_skin_type\",\n        \"combined_race\",\n        \"image_paths\",\n        \"main_label\"]]\nlabel_counts = df[\"main_label\"].value_counts()\nprint(f'New label counts: {label_counts}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-15T07:58:03.966609Z","iopub.status.idle":"2025-10-15T07:58:03.966895Z","shell.execute_reply.started":"2025-10-15T07:58:03.966774Z","shell.execute_reply":"2025-10-15T07:58:03.966788Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"all_image_lists = list(df[\"image_paths\"])\n\n# Flatten nested lists (in case each row has multiple images)\nall_image_paths = []\nfor item in all_image_lists:\n    if isinstance(item, (list, tuple)):\n        all_image_paths.extend(item)\n    elif isinstance(item, str):\n        all_image_paths.append(item)\n\n# Clean: remove None / nan / invalid\nall_image_paths = [\n    p for p in all_image_paths\n    if isinstance(p, str) and p.lower() != \"none\" and p.strip() != \"\"\n]\n\n# Deduplicate\nall_image_paths = list(set(all_image_paths))\nprint(f\"Total unique image paths: {len(all_image_paths)}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-15T07:58:03.968450Z","iopub.status.idle":"2025-10-15T07:58:03.968807Z","shell.execute_reply.started":"2025-10-15T07:58:03.968624Z","shell.execute_reply":"2025-10-15T07:58:03.968640Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# CONFIG\nbucket_name = \"dx-scin-public-data\"\nlocal_root = \"./data_cache\" \nos.makedirs(local_root, exist_ok=True)\n\ndef download_one(obj_path: str):\n    \"\"\"Download one file from GCS to local cache.\"\"\"\n    if obj_path is None or str(obj_path).lower() == \"none\":\n        return None\n\n    # Build local path\n    local_path = os.path.join(local_root, obj_path)\n    os.makedirs(os.path.dirname(local_path), exist_ok=True)\n\n    # Skip if already exists\n    if os.path.exists(local_path):\n        return local_path\n\n    # Build GCS public URL\n    url = f\"https://storage.googleapis.com/download/storage/v1/b/{bucket_name}/o/{obj_path.replace('/', '%2F')}?alt=media\"\n\n    for attempt in range(3):\n        try:\n            r = requests.get(url, timeout=15)\n            if r.status_code == 200:\n                with open(local_path, \"wb\") as f:\n                    f.write(r.content)\n                return local_path\n            else:\n                raise RuntimeError(f\"HTTP {r.status_code}\")\n        except Exception as e:\n            if attempt == 2:\n                print(f\"[FAIL] {obj_path}: {e}\")\n    return None\n\n\n# Parallel download\nresults = []\nwith ThreadPoolExecutor(max_workers=4) as ex:\n    futures = {ex.submit(download_one, p): p for p in all_image_paths}\n    for fut in tqdm(as_completed(futures), total=len(futures), desc=\"Downloading\"):\n        res = fut.result()\n        if res:\n            results.append(res)\n\nprint(f\"Downloaded {len(results)} / {len(all_image_paths)} files to {local_root}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-15T07:58:03.969576Z","iopub.status.idle":"2025-10-15T07:58:03.969884Z","shell.execute_reply.started":"2025-10-15T07:58:03.969739Z","shell.execute_reply":"2025-10-15T07:58:03.969753Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"len(os.listdir('/kaggle/working/data_cache/dataset/images'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-15T07:58:03.970576Z","iopub.status.idle":"2025-10-15T07:58:03.970795Z","shell.execute_reply.started":"2025-10-15T07:58:03.970690Z","shell.execute_reply":"2025-10-15T07:58:03.970700Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import shutil\nimport os\n\n# Path to your images folder\nfolder_path = \"/kaggle/working/data_cache/dataset/images\"\nzip_path = \"/kaggle/working/images_backup\"\n\nshutil.make_archive(zip_path, 'zip', folder_path)\nprint(f\"✅ Zipped folder created at: {zip_path}.zip\")\n\nshutil.rmtree(folder_path)\nprint(f\"🗑️ Removed original folder: {folder_path}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-15T07:58:03.971639Z","iopub.status.idle":"2025-10-15T07:58:03.971865Z","shell.execute_reply.started":"2025-10-15T07:58:03.971759Z","shell.execute_reply":"2025-10-15T07:58:03.971769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.to_csv('metadata.csv', index = False)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}