{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"dockerImageVersionId":31239,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <center> Computer Vision - Data Processing </center>","metadata":{}},{"cell_type":"code","source":"!nvidia-smi","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-13T13:46:03.849713Z","iopub.execute_input":"2026-01-13T13:46:03.850522Z","iopub.status.idle":"2026-01-13T13:46:03.968004Z","shell.execute_reply.started":"2026-01-13T13:46:03.850489Z","shell.execute_reply":"2026-01-13T13:46:03.967302Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Import","metadata":{}},{"cell_type":"code","source":"import os, json, hashlib\nfrom pathlib import Path\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nfrom tqdm.auto import tqdm\n\nfrom sklearn.model_selection import GroupKFold","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-13T13:49:15.985242Z","iopub.execute_input":"2026-01-13T13:49:15.985482Z","iopub.status.idle":"2026-01-13T13:49:16.635746Z","shell.execute_reply.started":"2026-01-13T13:49:15.985464Z","shell.execute_reply":"2026-01-13T13:49:16.634893Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Config","metadata":{}},{"cell_type":"code","source":"DATA_DIR  = Path(\"/kaggle/input/siim-isic-melanoma-classification\")\nIMG_DIR   = DATA_DIR / \"jpeg\" / \"train\"\nCSV_PATH  = DATA_DIR / \"train.csv\"\n\nOUT_ROOT  = Path(\"/kaggle/working/siim_isic_cache\")\nOUT_ROOT.mkdir(parents=True, exist_ok=True)\n\nIMG_SIZE = 384\nSEED = 42\nVAL_FOLD = 0\nN_SPLITS = 5","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-13T13:48:44.950694Z","iopub.execute_input":"2026-01-13T13:48:44.950984Z","iopub.status.idle":"2026-01-13T13:48:44.955943Z","shell.execute_reply.started":"2026-01-13T13:48:44.950965Z","shell.execute_reply":"2026-01-13T13:48:44.955182Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Fingerprint","metadata":{}},{"cell_type":"code","source":"PREP_CFG = {\n    \"img_size\": IMG_SIZE,\n    \"img_mode\": \"RGB\",\n    \"resize\": \"bilinear\",\n    \"tabular\": {\n        \"cat_cols\": [\"sex\", \"anatom_site_general_challenge\"],\n        \"num_cols\": [\"age_approx\"],\n        \"impute_cat\": \"unknown\",\n        \"impute_age\": \"median_train\",\n        \"one_hot\": True,\n        \"scale_age\": \"zscore_train\",\n    },\n    \"split\": {\"group\": \"patient_id\", \"n_splits\": N_SPLITS, \"val_fold\": VAL_FOLD, \"seed\": SEED}\n}\n\ndef cfg_fingerprint(cfg: dict) -> str:\n    s = json.dumps(cfg, sort_keys=True).encode(\"utf-8\")\n    return hashlib.md5(s).hexdigest()[:10]\n\nFP = cfg_fingerprint(PREP_CFG)\nOUT_DIR = OUT_ROOT / f\"fp_{FP}\"\nOUT_DIR.mkdir(parents=True, exist_ok=True)\nprint(\"Fingerprint:\", FP, \"\\nOUT_DIR:\", OUT_DIR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-13T13:48:54.036397Z","iopub.execute_input":"2026-01-13T13:48:54.036672Z","iopub.status.idle":"2026-01-13T13:48:54.043233Z","shell.execute_reply.started":"2026-01-13T13:48:54.036655Z","shell.execute_reply":"2026-01-13T13:48:54.042268Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Load CSV + Split","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(CSV_PATH)\n\ndf[\"sex\"] = df[\"sex\"].fillna(PREP_CFG[\"tabular\"][\"impute_cat\"]).astype(str)\ndf[\"anatom_site_general_challenge\"] = df[\"anatom_site_general_challenge\"].fillna(PREP_CFG[\"tabular\"][\"impute_cat\"]).astype(str)\ndf[\"age_approx\"] = pd.to_numeric(df[\"age_approx\"], errors=\"coerce\")\n\ngroups = df[\"patient_id\"].astype(str).values\ny = df[\"target\"].astype(int).values\n\ngkf = GroupKFold(n_splits=N_SPLITS)\nsplits = list(gkf.split(df, y, groups=groups))\ntrain_idx, val_idx = splits[VAL_FOLD]\n\nprint(\"N:\", len(df), \"train:\", len(train_idx), \"val:\", len(val_idx))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-13T13:49:56.490031Z","iopub.execute_input":"2026-01-13T13:49:56.490390Z","iopub.status.idle":"2026-01-13T13:49:56.618412Z","shell.execute_reply.started":"2026-01-13T13:49:56.490371Z","shell.execute_reply":"2026-01-13T13:49:56.617584Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Metadata preprocess","metadata":{}},{"cell_type":"code","source":"CAT_COLS = PREP_CFG[\"tabular\"][\"cat_cols\"]\nNUM_COLS = PREP_CFG[\"tabular\"][\"num_cols\"]\n\ntrain_df = df.iloc[train_idx].copy()\n\nage_median = float(np.nanmedian(train_df[\"age_approx\"].values))\ndf[\"age_approx\"] = df[\"age_approx\"].fillna(age_median)\n\n# fit cat levels on TRAIN only\ncat_levels = {c: sorted(train_df[c].unique().tolist()) for c in CAT_COLS}\n\n# build cat indices (one-hot positions)\ncat_index, offset = {}, 0\nfor c in CAT_COLS:\n    idx = {v: i + offset for i, v in enumerate(cat_levels[c])}\n    cat_index[c] = idx\n    offset += len(cat_levels[c])\n\n# z-score params on TRAIN only\nage_mean = float(train_df[\"age_approx\"].fillna(age_median).mean())\nage_std  = float(train_df[\"age_approx\"].fillna(age_median).std(ddof=0) + 1e-8)\n\nmeta_dim = offset + len(NUM_COLS)\n\nencoder_state = {\n    \"cfg_fp\": FP,\n    \"cat_cols\": CAT_COLS,\n    \"num_cols\": NUM_COLS,\n    \"cat_levels\": cat_levels,\n    \"cat_index\": cat_index,\n    \"age_median\": age_median,\n    \"age_mean\": age_mean,\n    \"age_std\": age_std,\n    \"meta_dim\": meta_dim\n}\n\nwith open(OUT_DIR / \"metadata_encoder.json\", \"w\") as f:\n    json.dump(encoder_state, f)\n\nprint(\"meta_dim:\", meta_dim, \"saved encoder.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-13T13:51:35.359422Z","iopub.execute_input":"2026-01-13T13:51:35.359646Z","iopub.status.idle":"2026-01-13T13:51:35.392063Z","shell.execute_reply.started":"2026-01-13T13:51:35.359630Z","shell.execute_reply":"2026-01-13T13:51:35.391471Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Encode metadata + cache Image","metadata":{}},{"cell_type":"code","source":"def encode_metadata_row(row, enc) -> np.ndarray:\n    x = np.zeros(enc[\"meta_dim\"], dtype=np.float32)\n\n    # one-hot cats\n    for c in enc[\"cat_cols\"]:\n        v = str(row[c]) if pd.notna(row[c]) else PREP_CFG[\"tabular\"][\"impute_cat\"]\n        j = enc[\"cat_index\"][c].get(v, None)\n        if j is not None:\n            x[j] = 1.0\n        # unseen categories in val/test => all-zero for that cat group\n\n    # numeric: age_approx (zscore)\n    age = float(row[\"age_approx\"]) if pd.notna(row[\"age_approx\"]) else float(enc[\"age_median\"])\n    age = (age - enc[\"age_mean\"]) / enc[\"age_std\"]\n    x[-1] = age\n\n    return x\n\nimg_cache = OUT_DIR / \"images_uint8.npy\"\nmeta_cache = OUT_DIR / \"meta_float32.npy\"\ny_cache = OUT_DIR / \"targets_uint8.npy\"\nnames_path = OUT_DIR / \"image_names.json\"\nsplits_path = OUT_DIR / \"splits.json\"","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-13T13:52:48.654054Z","iopub.execute_input":"2026-01-13T13:52:48.654294Z","iopub.status.idle":"2026-01-13T13:52:48.659544Z","shell.execute_reply.started":"2026-01-13T13:52:48.654279Z","shell.execute_reply":"2026-01-13T13:52:48.659020Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open(OUT_DIR / \"metadata_encoder.json\", \"r\") as f:\n    enc = json.load(f)\n\nimage_names = df[\"image_name\"].tolist()\nN = len(df)\n\nimgs = np.zeros((N, IMG_SIZE, IMG_SIZE, 3), dtype=np.uint8)\nmetas = np.zeros((N, enc[\"meta_dim\"]), dtype=np.float32)\nys = df[\"target\"].astype(np.uint8).values\n\nfor i, row in tqdm(list(df.iterrows()), total=N):\n    name = row[\"image_name\"]\n    img_path = IMG_DIR / f\"{name}.jpg\"\n    im = Image.open(img_path).convert(\"RGB\")\n    im = im.resize((IMG_SIZE, IMG_SIZE), resample=Image.BILINEAR)\n    imgs[i] = np.asarray(im, dtype=np.uint8)\n    metas[i] = encode_metadata_row(row, enc)\n\nnp.save(img_cache, imgs)\nnp.save(meta_cache, metas)\nnp.save(y_cache, ys)\n\nwith open(names_path, \"w\") as f:\n    json.dump(image_names, f)\n\nsplits_obj = {\n    \"train_idx\": train_idx.tolist(),\n    \"val_idx\": val_idx.tolist(),\n    \"n_splits\": N_SPLITS,\n    \"val_fold\": VAL_FOLD,\n    \"group_key\": \"patient_id\",\n    \"cfg_fp\": FP\n}\nwith open(splits_path, \"w\") as f:\n    json.dump(splits_obj, f)\n\nprint(\"Saved:\", OUT_DIR)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-13T13:53:10.097941Z","iopub.execute_input":"2026-01-13T13:53:10.098177Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"---","metadata":{}}]}