{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.10.0"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":14774,"databundleVersionId":875431,"isSourceIdPinned":false}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"id":"a0d8c72d-3efb-4e3d-90f9-775cd39d8cea","cell_type":"markdown","source":"# 🩺 APTOS 2019 — Full Pipeline\n### Preprocessing → Augmentation → Balancement → Class Weights","metadata":{}},{"id":"27ff8ea0-3b5b-4dc9-801b-6d2091214679","cell_type":"markdown","source":"## ⚙️ Configuration & Imports","metadata":{}},{"id":"5f167e90-9625-4448-800b-cd9efd7922ca","cell_type":"code","source":"import os\nimport cv2\nimport json\nimport random\nimport shutil\nimport pandas as pd\nimport numpy as np\nfrom tqdm import tqdm\nfrom collections import Counter\n\n# ── Chemins ──────────────────────────────────────\nDATASET_PATH   = \"/kaggle/input/competitions/aptos2019-blindness-detection\"\nIMG_PATH       = os.path.join(DATASET_PATH, \"train_images\")\nCSV_PATH       = os.path.join(DATASET_PATH, \"train.csv\")\nOUTPUT         = \"/kaggle/working/APTOS_processed\"\nOUTPUT_BALANCE = \"/kaggle/working/APTOS_balanced\"\n\nRANDOM_SEED = 42\n\n# ── Créer dossiers 0-4 ───────────────────────────\nfor i in range(5):\n    os.makedirs(os.path.join(OUTPUT, str(i)), exist_ok=True)\n\ndf = pd.read_csv(CSV_PATH)\nprint(f\"✅ Dataset chargé : {len(df)} images\")\nprint(df['diagnosis'].value_counts().sort_index())","metadata":{},"outputs":[],"execution_count":null},{"id":"a50887c4-4387-4395-93b3-4da56a573466","cell_type":"markdown","source":"## 🔬 Étape 1 — Preprocessing","metadata":{}},{"id":"25fd4c4f-ee68-4283-b855-2a940aafd86a","cell_type":"code","source":"def preprocess(img):\n    img = cv2.resize(img, (224, 224))\n    img = cv2.GaussianBlur(img, (5, 5), 0)\n\n    lab = cv2.cvtColor(img, cv2.COLOR_BGR2LAB)\n    l, a, b = cv2.split(lab)\n\n    clahe = cv2.createCLAHE(clipLimit=2.0, tileGridSize=(8, 8))\n    cl = clahe.apply(l)\n\n    limg = cv2.merge((cl, a, b))\n    img  = cv2.cvtColor(limg, cv2.COLOR_LAB2BGR)\n    return img","metadata":{},"outputs":[],"execution_count":null},{"id":"f2adcb89-05a2-4c7d-a17a-47219361cc8f","cell_type":"markdown","source":"## 🎨 Étape 2 — Augmentation","metadata":{}},{"id":"1128e85b-359a-4368-a888-55375a5ce715","cell_type":"code","source":"def augment(img, seed=None):\n    \"\"\"\n    seed=None  → 3 augmentations fixes (pipeline de base)\n    seed=int   → 1 augmentation aléatoire (oversampling)\n    \"\"\"\n    if seed is not None:\n        rng = random.Random(seed)\n        op  = rng.randint(0, 5)\n        if op == 0:\n            return cv2.flip(img, 1)\n        elif op == 1:\n            return cv2.flip(img, 0)\n        elif op == 2:\n            M = cv2.getRotationMatrix2D((112, 112), rng.uniform(-20, 20), 1)\n            return cv2.warpAffine(img, M, (224, 224))\n        elif op == 3:\n            return cv2.convertScaleAbs(img, alpha=1.2, beta=10)\n        elif op == 4:\n            return cv2.convertScaleAbs(img, alpha=0.85, beta=-10)\n        else:\n            flipped = cv2.flip(img, 1)\n            M = cv2.getRotationMatrix2D((112, 112), rng.uniform(-15, 15), 1)\n            return cv2.warpAffine(flipped, M, (224, 224))\n\n    # Mode original : 3 augmentations fixes\n    aug_images = []\n    aug_images.append(cv2.flip(img, 1))\n    M = cv2.getRotationMatrix2D((112, 112), 15, 1)\n    aug_images.append(cv2.warpAffine(img, M, (224, 224)))\n    aug_images.append(cv2.convertScaleAbs(img, alpha=1.2, beta=10))\n    return aug_images","metadata":{},"outputs":[],"execution_count":null},{"id":"4159adba-d15a-4b29-8025-5ef493567887","cell_type":"markdown","source":"## 🔄 Étape 3 — Preprocessing + Augmentation de base","metadata":{}},{"id":"3cd67c04-d95c-471c-a45a-438e0186f210","cell_type":"code","source":"for _, row in tqdm(df.iterrows(), total=len(df)):\n    img_name = row['id_code']\n    label    = row['diagnosis']\n    path     = os.path.join(IMG_PATH, img_name + \".png\")\n\n    img = cv2.imread(path)\n    if img is None:\n        continue\n\n    img = preprocess(img)\n\n    # Image originale préprocessée\n    cv2.imwrite(os.path.join(OUTPUT, str(label), img_name + \".png\"), img)\n\n    # 3 augmentations de base\n    for i, aug_img in enumerate(augment(img)):\n        aug_name = img_name + \"_aug\" + str(i) + \".png\"\n        cv2.imwrite(os.path.join(OUTPUT, str(label), aug_name), aug_img)\n\nprint(\"\\n✅ Preprocessing terminé\")\n\n# Distribution après preprocessing\ncounts = {}\nfor cls in range(5):\n    cls_dir = os.path.join(OUTPUT, str(cls))\n    counts[str(cls)] = len([f for f in os.listdir(cls_dir) if f.endswith(\".png\")])\n\nprint(\"\\nDistribution après preprocessing :\")\nfor cls, n in sorted(counts.items()):\n    bar = '█' * int(n * 25 / max(counts.values()))\n    print(f\"  Classe {cls}: {n:>5} images  {bar}\")\nprint(f\"  TOTAL   : {sum(counts.values())}\")","metadata":{},"outputs":[],"execution_count":null},{"id":"d50d5c8a-d6be-4e4f-b4b5-f44eb86efd38","cell_type":"markdown","source":"## ⚖️ Étape 4 — Balancement (Oversampling uniquement)","metadata":{}},{"id":"8d905c32-72f3-4c85-aff4-f2b63635b89b","cell_type":"code","source":"random.seed(RANDOM_SEED)\nTARGET = max(counts.values())  # cible = classe max → zéro perte\nprint(f\"Cible : {TARGET} images/classe\\n\")\n\nfor cls in range(5):\n    src_dir = os.path.join(OUTPUT, str(cls))\n    dst_dir = os.path.join(OUTPUT_BALANCE, str(cls))\n    os.makedirs(dst_dir, exist_ok=True)\n\n    images = [f for f in os.listdir(src_dir) if f.endswith(\".png\")]\n    n      = len(images)\n    copied = 0\n\n    # 1. Copie TOUS les originaux\n    for fname in images:\n        shutil.copy2(os.path.join(src_dir, fname), os.path.join(dst_dir, fname))\n        copied += 1\n\n    # 2. Oversampling si classe minoritaire\n    if n < TARGET:\n        aug_idx = 0\n        print(f\"Classe {cls} : {n} → {TARGET}  (+{TARGET - n} images)\", end=\" \")\n        while copied < TARGET:\n            src_name = images[aug_idx % n]\n            src_path = os.path.join(src_dir, src_name)\n            img = cv2.imread(src_path)\n            if img is not None:\n                aug_img  = augment(img, seed=aug_idx * 13 + cls * 7)\n                new_name = f\"over_{aug_idx:05d}_{src_name}\"\n                cv2.imwrite(os.path.join(dst_dir, new_name), aug_img)\n            aug_idx += 1\n            copied  += 1\n        print(\"✓\")\n    else:\n        print(f\"Classe {cls} : {n} — majoritaire, copiée intégralement ✓\")\n\nprint(\"\\n✅ Balancement terminé\")","metadata":{},"outputs":[],"execution_count":null},{"id":"2d58aa42-df53-45e9-9552-c8778bf0fc10","cell_type":"markdown","source":"## 📊 Étape 5 — Class Weights","metadata":{}},{"id":"c4a77d6b-cf37-4838-8ace-70c2233a03c6","cell_type":"code","source":"n_total   = sum(counts.values())\nn_classes = 5\nweights   = {}\n\nfor cls in sorted(counts):\n    w = n_total / (n_classes * counts[cls])\n    weights[cls] = round(w, 4)\n\nprint(\"Class Weights (basés sur distribution originale) :\")\nfor cls, w in sorted(weights.items()):\n    bar = '█' * int(w * 8)\n    print(f\"  Classe {cls}: {w:.4f}  {bar}\")\n\n# Sauvegarde JSON\nweights_path = os.path.join(OUTPUT_BALANCE, \"class_weights.json\")\nwith open(weights_path, \"w\") as f:\n    json.dump(weights, f, indent=2)\nprint(f\"\\n💾 Sauvegardés → {weights_path}\")","metadata":{},"outputs":[],"execution_count":null},{"id":"bf639504-ebe1-455c-be5e-aea679fd23bc","cell_type":"markdown","source":"## 🤖 Utilisation dans ton modèle","metadata":{}},{"id":"ce74e529-acc0-4bad-b650-f75f7fabec34","cell_type":"code","source":"w_list = [weights[str(c)] for c in range(5)]\n\n# ── PyTorch ───────────────────────────────────────\n# import torch\n# import torch.nn as nn\n#\n# class_weights = torch.tensor(w_list, dtype=torch.float32).cuda()\n# criterion = nn.CrossEntropyLoss(weight=class_weights)\n\n# ── Keras / TensorFlow ────────────────────────────\n# class_weight = {i: w_list[i] for i in range(5)}\n#\n# model.fit(\n#     train_dataset,\n#     epochs=50,\n#     class_weight=class_weight,\n#     validation_data=val_dataset,\n# )\n\nprint(\"PyTorch weights :\", w_list)\nprint(\"Keras  weights  :\", {i: w_list[i] for i in range(5)})","metadata":{},"outputs":[],"execution_count":null},{"id":"6f81ce7a-297c-4187-8410-1cbf80eef75f","cell_type":"markdown","source":"## ✅ Résumé Final","metadata":{}},{"id":"b4826c30-4b63-4d2b-aed2-33e41eea5621","cell_type":"code","source":"print(\"=\" * 50)\nprint(\"  RÉSUMÉ FINAL\")\nprint(\"=\" * 50)\n\ntotal_balanced = 0\nfor cls in range(5):\n    n = len(os.listdir(os.path.join(OUTPUT_BALANCE, str(cls))))\n    total_balanced += n\n    print(f\"  Classe {cls}: {n:>5} images\")\n\nprint(f\"  {'─'*30}\")\nprint(f\"  TOTAL   : {total_balanced} images\")\nprint(f\"\\n  📁 Preprocessed : {OUTPUT}\")\nprint(f\"  📁 Balanced     : {OUTPUT_BALANCE}\")\nprint(\"=\" * 50)","metadata":{},"outputs":[],"execution_count":null}]}