{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":4104,"databundleVersionId":46661,"sourceType":"competition"},{"sourceId":7866129,"sourceType":"datasetVersion","datasetId":4614938},{"sourceId":7869237,"sourceType":"datasetVersion","datasetId":4617269}],"dockerImageVersionId":30761,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom glob import glob\nfrom PIL import Image\nfrom tqdm import tqdm\n\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\n\nimport warnings\nwarnings.filterwarnings('ignore')\n\nimport torch\nfrom torchvision import transforms\nfrom torch.utils.data import Dataset, DataLoader","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-09T14:41:09.858638Z","iopub.execute_input":"2025-05-09T14:41:09.859016Z","iopub.status.idle":"2025-05-09T14:41:09.864125Z","shell.execute_reply.started":"2025-05-09T14:41:09.858988Z","shell.execute_reply":"2025-05-09T14:41:09.863245Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!unzip /kaggle/input/diabetic-retinopathy-detection/trainLabels.csv.zip -d /kaggle/working/\n!unzip /kaggle/input/diabetic-retinopathy-detection/sampleSubmission.csv.zip -d /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2025-05-09T14:38:19.161914Z","iopub.execute_input":"2025-05-09T14:38:19.162747Z","iopub.status.idle":"2025-05-09T14:38:19.464768Z","shell.execute_reply.started":"2025-05-09T14:38:19.162714Z","shell.execute_reply":"2025-05-09T14:38:19.463880Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_lbl = pd.read_csv('/kaggle/working/trainLabels.csv')\ntrain_img = glob('/kaggle/input/diabetic-retinopathy-train-unzipped/train/*.jpeg')\ntrain_names = [os.path.basename(path).replace('.jpeg', '') for path in train_img]\ntrain_df = pd.DataFrame({'image': train_names, 'image_path': train_img})\ntrain_df = pd.merge(train_lbl, train_df, on='image')\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-09T14:38:23.211238Z","iopub.execute_input":"2025-05-09T14:38:23.211851Z","iopub.status.idle":"2025-05-09T14:38:23.869999Z","shell.execute_reply.started":"2025-05-09T14:38:23.211813Z","shell.execute_reply":"2025-05-09T14:38:23.869199Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"samplesub = pd.read_csv('/kaggle/working/sampleSubmission.csv')\ntest_img = glob('/kaggle/input/diabetic-retinopathy-test-unzipped/test/*.jpeg')\ntest_names = [os.path.basename(path).replace('.jpeg', '') for path in test_img]\ntest_df = pd.DataFrame({'image': test_names, 'image_path': test_img})\ntest_df = pd.merge(samplesub, test_df, on='image')\ntest_df.head()\n","metadata":{"execution":{"iopub.status.busy":"2025-05-09T14:38:27.107661Z","iopub.execute_input":"2025-05-09T14:38:27.107985Z","iopub.status.idle":"2025-05-09T14:38:27.894700Z","shell.execute_reply.started":"2025-05-09T14:38:27.107957Z","shell.execute_reply":"2025-05-09T14:38:27.894094Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig, axes = plt.subplots(nrows=2, ncols=5, figsize=(20, 10))\nax = axes.flatten()\n\nfor i in range(10):\n    row = train_df.sample(10).iloc[i]\n    img = Image.open(row['image_path'])\n    ax[i].imshow(img)\n    ax[i].set_title(f\"Label: {row['level']}\")\n    ax[i].axis('off')\n\nplt.tight_layout()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2025-05-09T14:38:30.794357Z","iopub.execute_input":"2025-05-09T14:38:30.795165Z","iopub.status.idle":"2025-05-09T14:38:39.102808Z","shell.execute_reply.started":"2025-05-09T14:38:30.795130Z","shell.execute_reply":"2025-05-09T14:38:39.102047Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# الإعداد\ntarget_class = 1                # الكلاس اللي عايز تزود صوره\ndesired_total = 5000           # العدد النهائي المطلوب\nimage_size = (224, 224)        # حجم الصورة بعد التحجيم\noutput_suffix = 'aug'          # لاحقة اسم الصورة الجديدة\noutput_dir = '/kaggle/working/augmented_images'\nos.makedirs(output_dir, exist_ok=True)\n\n# التحويلات الهندسية\ngeo_transforms = [\n    transforms.RandomHorizontalFlip(p=1.0),\n    transforms.RandomRotation(degrees=15),\n    transforms.RandomVerticalFlip(p=1.0),\n    transforms.RandomAffine(degrees=0, translate=(0.05, 0.05), scale=(0.95, 1.05)),\n]\nresize = transforms.Resize(image_size)\n\n# تحميل البيانات\ntrain_df = pd.read_csv('/kaggle/working/trainLabels.csv')\nimage_dir = '/kaggle/input/diabetic-retinopathy-train-unzipped/train/'\ntrain_df['image_path'] = train_df['image'].apply(lambda x: os.path.join(image_dir, f\"{x}.jpeg\"))\ntarget_df = train_df[train_df['level'] == target_class]\n\n# حساب عدد الصور المطلوبة\ncurrent_count = len(target_df)\nneeded_augmented = desired_total - current_count\nif needed_augmented <= 0:\n    print(\"✅ عدد الصور كافي بالفعل.\")\nelse:\n    augmentations_per_image = max(1, needed_augmented // current_count)\n    extra = needed_augmented % current_count  # نوزع الزيادة على بعض الصور\n\n    print(f\"🧮 عدد الصور الأصلية: {current_count}\")\n    print(f\"🛠️ عدد النسخ لكل صورة: {augmentations_per_image}\")\n    print(f\"➕ عدد الصور الإضافية للتعويض: {extra}\")\n\n    # تنفيذ التوليد\n    for idx, (_, row) in enumerate(tqdm(target_df.iterrows(), total=current_count)):\n        img_path = row['image_path']\n        image_name = os.path.splitext(os.path.basename(img_path))[0]\n\n        try:\n            image = Image.open(img_path).convert('RGB')\n            image = resize(image)\n\n            # توليد نسخ معززة\n            for i in range(augmentations_per_image):\n                aug = geo_transforms[(i + idx) % len(geo_transforms)]\n                aug_image = aug(image)\n                aug_image_name = f\"{image_name}_{output_suffix}{i}.jpeg\"\n                aug_image.save(os.path.join(output_dir, aug_image_name))\n\n            # لو لسه فاضل extra صور نولدها لبعض الصور\n            if idx < extra:\n                aug = geo_transforms[(augmentations_per_image + idx) % len(geo_transforms)]\n                aug_image = aug(image)\n                aug_image_name = f\"{image_name}_{output_suffix}_extra.jpeg\"\n                aug_image.save(os.path.join(output_dir, aug_image_name))\n\n        except Exception as e:\n            print(f\"❌ خطأ في الصورة {img_path}: {e}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-09T14:54:14.380031Z","iopub.execute_input":"2025-05-09T14:54:14.380784Z","iopub.status.idle":"2025-05-09T14:59:27.773945Z","shell.execute_reply.started":"2025-05-09T14:54:14.380744Z","shell.execute_reply":"2025-05-09T14:59:27.772963Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# الإعداد\ntarget_class = 3                # الكلاس اللي عايز تزود صوره\ndesired_total = 5000           # العدد النهائي المطلوب\nimage_size = (224, 224)        # حجم الصورة بعد التحجيم\noutput_suffix = 'aug'          # لاحقة اسم الصورة الجديدة\noutput_dir = '/kaggle/working/augmented_images'\nos.makedirs(output_dir, exist_ok=True)\n\n# التحويلات الهندسية\ngeo_transforms = [\n    transforms.RandomHorizontalFlip(p=1.0),\n    transforms.RandomRotation(degrees=15),\n    transforms.RandomVerticalFlip(p=1.0),\n    transforms.RandomAffine(degrees=0, translate=(0.05, 0.05), scale=(0.95, 1.05)),\n]\nresize = transforms.Resize(image_size)\n\n# تحميل البيانات\ntrain_df = pd.read_csv('/kaggle/working/trainLabels.csv')\nimage_dir = '/kaggle/input/diabetic-retinopathy-train-unzipped/train/'\ntrain_df['image_path'] = train_df['image'].apply(lambda x: os.path.join(image_dir, f\"{x}.jpeg\"))\ntarget_df = train_df[train_df['level'] == target_class]\n\n# حساب عدد الصور المطلوبة\ncurrent_count = len(target_df)\nneeded_augmented = desired_total - current_count\nif needed_augmented <= 0:\n    print(\"✅ عدد الصور كافي بالفعل.\")\nelse:\n    augmentations_per_image = max(1, needed_augmented // current_count)\n    extra = needed_augmented % current_count  # نوزع الزيادة على بعض الصور\n\n    print(f\"🧮 عدد الصور الأصلية: {current_count}\")\n    print(f\"🛠️ عدد النسخ لكل صورة: {augmentations_per_image}\")\n    print(f\"➕ عدد الصور الإضافية للتعويض: {extra}\")\n\n    # تنفيذ التوليد\n    for idx, (_, row) in enumerate(tqdm(target_df.iterrows(), total=current_count)):\n        img_path = row['image_path']\n        image_name = os.path.splitext(os.path.basename(img_path))[0]\n\n        try:\n            image = Image.open(img_path).convert('RGB')\n            image = resize(image)\n\n            # توليد نسخ معززة\n            for i in range(augmentations_per_image):\n                aug = geo_transforms[(i + idx) % len(geo_transforms)]\n                aug_image = aug(image)\n                aug_image_name = f\"{image_name}_{output_suffix}{i}.jpeg\"\n                aug_image.save(os.path.join(output_dir, aug_image_name))\n\n            # لو لسه فاضل extra صور نولدها لبعض الصور\n            if idx < extra:\n                aug = geo_transforms[(augmentations_per_image + idx) % len(geo_transforms)]\n                aug_image = aug(image)\n                aug_image_name = f\"{image_name}_{output_suffix}_extra.jpeg\"\n                aug_image.save(os.path.join(output_dir, aug_image_name))\n\n        except Exception as e:\n            print(f\"❌ خطأ في الصورة {img_path}: {e}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-09T14:59:32.508607Z","iopub.execute_input":"2025-05-09T14:59:32.509017Z","iopub.status.idle":"2025-05-09T15:01:52.490431Z","shell.execute_reply.started":"2025-05-09T14:59:32.508985Z","shell.execute_reply":"2025-05-09T15:01:52.489688Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# الإعداد\ntarget_class = 4                # الكلاس اللي عايز تزود صوره\ndesired_total = 5000           # العدد النهائي المطلوب\nimage_size = (224, 224)        # حجم الصورة بعد التحجيم\noutput_suffix = 'aug'          # لاحقة اسم الصورة الجديدة\noutput_dir = '/kaggle/working/augmented_images'\nos.makedirs(output_dir, exist_ok=True)\n\n# التحويلات الهندسية\ngeo_transforms = [\n    transforms.RandomHorizontalFlip(p=1.0),\n    transforms.RandomRotation(degrees=15),\n    transforms.RandomVerticalFlip(p=1.0),\n    transforms.RandomAffine(degrees=0, translate=(0.05, 0.05), scale=(0.95, 1.05)),\n]\nresize = transforms.Resize(image_size)\n\n# تحميل البيانات\ntrain_df = pd.read_csv('/kaggle/working/trainLabels.csv')\nimage_dir = '/kaggle/input/diabetic-retinopathy-train-unzipped/train/'\ntrain_df['image_path'] = train_df['image'].apply(lambda x: os.path.join(image_dir, f\"{x}.jpeg\"))\ntarget_df = train_df[train_df['level'] == target_class]\n\n# حساب عدد الصور المطلوبة\ncurrent_count = len(target_df)\nneeded_augmented = desired_total - current_count\nif needed_augmented <= 0:\n    print(\"✅ عدد الصور كافي بالفعل.\")\nelse:\n    augmentations_per_image = max(1, needed_augmented // current_count)\n    extra = needed_augmented % current_count  # نوزع الزيادة على بعض الصور\n\n    print(f\"🧮 عدد الصور الأصلية: {current_count}\")\n    print(f\"🛠️ عدد النسخ لكل صورة: {augmentations_per_image}\")\n    print(f\"➕ عدد الصور الإضافية للتعويض: {extra}\")\n\n    # تنفيذ التوليد\n    for idx, (_, row) in enumerate(tqdm(target_df.iterrows(), total=current_count)):\n        img_path = row['image_path']\n        image_name = os.path.splitext(os.path.basename(img_path))[0]\n\n        try:\n            image = Image.open(img_path).convert('RGB')\n            image = resize(image)\n\n            # توليد نسخ معززة\n            for i in range(augmentations_per_image):\n                aug = geo_transforms[(i + idx) % len(geo_transforms)]\n                aug_image = aug(image)\n                aug_image_name = f\"{image_name}_{output_suffix}{i}.jpeg\"\n                aug_image.save(os.path.join(output_dir, aug_image_name))\n\n            # لو لسه فاضل extra صور نولدها لبعض الصور\n            if idx < extra:\n                aug = geo_transforms[(augmentations_per_image + idx) % len(geo_transforms)]\n                aug_image = aug(image)\n                aug_image_name = f\"{image_name}_{output_suffix}_extra.jpeg\"\n                aug_image.save(os.path.join(output_dir, aug_image_name))\n\n        except Exception as e:\n            print(f\"❌ خطأ في الصورة {img_path}: {e}\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-09T15:02:00.515651Z","iopub.execute_input":"2025-05-09T15:02:00.516365Z","iopub.status.idle":"2025-05-09T15:03:56.511121Z","shell.execute_reply.started":"2025-05-09T15:02:00.516328Z","shell.execute_reply":"2025-05-09T15:03:56.510394Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}