{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":30919,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-02-24T23:14:01.841869Z","iopub.execute_input":"2025-02-24T23:14:01.842141Z","iopub.status.idle":"2025-02-24T23:14:10.747966Z","shell.execute_reply.started":"2025-02-24T23:14:01.842118Z","shell.execute_reply":"2025-02-24T23:14:10.740401Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torchvision.transforms as transforms\nfrom torch.utils.data import DataLoader, Dataset\nfrom transformers import ViTForImageClassification, ViTFeatureExtractor\nfrom PIL import Image\nimport os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-24T23:14:10.748802Z","iopub.execute_input":"2025-02-24T23:14:10.749170Z","iopub.status.idle":"2025-02-24T23:14:31.240446Z","shell.execute_reply.started":"2025-02-24T23:14:10.749136Z","shell.execute_reply":"2025-02-24T23:14:31.239446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load dataset (Example: APTOS 2019 from Kaggle)\ndata_path = \"/kaggle/input/aptos2019-blindness-detection/\"\ntrain_csv = os.path.join(data_path, \"train.csv\")\ntest_csv = os.path.join(data_path, \"test.csv\")\ntrain_images_dir = os.path.join(data_path, \"train_images\")\ntest_images_dir = os.path.join(data_path, \"test_images\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-24T23:14:31.241484Z","iopub.execute_input":"2025-02-24T23:14:31.242041Z","iopub.status.idle":"2025-02-24T23:14:31.246187Z","shell.execute_reply.started":"2025-02-24T23:14:31.242009Z","shell.execute_reply":"2025-02-24T23:14:31.245433Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load CSV files\ntrain_df = pd.read_csv(train_csv)\ntest_df = pd.read_csv(test_csv)\n\n# Display dataset information\nprint(\"Train Data:\")\nprint(train_df.head())\nprint(\"\\nTest Data:\")\nprint(test_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-24T23:14:31.248407Z","iopub.execute_input":"2025-02-24T23:14:31.248749Z","iopub.status.idle":"2025-02-24T23:14:31.314031Z","shell.execute_reply.started":"2025-02-24T23:14:31.248716Z","shell.execute_reply":"2025-02-24T23:14:31.313293Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot class distribution\nsns.countplot(x=train_df['diagnosis'])\nplt.title(\"Distribution of Diabetic Retinopathy Stages\")\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-24T23:14:31.315043Z","iopub.execute_input":"2025-02-24T23:14:31.315258Z","iopub.status.idle":"2025-02-24T23:14:31.553440Z","shell.execute_reply.started":"2025-02-24T23:14:31.315240Z","shell.execute_reply":"2025-02-24T23:14:31.552561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to visualize images\ndef show_images(image_paths, labels=None, cols=5):\n    rows = len(image_paths) // cols + 1\n    fig, axes = plt.subplots(rows, cols, figsize=(15, 5))\n    axes = axes.flatten()\n    for idx, img_path in enumerate(image_paths):\n        img = Image.open(img_path)\n        axes[idx].imshow(img)\n        if labels is not None:\n            axes[idx].set_title(f\"Stage: {labels[idx]}\")\n        axes[idx].axis(\"off\")\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-24T23:14:31.554336Z","iopub.execute_input":"2025-02-24T23:14:31.554662Z","iopub.status.idle":"2025-02-24T23:14:31.559651Z","shell.execute_reply.started":"2025-02-24T23:14:31.554630Z","shell.execute_reply":"2025-02-24T23:14:31.558767Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Show sample images\nsample_images = train_df.sample(10)\nimage_paths = [os.path.join(train_images_dir, img + \".png\") for img in sample_images['id_code']]\nlabels = sample_images['diagnosis'].tolist()\nshow_images(image_paths, labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-24T23:14:31.560468Z","iopub.execute_input":"2025-02-24T23:14:31.560736Z","iopub.status.idle":"2025-02-24T23:14:39.310742Z","shell.execute_reply.started":"2025-02-24T23:14:31.560714Z","shell.execute_reply":"2025-02-24T23:14:39.309860Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Data preprocessing\nfeature_extractor = ViTFeatureExtractor.from_pretrained(\"google/vit-base-patch16-224\")\ntransform = transforms.Compose([\n    transforms.Resize((224, 224)),\n    transforms.ToTensor(),\n    transforms.Normalize(mean=feature_extractor.image_mean, std=feature_extractor.image_std)\n])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-24T23:14:39.311775Z","iopub.execute_input":"2025-02-24T23:14:39.312095Z","iopub.status.idle":"2025-02-24T23:14:39.575823Z","shell.execute_reply.started":"2025-02-24T23:14:39.312067Z","shell.execute_reply":"2025-02-24T23:14:39.574719Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load data\nclass DRDataset(Dataset):\n    def __init__(self, csv_file, img_dir, transform=None):\n        self.labels = pd.read_csv(csv_file)\n        self.img_dir = img_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.labels)\n\n    def __getitem__(self, idx):\n        img_path = os.path.join(self.img_dir, self.labels.iloc[idx, 0] + \".png\")\n        image = Image.open(img_path).convert(\"RGB\")\n        label = self.labels.iloc[idx, 1]  # Class label (0 to 4)\n\n        if self.transform:\n            image = self.transform(image)\n        \n        return image, label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-24T23:14:39.576806Z","iopub.execute_input":"2025-02-24T23:14:39.577031Z","iopub.status.idle":"2025-02-24T23:14:39.583408Z","shell.execute_reply.started":"2025-02-24T23:14:39.577000Z","shell.execute_reply":"2025-02-24T23:14:39.582486Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dataset = DRDataset(csv_file=train_csv, img_dir=train_images_dir, transform=transform)\nval_dataset = DRDataset(csv_file=test_csv, img_dir=test_images_dir, transform=transform)\n\ntrain_loader = DataLoader(train_dataset, batch_size=16, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=16, shuffle=False)\n\n# Load Vision Transformer model\nmodel = ViTForImageClassification.from_pretrained(\n    \"google/vit-base-patch16-224\",\n    num_labels=5,\n    ignore_mismatched_sizes=True\n)\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\nmodel.to(device)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-24T23:15:10.231362Z","iopub.execute_input":"2025-02-24T23:15:10.231675Z","iopub.status.idle":"2025-02-24T23:15:11.895294Z","shell.execute_reply.started":"2025-02-24T23:15:10.231643Z","shell.execute_reply":"2025-02-24T23:15:11.894284Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from PIL import Image\nimport os\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport plotly.express as px\nimport cv2\nimport numpy as np","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-24T23:19:36.332934Z","iopub.execute_input":"2025-02-24T23:19:36.333220Z","iopub.status.idle":"2025-02-24T23:19:37.379012Z","shell.execute_reply.started":"2025-02-24T23:19:36.333199Z","shell.execute_reply":"2025-02-24T23:19:37.378296Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot class distribution using seaborn\nsns.countplot(x=train_df['diagnosis'])\nplt.title(\"Distribution of Diabetic Retinopathy Stages\")\nplt.show()\n\n# Interactive class distribution using plotly\nfig = px.histogram(train_df, x='diagnosis', title=\"Distribution of Diabetic Retinopathy Stages\")\nfig.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-24T23:19:39.102888Z","iopub.execute_input":"2025-02-24T23:19:39.103170Z","iopub.status.idle":"2025-02-24T23:19:40.995275Z","shell.execute_reply.started":"2025-02-24T23:19:39.103149Z","shell.execute_reply":"2025-02-24T23:19:40.994340Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Object Detection - Extract ROI using OpenCV\ndef extract_roi(image_path):\n    img = cv2.imread(image_path)\n    gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    _, thresh = cv2.threshold(gray, 30, 255, cv2.THRESH_BINARY_INV)\n    contours, _ = cv2.findContours(thresh, cv2.RETR_EXTERNAL, cv2.CHAIN_APPROX_SIMPLE)\n    \n    for contour in contours:\n        x, y, w, h = cv2.boundingRect(contour)\n        cv2.rectangle(img, (x, y), (x + w, y + h), (255, 0, 0), 2)\n    \n    plt.imshow(cv2.cvtColor(img, cv2.COLOR_BGR2RGB))\n    plt.title(\"Detected ROI\")\n    plt.axis(\"off\")\n    plt.show()\n\n# Show ROI extraction on sample image\nextract_roi(image_paths[0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-24T23:19:47.118028Z","iopub.execute_input":"2025-02-24T23:19:47.118380Z","iopub.status.idle":"2025-02-24T23:19:47.822647Z","shell.execute_reply.started":"2025-02-24T23:19:47.118346Z","shell.execute_reply":"2025-02-24T23:19:47.821816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to visualize images using OpenCV\ndef show_images_opencv(image_paths, labels=None):\n    for idx, img_path in enumerate(image_paths):\n        img = cv2.imread(img_path)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        plt.imshow(img)\n        if labels is not None:\n            plt.title(f\"Stage: {labels[idx]}\")\n        plt.axis(\"off\")\n        plt.show()\n\n# Show sample images\nsample_images = train_df.sample(5)\nimage_paths = [os.path.join(train_images_dir, img + \".png\") for img in sample_images['id_code']]\nlabels = sample_images['diagnosis'].tolist()\nshow_images_opencv(image_paths, labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-02-24T23:20:28.726290Z","iopub.execute_input":"2025-02-24T23:20:28.726660Z","iopub.status.idle":"2025-02-24T23:20:33.755542Z","shell.execute_reply.started":"2025-02-24T23:20:28.726631Z","shell.execute_reply":"2025-02-24T23:20:33.754678Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}