{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":14774,"databundleVersionId":875431,"sourceType":"competition"}],"dockerImageVersionId":31260,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-01-23T23:11:57.116373Z","iopub.execute_input":"2026-01-23T23:11:57.116801Z","iopub.status.idle":"2026-01-23T23:11:58.382223Z","shell.execute_reply.started":"2026-01-23T23:11:57.116773Z","shell.execute_reply":"2026-01-23T23:11:58.381608Z"},"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms\nfrom PIL import Image\nimport timm\nimport random\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-23T23:11:58.383316Z","iopub.execute_input":"2026-01-23T23:11:58.383542Z","iopub.status.idle":"2026-01-23T23:12:09.255624Z","shell.execute_reply.started":"2026-01-23T23:11:58.383521Z","shell.execute_reply":"2026-01-23T23:12:09.254887Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE_PATH = \"/kaggle/input/aptos2019-blindness-detection\"\nTRAIN_IMG_PATH = os.path.join(BASE_PATH, \"train_images\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-23T23:12:09.256554Z","iopub.execute_input":"2026-01-23T23:12:09.256842Z","iopub.status.idle":"2026-01-23T23:12:09.260586Z","shell.execute_reply.started":"2026-01-23T23:12:09.256806Z","shell.execute_reply":"2026-01-23T23:12:09.259893Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class SimCLRAugment:\n    def __init__(self, size=224):\n        self.transform = transforms.Compose([\n            transforms.RandomResizedCrop(size),\n            transforms.RandomHorizontalFlip(),\n            transforms.RandomApply([\n                transforms.ColorJitter(0.4, 0.4, 0.4, 0.1)\n            ], p=0.8),\n            transforms.RandomGrayscale(p=0.2),\n            transforms.ToTensor(),\n            transforms.Normalize(mean=[0.5]*3, std=[0.5]*3)\n        ])\n\n    def __call__(self, x):\n        return self.transform(x), self.transform(x)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-23T23:12:09.262347Z","iopub.execute_input":"2026-01-23T23:12:09.262621Z","iopub.status.idle":"2026-01-23T23:12:09.272782Z","shell.execute_reply.started":"2026-01-23T23:12:09.262596Z","shell.execute_reply":"2026-01-23T23:12:09.272098Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class UnlabeledImageDataset(Dataset):\n    def __init__(self, img_dir, transform):\n        self.img_dir = img_dir\n        self.images = os.listdir(img_dir)\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.images)\n\n    def __getitem__(self, idx):\n        img_path = os.path.join(self.img_dir, self.images[idx])\n        image = Image.open(img_path).convert(\"RGB\")\n        x1, x2 = self.transform(image)\n        return x1, x2\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-23T23:12:09.273643Z","iopub.execute_input":"2026-01-23T23:12:09.273919Z","iopub.status.idle":"2026-01-23T23:12:09.283308Z","shell.execute_reply.started":"2026-01-23T23:12:09.273892Z","shell.execute_reply":"2026-01-23T23:12:09.282663Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class SimCLR(nn.Module):\n    def __init__(self, encoder_name=\"vit_base_patch16_224\", proj_dim=128):\n        super().__init__()\n\n        self.encoder = timm.create_model(\n            encoder_name,\n            pretrained=False,\n            num_classes=0  #  Encoder \n        )\n\n        feat_dim = self.encoder.num_features\n\n        self.projector = nn.Sequential(\n            nn.Linear(feat_dim, 512),\n            nn.ReLU(),\n            nn.Linear(512, proj_dim)\n        )\n\n    def forward(self, x):\n        h = self.encoder(x)\n        z = self.projector(h)\n        return z\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-23T23:12:09.284095Z","iopub.execute_input":"2026-01-23T23:12:09.284551Z","iopub.status.idle":"2026-01-23T23:12:09.295563Z","shell.execute_reply.started":"2026-01-23T23:12:09.284530Z","shell.execute_reply":"2026-01-23T23:12:09.295005Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def nt_xent_loss(z1, z2, temperature=0.5):\n    z1 = nn.functional.normalize(z1, dim=1)\n    z2 = nn.functional.normalize(z2, dim=1)\n\n    batch_size = z1.size(0)\n    z = torch.cat([z1, z2], dim=0)\n\n    similarity = torch.matmul(z, z.T) / temperature\n    labels = torch.arange(batch_size).to(z.device)\n    labels = torch.cat([labels + batch_size, labels])\n\n    mask = torch.eye(2 * batch_size, device=z.device).bool()\n    similarity.masked_fill_(mask, -9e15)\n\n    loss = nn.CrossEntropyLoss()(similarity, labels)\n    return loss\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-23T23:12:09.296293Z","iopub.execute_input":"2026-01-23T23:12:09.296516Z","iopub.status.idle":"2026-01-23T23:12:09.308050Z","shell.execute_reply.started":"2026-01-23T23:12:09.296492Z","shell.execute_reply":"2026-01-23T23:12:09.307458Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = \"cuda\" if torch.cuda.is_available() else \"cpu\"\n\ndataset = UnlabeledImageDataset(\n    TRAIN_IMG_PATH,\n    transform=SimCLRAugment()\n)\n\nloader = DataLoader(dataset, batch_size=32, shuffle=True, num_workers=2)\n\nmodel = SimCLR().to(device)\noptimizer = optim.Adam(model.parameters(), lr=3e-4)\n\nepochs = 10  \n\n\n\nfor epoch in range(epochs):\n    model.train()\n    total_loss = 0\n    num_batches = len(loader)\n\n    for batch_idx, (x1, x2) in enumerate(loader):\n        x1, x2 = x1.to(device), x2.to(device)\n\n        z1 = model(x1)\n        z2 = model(x2)\n\n        loss = nt_xent_loss(z1, z2)\n\n        optimizer.zero_grad()\n        loss.backward()\n        optimizer.step()\n\n        total_loss += loss.item()\n\n       \n        progress = (batch_idx + 1) / num_batches * 100\n\n    \n        if (batch_idx + 1) % max(1, num_batches // 10) == 0:\n            print(\n                f\"Epoch [{epoch+1}/{epochs}] \"\n                f\"- Progress: {progress:.1f}% \"\n                f\"- Batch Loss: {loss.item():.4f}\"\n            )\n\n    print(\n        f\"✅ Epoch [{epoch+1}/{epochs}] Completed \"\n        f\"- Avg Loss: {total_loss/num_batches:.4f}\\n\"\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-23T23:14:01.099178Z","iopub.execute_input":"2026-01-23T23:14:01.099541Z","iopub.status.idle":"2026-01-24T00:04:13.955325Z","shell.execute_reply.started":"2026-01-23T23:14:01.099507Z","shell.execute_reply":"2026-01-24T00:04:13.954481Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ntorch.save(model.encoder.state_dict(), \"simclr_encoder.pth\")\nprint(\"Encoder saved successfully!\")\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-01-24T00:06:56.949086Z","iopub.execute_input":"2026-01-24T00:06:56.949980Z","iopub.status.idle":"2026-01-24T00:06:57.307008Z","shell.execute_reply.started":"2026-01-24T00:06:56.949944Z","shell.execute_reply":"2026-01-24T00:06:57.306322Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}