{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceType":"competition","sourceId":22990,"datasetId":1136396,"databundleVersionId":2048213},{"sourceType":"datasetVersion","sourceId":15088959,"datasetId":9660413,"databundleVersionId":15973037},{"sourceType":"modelInstanceVersion","sourceId":777819,"databundleVersionId":15974343,"modelInstanceId":593626,"modelId":605901}],"dockerImageVersionId":31286,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input/models/smohanesh/model/other/default/'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-03-08T08:41:52.007097Z","iopub.execute_input":"2026-03-08T08:41:52.007433Z","iopub.status.idle":"2026-03-08T08:41:52.022276Z","shell.execute_reply.started":"2026-03-08T08:41:52.007407Z","shell.execute_reply":"2026-03-08T08:41:52.020561Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install imagecodecs -q","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T07:36:03.635667Z","iopub.execute_input":"2026-03-08T07:36:03.636184Z","iopub.status.idle":"2026-03-08T07:36:10.479452Z","shell.execute_reply.started":"2026-03-08T07:36:03.636158Z","shell.execute_reply":"2026-03-08T07:36:10.478249Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom PIL import Image\nimport tifffile as tiff\nimport cv2\nimport os\nfrom tqdm.notebook import tqdm\nimport zipfile\nimport rasterio\nfrom rasterio.windows import Window\nfrom torch.utils.data import Dataset\nimport gc","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T07:36:10.481206Z","iopub.execute_input":"2026-03-08T07:36:10.481511Z","iopub.status.idle":"2026-03-08T07:36:16.991592Z","shell.execute_reply.started":"2026-03-08T07:36:10.481484Z","shell.execute_reply":"2026-03-08T07:36:16.990606Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sz = 256   #the size of tiles\nreduce = 4 #reduce the original images by 4 times \nMASKS = '/kaggle/input/competitions/hubmap-kidney-segmentation/train.csv'\nDATA = '/kaggle/input/competitions/hubmap-kidney-segmentation/train/'\nOUT_TRAIN = 'train.zip'\nOUT_MASKS = 'masks.zip'","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:44:38.737050Z","iopub.execute_input":"2026-03-04T09:44:38.737474Z","iopub.status.idle":"2026-03-04T09:44:38.741927Z","shell.execute_reply.started":"2026-03-04T09:44:38.737438Z","shell.execute_reply":"2026-03-04T09:44:38.741099Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def enc2mask(encs, shape):\n    img = np.zeros(shape[0]*shape[1], dtype=np.uint8)\n    for m,enc in enumerate(encs):\n        if isinstance(enc, float) and np.isnan(enc): continue\n        s = enc.split()\n        for i in range(len(s)//2):\n            start = int(s[2*i]) - 1\n            length = int(s[2*i+1])\n            img[start:start+length] = 1 + m\n    return img.reshape(shape).T\n\ndef mask2enc(mask, n=1):\n    pixels = mask.T.flatten()\n    encs = []\n    for i in range(1,n+1):\n        p = (pixels == i).astype(np.int8)\n        if p.sum() == 0: encs.append(np.nan)\n        else:\n            p = np.concatenate([[0], p, [0]])\n            runs = np.where(p[1:] != p[:-1])[0] + 1\n            runs[1::2] -= runs[::2]\n            encs.append(' '.join(str(x) for x in runs))\n    return encs\n\ndf_masks = pd.read_csv(MASKS).set_index('id')\ndf_masks.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:44:38.742920Z","iopub.execute_input":"2026-03-04T09:44:38.743201Z","iopub.status.idle":"2026-03-04T09:44:39.143321Z","shell.execute_reply.started":"2026-03-04T09:44:38.743166Z","shell.execute_reply":"2026-03-04T09:44:39.142548Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"s_th = 40  #saturation blancking threshold\np_th = 1000*(sz//256)**2 #threshold for the minimum number of pixels\n\n\nclass HuBMAPDataset(Dataset):\n    def __init__(self, idx, sz=sz, reduce=reduce, encs=None):\n        self.data = rasterio.open(os.path.join(DATA,idx+'.tiff'),num_threads='all_cpus')\n        # some images have issues with their format \n        # and must be saved correctly before reading with rasterio\n        if self.data.count != 3:\n            subdatasets = self.data.subdatasets\n            self.layers = []\n            if len(subdatasets) > 0:\n                for i, subdataset in enumerate(subdatasets, 0):\n                    self.layers.append(rasterio.open(subdataset))\n        self.shape = self.data.shape\n        self.reduce = reduce\n        self.sz = reduce*sz\n        self.pad0 = (self.sz - self.shape[0]%self.sz)%self.sz\n        self.pad1 = (self.sz - self.shape[1]%self.sz)%self.sz\n        self.n0max = (self.shape[0] + self.pad0)//self.sz\n        self.n1max = (self.shape[1] + self.pad1)//self.sz\n        self.mask = enc2mask(encs,(self.shape[1],self.shape[0])) if encs is not None else None\n        \n    def __len__(self):\n        return self.n0max*self.n1max\n    \n    def __getitem__(self, idx):\n        # the code below may be a little bit difficult to understand,\n        # but the thing it does is mapping the original image to\n        # tiles created with adding padding (like in the previous version of the kernel)\n        # then the tiles are loaded with rasterio\n        # n0,n1 - are the x and y index of the tile (idx = n0*self.n1max + n1)\n        n0,n1 = idx//self.n1max, idx%self.n1max\n        # x0,y0 - are the coordinates of the lower left corner of the tile in the image\n        # negative numbers correspond to padding (which must not be loaded)\n        x0,y0 = -self.pad0//2 + n0*self.sz, -self.pad1//2 + n1*self.sz\n\n        # make sure that the region to read is within the image\n        p00,p01 = max(0,x0), min(x0+self.sz,self.shape[0])\n        p10,p11 = max(0,y0), min(y0+self.sz,self.shape[1])\n        img = np.zeros((self.sz,self.sz,3),np.uint8)\n        mask = np.zeros((self.sz,self.sz),np.uint8)\n        # mapping the loade region to the tile\n        if self.data.count == 3:\n            img[(p00-x0):(p01-x0),(p10-y0):(p11-y0)] = np.moveaxis(self.data.read([1,2,3],\n                window=Window.from_slices((p00,p01),(p10,p11))), 0, -1)\n        else:\n            for i,layer in enumerate(self.layers):\n                img[(p00-x0):(p01-x0),(p10-y0):(p11-y0),i] =\\\n                  layer.read(1,window=Window.from_slices((p00,p01),(p10,p11)))\n        if self.mask is not None: mask[(p00-x0):(p01-x0),(p10-y0):(p11-y0)] = self.mask[p00:p01,p10:p11]\n        \n        if self.reduce != 1:\n            img = cv2.resize(img,(self.sz//reduce,self.sz//reduce),\n                             interpolation = cv2.INTER_AREA)\n            mask = cv2.resize(mask,(self.sz//reduce,self.sz//reduce),\n                             interpolation = cv2.INTER_NEAREST)\n        #check for empty imges\n        hsv = cv2.cvtColor(img, cv2.COLOR_BGR2HSV)\n        h,s,v = cv2.split(hsv)\n        #return -1 for empty images\n        return img, mask, (-1 if (s>s_th).sum() <= p_th or img.sum() <= p_th else idx)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:44:39.144407Z","iopub.execute_input":"2026-03-04T09:44:39.145109Z","iopub.status.idle":"2026-03-04T09:44:39.158542Z","shell.execute_reply.started":"2026-03-04T09:44:39.145081Z","shell.execute_reply":"2026-03-04T09:44:39.157783Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"x_tot,x2_tot = [],[]\nwith zipfile.ZipFile(OUT_TRAIN, 'w') as img_out,\\\n zipfile.ZipFile(OUT_MASKS, 'w') as mask_out:\n    for index, encs in tqdm(df_masks.iterrows(),total=len(df_masks)):\n        #image+mask dataset\n        ds = HuBMAPDataset(index,encs=encs)\n        for i in range(len(ds)):\n            im,m,idx = ds[i]\n            if idx < 0: continue\n                \n            x_tot.append((im/255.0).reshape(-1,3).mean(0))\n            x2_tot.append(((im/255.0)**2).reshape(-1,3).mean(0))\n            \n            #write data   \n            im = cv2.imencode('.png',cv2.cvtColor(im, cv2.COLOR_RGB2BGR))[1]\n            img_out.writestr(f'{index}_{idx:04d}.png', im)\n            m = cv2.imencode('.png',m)[1]\n            mask_out.writestr(f'{index}_{idx:04d}.png', m)\n        \n#image stats\nimg_avr =  np.array(x_tot).mean(0)\nimg_std =  np.sqrt(np.array(x2_tot).mean(0) - img_avr**2)\nprint('mean:',img_avr, ', std:', img_std)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:44:39.159567Z","iopub.execute_input":"2026-03-04T09:44:39.159906Z","iopub.status.idle":"2026-03-04T09:53:13.182688Z","shell.execute_reply.started":"2026-03-04T09:44:39.159867Z","shell.execute_reply":"2026-03-04T09:53:13.181503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"columns, rows = 4,4\nidx0 = 20\nfig=plt.figure(figsize=(columns*4, rows*4))\nwith zipfile.ZipFile(OUT_TRAIN, 'r') as img_arch, \\\n     zipfile.ZipFile(OUT_MASKS, 'r') as msk_arch:\n    fnames = sorted(img_arch.namelist())[8:]\n    for i in range(rows):\n        for j in range(columns):\n            idx = i+j*columns\n            img = cv2.imdecode(np.frombuffer(img_arch.read(fnames[idx0+idx]), \n                                             np.uint8), cv2.IMREAD_COLOR)\n            img = cv2.cvtColor(img, cv2.COLOR_RGB2BGR)\n            mask = cv2.imdecode(np.frombuffer(msk_arch.read(fnames[idx0+idx]), \n                                              np.uint8), cv2.IMREAD_GRAYSCALE)\n    \n            fig.add_subplot(rows, columns, idx+1)\n            plt.axis('off')\n            plt.imshow(Image.fromarray(img))\n            plt.imshow(Image.fromarray(mask), alpha=0.2)\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:53:13.184751Z","iopub.execute_input":"2026-03-04T09:53:13.185126Z","iopub.status.idle":"2026-03-04T09:53:15.264477Z","shell.execute_reply.started":"2026-03-04T09:53:13.185083Z","shell.execute_reply":"2026-03-04T09:53:15.263379Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nfor dirname, _, filenames in os.walk('/kaggle/working'):\n    for filename in filenames[:5]:\n        print(os.path.join(dirname, filename))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:53:15.265443Z","iopub.execute_input":"2026-03-04T09:53:15.265687Z","iopub.status.idle":"2026-03-04T09:53:15.271081Z","shell.execute_reply.started":"2026-03-04T09:53:15.265664Z","shell.execute_reply":"2026-03-04T09:53:15.270106Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install scikit-image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:53:15.273597Z","iopub.execute_input":"2026-03-04T09:53:15.273891Z","iopub.status.idle":"2026-03-04T09:53:19.558379Z","shell.execute_reply.started":"2026-03-04T09:53:15.273866Z","shell.execute_reply":"2026-03-04T09:53:19.557265Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import zipfile\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\n\nfrom skimage.feature import graycomatrix, graycoprops\nfrom skimage.measure import regionprops\nfrom scipy.stats import entropy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:53:19.559927Z","iopub.execute_input":"2026-03-04T09:53:19.560250Z","iopub.status.idle":"2026-03-04T09:53:20.712722Z","shell.execute_reply.started":"2026-03-04T09:53:19.560217Z","shell.execute_reply":"2026-03-04T09:53:20.711886Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_zip = zipfile.ZipFile('/kaggle/working/train.zip')\nmask_zip = zipfile.ZipFile('/kaggle/working/masks.zip')\n\nfnames = sorted(img_zip.namelist())\n\nprint(\"Total patches:\", len(fnames))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:53:20.713917Z","iopub.execute_input":"2026-03-04T09:53:20.714479Z","iopub.status.idle":"2026-03-04T09:53:20.843650Z","shell.execute_reply.started":"2026-03-04T09:53:20.714449Z","shell.execute_reply":"2026-03-04T09:53:20.842697Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_features(img, mask):\n\n    features = {}\n\n    mask = mask.astype(np.uint8)\n\n    # -----------------------------\n    # Remove tiny segmentation noise\n    # -----------------------------\n    if mask.sum() < 200:\n        return None\n\n\n    # -----------------------------\n    # FIND CONTOUR\n    # -----------------------------\n    contours,_ = cv2.findContours(mask,\n                                  cv2.RETR_EXTERNAL,\n                                  cv2.CHAIN_APPROX_SIMPLE)\n\n    if len(contours) == 0:\n        return None\n\n    cnt = contours[0]\n\n\n    # -----------------------------\n    # MORPHOLOGICAL FEATURES\n    # -----------------------------\n\n    # Area (correct way)\n    area = cv2.contourArea(cnt)\n    features[\"area\"] = area\n\n    # Remove extremely small areas\n    if area < 100:\n        return None\n\n\n    # Perimeter\n    perimeter = cv2.arcLength(cnt, True)\n    features[\"perimeter\"] = perimeter\n\n\n    # Circularity\n    if perimeter > 0:\n        circularity = (4 * np.pi * area) / (perimeter ** 2)\n    else:\n        circularity = 0\n\n    # Clamp extreme values\n    circularity = min(circularity, 1.2)\n\n    features[\"circularity\"] = circularity\n\n\n    # Solidity\n    hull = cv2.convexHull(cnt)\n    hull_area = cv2.contourArea(hull)\n\n    if hull_area > 0:\n        solidity = area / hull_area\n    else:\n        solidity = 0\n\n    features[\"solidity\"] = solidity\n\n\n    # -----------------------------\n    # REGION PROPERTIES\n    # -----------------------------\n\n    from skimage.measure import regionprops\n\n    props = regionprops(mask)\n\n    if len(props) > 0:\n\n        prop = props[0]\n\n        features[\"eccentricity\"] = prop.eccentricity\n        features[\"major_axis_length\"] = prop.major_axis_length\n        features[\"minor_axis_length\"] = prop.minor_axis_length\n\n        if prop.minor_axis_length > 0:\n            aspect_ratio = prop.major_axis_length / prop.minor_axis_length\n        else:\n            aspect_ratio = 0\n\n        features[\"aspect_ratio\"] = aspect_ratio\n\n        # Compactness\n        if area > 0:\n            compactness = (perimeter ** 2) / (4 * np.pi * area)\n        else:\n            compactness = 0\n\n        features[\"compactness\"] = compactness\n\n    else:\n\n        features[\"eccentricity\"] = 0\n        features[\"major_axis_length\"] = 0\n        features[\"minor_axis_length\"] = 0\n        features[\"aspect_ratio\"] = 0\n        features[\"compactness\"] = 0\n\n\n    # -----------------------------\n    # INTENSITY FEATURES\n    # -----------------------------\n\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n\n    pixels = gray[mask == 1]\n\n    if len(pixels) == 0:\n        return None\n\n    features[\"mean_intensity\"] = np.mean(pixels)\n    features[\"std_intensity\"] = np.std(pixels)\n\n    from scipy.stats import entropy\n\n    hist,_ = np.histogram(pixels, bins=256, range=(0,255))\n    features[\"entropy\"] = entropy(hist + 1)\n\n\n    # -----------------------------\n    # TEXTURE FEATURES (GLCM)\n    # -----------------------------\n\n    from skimage.feature import graycomatrix, graycoprops\n\n    glcm = graycomatrix(gray,\n                        distances=[1],\n                        angles=[0],\n                        levels=256,\n                        symmetric=True,\n                        normed=True)\n\n    features[\"contrast\"] = graycoprops(glcm,'contrast')[0,0]\n    features[\"homogeneity\"] = graycoprops(glcm,'homogeneity')[0,0]\n    features[\"energy\"] = graycoprops(glcm,'energy')[0,0]\n    features[\"correlation\"] = graycoprops(glcm,'correlation')[0,0]\n\n\n    return features","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:53:20.845743Z","iopub.execute_input":"2026-03-04T09:53:20.846068Z","iopub.status.idle":"2026-03-04T09:53:20.860581Z","shell.execute_reply.started":"2026-03-04T09:53:20.846042Z","shell.execute_reply":"2026-03-04T09:53:20.859704Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"feature_list = []\n\nfor fname in tqdm(fnames):\n\n    # Load image\n    img = cv2.imdecode(\n        np.frombuffer(img_zip.read(fname), np.uint8),\n        cv2.IMREAD_COLOR\n    )\n\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    # Load mask\n    mask = cv2.imdecode(\n        np.frombuffer(mask_zip.read(fname), np.uint8),\n        cv2.IMREAD_GRAYSCALE\n    )\n\n    mask = (mask > 0).astype(np.uint8)\n\n    features = extract_features(img, mask)\n\n    if features is not None:\n        feature_list.append(features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:53:20.861658Z","iopub.execute_input":"2026-03-04T09:53:20.861977Z","iopub.status.idle":"2026-03-04T09:54:04.310126Z","shell.execute_reply.started":"2026-03-04T09:53:20.861942Z","shell.execute_reply":"2026-03-04T09:54:04.309294Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.DataFrame(feature_list)\n\nprint(df.shape)\ndf.head(10)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:54:04.311311Z","iopub.execute_input":"2026-03-04T09:54:04.311619Z","iopub.status.idle":"2026-03-04T09:54:04.344695Z","shell.execute_reply.started":"2026-03-04T09:54:04.311592Z","shell.execute_reply":"2026-03-04T09:54:04.344031Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nsns.histplot(df[\"circularity\"], bins=30)\nplt.title(\"Circularity Distribution\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:54:04.345781Z","iopub.execute_input":"2026-03-04T09:54:04.346227Z","iopub.status.idle":"2026-03-04T09:54:04.842654Z","shell.execute_reply.started":"2026-03-04T09:54:04.346189Z","shell.execute_reply":"2026-03-04T09:54:04.841772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA\nimport matplotlib.pyplot as plt\n\nX = df.values\n\nscaler = StandardScaler()\nX_scaled = scaler.fit_transform(X)\n\npca = PCA(n_components=2)\nX_pca = pca.fit_transform(X_scaled)\n\nplt.figure(figsize=(6,6))\nplt.scatter(X_pca[:,0], X_pca[:,1], s=5)\nplt.title(\"PCA of Glomerulus Features\")\nplt.xlabel(\"PC1\")\nplt.ylabel(\"PC2\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:54:04.843772Z","iopub.execute_input":"2026-03-04T09:54:04.844145Z","iopub.status.idle":"2026-03-04T09:54:05.352390Z","shell.execute_reply.started":"2026-03-04T09:54:04.844109Z","shell.execute_reply":"2026-03-04T09:54:05.351550Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nplt.figure(figsize=(7,6))\n\nsns.scatterplot(\n    data=df,\n    x=\"circularity\",\n    y=\"solidity\",\n    alpha=0.6\n)\n\nplt.title(\"Circularity vs Solidity of Glomeruli\")\nplt.xlabel(\"Circularity\")\nplt.ylabel(\"Solidity\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:54:05.353545Z","iopub.execute_input":"2026-03-04T09:54:05.354314Z","iopub.status.idle":"2026-03-04T09:54:05.536263Z","shell.execute_reply.started":"2026-03-04T09:54:05.354285Z","shell.execute_reply":"2026-03-04T09:54:05.535459Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!apt-get install openslide-tools\n!pip install openslide-python scikit-image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-04T09:54:49.457595Z","iopub.execute_input":"2026-03-04T09:54:49.457945Z","iopub.status.idle":"2026-03-04T09:55:02.149565Z","shell.execute_reply.started":"2026-03-04T09:54:49.457915Z","shell.execute_reply":"2026-03-04T09:55:02.148538Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!apt-get install openslide-tools\n!pip install openslide-python","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T08:56:04.041534Z","iopub.execute_input":"2026-03-08T08:56:04.042293Z","iopub.status.idle":"2026-03-08T08:56:16.702483Z","shell.execute_reply.started":"2026-03-08T08:56:04.042263Z","shell.execute_reply":"2026-03-08T08:56:16.701574Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!apt-get install openslide-tools\n!pip install openslide-python","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import openslide\nimport os\nimport cv2\nimport numpy as np\nfrom tqdm import tqdm","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T08:56:16.703953Z","iopub.execute_input":"2026-03-08T08:56:16.704279Z","iopub.status.idle":"2026-03-08T08:56:17.010585Z","shell.execute_reply.started":"2026-03-08T08:56:16.704243Z","shell.execute_reply":"2026-03-08T08:56:17.009850Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"svs_folder = \"/kaggle/input/datasets/smohanesh/dataset-a\"\nsave_folder = \"/kaggle/working/patches\"\n\nos.makedirs(save_folder, exist_ok=True)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T08:56:26.182734Z","iopub.execute_input":"2026-03-08T08:56:26.183088Z","iopub.status.idle":"2026-03-08T08:56:26.187175Z","shell.execute_reply.started":"2026-03-08T08:56:26.183063Z","shell.execute_reply":"2026-03-08T08:56:26.186593Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def detect_tissue_mask(img):\n\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n\n    # Otsu threshold to separate tissue\n    _, mask = cv2.threshold(gray, 0, 255, cv2.THRESH_BINARY_INV + cv2.THRESH_OTSU)\n\n    # remove small noise\n    kernel = np.ones((5,5), np.uint8)\n    mask = cv2.morphologyEx(mask, cv2.MORPH_CLOSE, kernel)\n\n    return mask","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T08:56:31.152685Z","iopub.execute_input":"2026-03-08T08:56:31.153239Z","iopub.status.idle":"2026-03-08T08:56:31.157606Z","shell.execute_reply.started":"2026-03-08T08:56:31.153210Z","shell.execute_reply":"2026-03-08T08:56:31.156929Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"patch_size = 256\nstride = 256\npatch_id = 0\n\nsvs_files = [f for f in os.listdir(svs_folder) if f.endswith(\".svs\")]\n\nfor file in svs_files:\n\n    slide_path = os.path.join(svs_folder, file)\n    slide = openslide.OpenSlide(slide_path)\n\n    width, height = slide.dimensions\n\n    print(\"Processing:\", file)\n    print(\"Size:\", width, height)\n\n    for y in tqdm(range(0, height - patch_size, stride)):\n        for x in range(0, width - patch_size, stride):\n\n            patch = slide.read_region((x, y), 0, (patch_size, patch_size))\n            patch = np.array(patch)[:,:,:3]\n\n            # detect tissue\n            mask = detect_tissue_mask(patch)\n\n            tissue_ratio = np.sum(mask > 0) / (patch_size * patch_size)\n\n            # skip patches with little tissue\n            if tissue_ratio < 0.20:\n                continue\n\n            save_path = os.path.join(save_folder, f\"patch_{patch_id}.png\")\n\n            cv2.imwrite(save_path, cv2.cvtColor(patch, cv2.COLOR_RGB2BGR))\n\n            patch_id += 1\n\nprint(\"Total useful patches extracted:\", patch_id)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T08:56:34.240544Z","iopub.execute_input":"2026-03-08T08:56:34.241078Z","iopub.status.idle":"2026-03-08T09:26:23.482916Z","shell.execute_reply.started":"2026-03-08T08:56:34.241049Z","shell.execute_reply":"2026-03-08T09:26:23.482235Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install scikit-image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T09:27:06.419208Z","iopub.execute_input":"2026-03-08T09:27:06.419518Z","iopub.status.idle":"2026-03-08T09:27:09.663821Z","shell.execute_reply.started":"2026-03-08T09:27:06.419492Z","shell.execute_reply":"2026-03-08T09:27:09.663155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\nimport cv2\nimport torch\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nfrom skimage.measure import label, regionprops\nfrom skimage.feature import graycomatrix, graycoprops\nfrom scipy.stats import entropy","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T09:27:13.708478Z","iopub.execute_input":"2026-03-08T09:27:13.708798Z","iopub.status.idle":"2026-03-08T09:27:18.125657Z","shell.execute_reply.started":"2026-03-08T09:27:13.708755Z","shell.execute_reply":"2026-03-08T09:27:18.125091Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.nn.functional as F\n\nclass DilatedResBlock(nn.Module):\n    def __init__(self, in_channels, out_channels, dilation=1):\n        super(DilatedResBlock, self).__init__()\n\n        self.conv1 = nn.Conv2d(in_channels, out_channels, 3,\n                               padding=dilation, dilation=dilation)\n        self.bn1 = nn.BatchNorm2d(out_channels)\n\n        self.conv2 = nn.Conv2d(out_channels, out_channels, 3,\n                               padding=dilation, dilation=dilation)\n        self.bn2 = nn.BatchNorm2d(out_channels)\n\n        self.skip = nn.Conv2d(in_channels, out_channels, 1)\n\n    def forward(self, x):\n        identity = self.skip(x)\n\n        out = F.relu(self.bn1(self.conv1(x)))\n        out = self.bn2(self.conv2(out))\n\n        out += identity\n        return F.relu(out)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T09:27:20.345204Z","iopub.execute_input":"2026-03-08T09:27:20.345977Z","iopub.status.idle":"2026-03-08T09:27:20.351934Z","shell.execute_reply.started":"2026-03-08T09:27:20.345947Z","shell.execute_reply":"2026-03-08T09:27:20.351184Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class TransformerBlock(nn.Module):\n    def __init__(self, dim, num_heads=4):\n        super(TransformerBlock, self).__init__()\n\n        self.norm1 = nn.LayerNorm(dim)\n        self.attn = nn.MultiheadAttention(dim, num_heads)\n        self.norm2 = nn.LayerNorm(dim)\n\n        self.mlp = nn.Sequential(\n            nn.Linear(dim, dim*4),\n            nn.GELU(),\n            nn.Linear(dim*4, dim)\n        )\n\n    def forward(self, x):\n        B, C, H, W = x.shape\n\n        x = x.view(B, C, -1).permute(2, 0, 1)  # (HW, B, C)\n\n        x2 = self.norm1(x)\n        attn_out, _ = self.attn(x2, x2, x2)\n        x = x + attn_out\n\n        x2 = self.norm2(x)\n        x = x + self.mlp(x2)\n\n        x = x.permute(1, 2, 0).view(B, C, H, W)\n        return x","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T09:27:22.472911Z","iopub.execute_input":"2026-03-08T09:27:22.473666Z","iopub.status.idle":"2026-03-08T09:27:22.479427Z","shell.execute_reply.started":"2026-03-08T09:27:22.473636Z","shell.execute_reply":"2026-03-08T09:27:22.478708Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class ADCResTransXNet(nn.Module):\n    def __init__(self):\n        super(ADCResTransXNet, self).__init__()\n\n        # Encoder\n        self.enc1 = DilatedResBlock(3, 64, dilation=1)\n        self.pool1 = nn.MaxPool2d(2)\n\n        self.enc2 = DilatedResBlock(64, 128, dilation=2)\n        self.pool2 = nn.MaxPool2d(2)\n\n        self.enc3 = DilatedResBlock(128, 256, dilation=4)\n        self.pool3 = nn.MaxPool2d(2)\n\n        # Transformer\n        self.transformer = TransformerBlock(256)\n\n        # Decoder (FIXED CHANNELS)\n\n        self.up1 = nn.ConvTranspose2d(256, 128, 2, stride=2)\n        self.dec1 = DilatedResBlock(384, 128)   # 128 + 256 = 384\n\n        self.up2 = nn.ConvTranspose2d(128, 64, 2, stride=2)\n        self.dec2 = DilatedResBlock(192, 64)    # 64 + 128 = 192\n\n        self.up3 = nn.ConvTranspose2d(64, 32, 2, stride=2)\n        self.dec3 = DilatedResBlock(96, 32)     # 32 + 64 = 96\n\n        self.final = nn.Conv2d(32, 1, kernel_size=1)\n\n    def forward(self, x):\n\n        e1 = self.enc1(x)\n        p1 = self.pool1(e1)\n\n        e2 = self.enc2(p1)\n        p2 = self.pool2(e2)\n\n        e3 = self.enc3(p2)\n        p3 = self.pool3(e3)\n\n        bottleneck = self.transformer(p3)\n\n        d1 = self.up1(bottleneck)\n        d1 = torch.cat([d1, e3], dim=1)\n        d1 = self.dec1(d1)\n\n        d2 = self.up2(d1)\n        d2 = torch.cat([d2, e2], dim=1)\n        d2 = self.dec2(d2)\n\n        d3 = self.up3(d2)\n        d3 = torch.cat([d3, e1], dim=1)\n        d3 = self.dec3(d3)\n\n        return self.final(d3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T09:27:24.446063Z","iopub.execute_input":"2026-03-08T09:27:24.446369Z","iopub.status.idle":"2026-03-08T09:27:24.454149Z","shell.execute_reply.started":"2026-03-08T09:27:24.446342Z","shell.execute_reply":"2026-03-08T09:27:24.453499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"device = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")\n\nmodel = ADCResTransXNet().to(device)\n\nmodel.load_state_dict(\n    torch.load(\"/kaggle/input/models/smohanesh/model/other/default/1/adc_res_transxnet_seg (1).pth\", map_location=device)\n)\n\nmodel.eval()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T09:27:27.203281Z","iopub.execute_input":"2026-03-08T09:27:27.203576Z","iopub.status.idle":"2026-03-08T09:27:27.871022Z","shell.execute_reply.started":"2026-03-08T09:27:27.203550Z","shell.execute_reply":"2026-03-08T09:27:27.870299Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"patch_folder = \"/kaggle/working/patches\"\n\npatch_files = [f for f in os.listdir(patch_folder) if f.endswith(\".png\")]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T09:27:31.104256Z","iopub.execute_input":"2026-03-08T09:27:31.104948Z","iopub.status.idle":"2026-03-08T09:27:31.197493Z","shell.execute_reply.started":"2026-03-08T09:27:31.104919Z","shell.execute_reply":"2026-03-08T09:27:31.196701Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def predict_mask(img):\n\n    img_norm = img / 255.0\n\n    tensor = torch.tensor(img_norm).permute(2,0,1).unsqueeze(0).float().to(device)\n\n    with torch.no_grad():\n        output = model(tensor)\n        prob = torch.sigmoid(output)\n\n    mask = (prob > 0.5).float().squeeze().cpu().numpy()\n\n    return mask","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T09:27:33.272666Z","iopub.execute_input":"2026-03-08T09:27:33.273274Z","iopub.status.idle":"2026-03-08T09:27:33.277794Z","shell.execute_reply.started":"2026-03-08T09:27:33.273244Z","shell.execute_reply":"2026-03-08T09:27:33.277172Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def extract_features(img, mask):\n\n    labeled = label(mask)\n\n    regions = regionprops(labeled)\n\n    if len(regions) == 0:\n        return None\n\n    region = max(regions, key=lambda x: x.area)\n\n    area = region.area\n    perimeter = region.perimeter\n\n    if perimeter == 0:\n        circularity = 0\n    else:\n        circularity = 4 * np.pi * area / (perimeter**2)\n\n    solidity = region.solidity\n    eccentricity = region.eccentricity\n\n    major_axis = region.major_axis_length\n    minor_axis = region.minor_axis_length\n\n    if minor_axis == 0:\n        aspect_ratio = 0\n    else:\n        aspect_ratio = major_axis / minor_axis\n\n    compactness = perimeter**2 / (4*np.pi*area) if area > 0 else 0\n\n    # Intensity features\n    gray = cv2.cvtColor(img, cv2.COLOR_RGB2GRAY)\n\n    mean_intensity = gray.mean()\n    std_intensity = gray.std()\n\n    # Entropy\n    hist = np.histogram(gray, bins=256)[0]\n    ent = entropy(hist + 1)\n\n    # GLCM texture features\n    glcm = graycomatrix(gray,\n                        distances=[1],\n                        angles=[0],\n                        levels=256,\n                        symmetric=True,\n                        normed=True)\n\n    contrast = graycoprops(glcm, 'contrast')[0,0]\n    homogeneity = graycoprops(glcm, 'homogeneity')[0,0]\n    energy = graycoprops(glcm, 'energy')[0,0]\n    correlation = graycoprops(glcm, 'correlation')[0,0]\n\n    return {\n        \"area\": area,\n        \"perimeter\": perimeter,\n        \"circularity\": circularity,\n        \"solidity\": solidity,\n        \"eccentricity\": eccentricity,\n        \"major_axis_length\": major_axis,\n        \"minor_axis_length\": minor_axis,\n        \"aspect_ratio\": aspect_ratio,\n        \"compactness\": compactness,\n        \"mean_intensity\": mean_intensity,\n        \"std_intensity\": std_intensity,\n        \"entropy\": ent,\n        \"contrast\": contrast,\n        \"homogeneity\": homogeneity,\n        \"energy\": energy,\n        \"correlation\": correlation\n    }","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T09:27:35.780244Z","iopub.execute_input":"2026-03-08T09:27:35.780827Z","iopub.status.idle":"2026-03-08T09:27:35.788651Z","shell.execute_reply.started":"2026-03-08T09:27:35.780769Z","shell.execute_reply":"2026-03-08T09:27:35.788046Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"dataset = []\n\nfor file in tqdm(patch_files):\n\n    path = os.path.join(patch_folder, file)\n\n    img = cv2.imread(path)\n    img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n    mask = predict_mask(img)\n\n    features = extract_features(img, mask)\n\n    if features is None:\n        continue\n\n    features[\"image\"] = file\n\n    dataset.append(features)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T09:27:39.469685Z","iopub.execute_input":"2026-03-08T09:27:39.470451Z","iopub.status.idle":"2026-03-08T09:56:54.475632Z","shell.execute_reply.started":"2026-03-08T09:27:39.470419Z","shell.execute_reply":"2026-03-08T09:56:54.474850Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.DataFrame(dataset)\n\ndf.to_csv(\"glomeruli_features_dataset.csv\", index=False)\n\nprint(\"Dataset Created\")\nprint(df.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T09:57:35.297195Z","iopub.execute_input":"2026-03-08T09:57:35.297646Z","iopub.status.idle":"2026-03-08T09:57:35.770351Z","shell.execute_reply.started":"2026-03-08T09:57:35.297620Z","shell.execute_reply":"2026-03-08T09:57:35.769742Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import seaborn as sns\nimport matplotlib.pyplot as plt\n\nsns.histplot(df[\"circularity\"], bins=30)\nplt.title(\"Circularity Distribution\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T09:58:42.873234Z","iopub.execute_input":"2026-03-08T09:58:42.873890Z","iopub.status.idle":"2026-03-08T09:58:43.448313Z","shell.execute_reply.started":"2026-03-08T09:58:42.873863Z","shell.execute_reply":"2026-03-08T09:58:43.447692Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:02:03.758059Z","iopub.execute_input":"2026-03-08T10:02:03.758348Z","iopub.status.idle":"2026-03-08T10:02:03.810807Z","shell.execute_reply.started":"2026-03-08T10:02:03.758323Z","shell.execute_reply":"2026-03-08T10:02:03.810155Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_clean = df.copy()\n\n# remove invalid area\ndf_clean = df_clean[df_clean[\"area\"] > 50]\n\n# remove invalid perimeter\ndf_clean = df_clean[df_clean[\"perimeter\"] > 10]\ndf_clean = df_clean[df_clean[\"area\"] > 200]\n# fix circularity\ndf_clean = df_clean[df_clean[\"circularity\"] <= 1]\n\n# remove extreme aspect ratios\ndf_clean = df_clean[df_clean[\"aspect_ratio\"] < 5]\n\nprint(\"Original size:\", len(df))\nprint(\"Clean dataset size:\", len(df_clean))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:09:07.744749Z","iopub.execute_input":"2026-03-08T10:09:07.745277Z","iopub.status.idle":"2026-03-08T10:09:07.755374Z","shell.execute_reply.started":"2026-03-08T10:09:07.745249Z","shell.execute_reply":"2026-03-08T10:09:07.754661Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.hist(df_clean[\"circularity\"], bins=40)\nplt.title(\"Circularity Distribution\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:09:11.348193Z","iopub.execute_input":"2026-03-08T10:09:11.348688Z","iopub.status.idle":"2026-03-08T10:09:11.479173Z","shell.execute_reply.started":"2026-03-08T10:09:11.348659Z","shell.execute_reply":"2026-03-08T10:09:11.478514Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.hist(df_clean[\"area\"], bins=40)\nplt.title(\"Area Distribution of Detected Glomeruli\")\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:11:21.815634Z","iopub.execute_input":"2026-03-08T10:11:21.816236Z","iopub.status.idle":"2026-03-08T10:11:21.964709Z","shell.execute_reply.started":"2026-03-08T10:11:21.816208Z","shell.execute_reply":"2026-03-08T10:11:21.964118Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nplt.figure(figsize=(7,6))\n\nplt.scatter(df_clean[\"circularity\"],\n            df_clean[\"solidity\"],\n            alpha=0.4)\n\nplt.xlabel(\"Circularity\")\nplt.ylabel(\"Solidity\")\n\nplt.title(\"Circularity vs Solidity of Detected Glomeruli\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:13:10.704487Z","iopub.execute_input":"2026-03-08T10:13:10.705142Z","iopub.status.idle":"2026-03-08T10:13:10.859463Z","shell.execute_reply.started":"2026-03-08T10:13:10.705110Z","shell.execute_reply":"2026-03-08T10:13:10.858798Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(7,6))\n\nplt.scatter(df_clean[\"aspect_ratio\"],\n            df_clean[\"eccentricity\"],\n            alpha=0.4)\n\nplt.xlabel(\"Aspect Ratio\")\nplt.ylabel(\"Eccentricity\")\n\nplt.title(\"Shape Characteristics of Detected Glomeruli\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:14:21.484700Z","iopub.execute_input":"2026-03-08T10:14:21.485114Z","iopub.status.idle":"2026-03-08T10:14:21.627190Z","shell.execute_reply.started":"2026-03-08T10:14:21.485083Z","shell.execute_reply":"2026-03-08T10:14:21.626542Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndf_clean.to_csv(\"glomeruli_features_dataset.csv\", index=False)\n\nprint(\"Dataset Created\")\nprint(df_clean.head())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:17:34.200416Z","iopub.execute_input":"2026-03-08T10:17:34.200989Z","iopub.status.idle":"2026-03-08T10:17:34.336905Z","shell.execute_reply.started":"2026-03-08T10:17:34.200960Z","shell.execute_reply":"2026-03-08T10:17:34.336181Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/working/glomeruli_features_dataset.csv\")\n\nprint(df.shape)\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:18:57.826380Z","iopub.execute_input":"2026-03-08T10:18:57.826664Z","iopub.status.idle":"2026-03-08T10:18:57.865112Z","shell.execute_reply.started":"2026-03-08T10:18:57.826639Z","shell.execute_reply":"2026-03-08T10:18:57.864367Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.decomposition import PCA","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:19:22.706506Z","iopub.execute_input":"2026-03-08T10:19:22.706808Z","iopub.status.idle":"2026-03-08T10:19:22.996317Z","shell.execute_reply.started":"2026-03-08T10:19:22.706770Z","shell.execute_reply":"2026-03-08T10:19:22.995578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df.drop(columns=[\"image\"], errors=\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:19:37.759499Z","iopub.execute_input":"2026-03-08T10:19:37.760030Z","iopub.status.idle":"2026-03-08T10:19:37.764705Z","shell.execute_reply.started":"2026-03-08T10:19:37.760003Z","shell.execute_reply":"2026-03-08T10:19:37.763969Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = StandardScaler()\n\nX_scaled = scaler.fit_transform(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:19:48.962035Z","iopub.execute_input":"2026-03-08T10:19:48.962561Z","iopub.status.idle":"2026-03-08T10:19:48.970503Z","shell.execute_reply.started":"2026-03-08T10:19:48.962533Z","shell.execute_reply":"2026-03-08T10:19:48.969687Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pca = PCA(n_components=2)\n\nX_pca = pca.fit_transform(X_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:19:58.510542Z","iopub.execute_input":"2026-03-08T10:19:58.510880Z","iopub.status.idle":"2026-03-08T10:19:58.518148Z","shell.execute_reply.started":"2026-03-08T10:19:58.510854Z","shell.execute_reply":"2026-03-08T10:19:58.517591Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(7,6))\n\nplt.scatter(X_pca[:,0],\n            X_pca[:,1],\n            alpha=0.5)\n\nplt.xlabel(\"Principal Component 1\")\nplt.ylabel(\"Principal Component 2\")\n\nplt.title(\"PCA of Glomerulus Features\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:20:08.949956Z","iopub.execute_input":"2026-03-08T10:20:08.950573Z","iopub.status.idle":"2026-03-08T10:20:09.106072Z","shell.execute_reply.started":"2026-03-08T10:20:08.950546Z","shell.execute_reply":"2026-03-08T10:20:09.105360Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(\"Explained variance ratio:\")\n\nprint(pca.explained_variance_ratio_)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:20:26.886886Z","iopub.execute_input":"2026-03-08T10:20:26.887194Z","iopub.status.idle":"2026-03-08T10:20:26.891622Z","shell.execute_reply.started":"2026-03-08T10:20:26.887170Z","shell.execute_reply":"2026-03-08T10:20:26.890837Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df.describe()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:22:37.492280Z","iopub.execute_input":"2026-03-08T10:22:37.492633Z","iopub.status.idle":"2026-03-08T10:22:37.539709Z","shell.execute_reply.started":"2026-03-08T10:22:37.492605Z","shell.execute_reply":"2026-03-08T10:22:37.539054Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.cluster import KMeans\nimport matplotlib.pyplot as plt\n\ninertia = []\n\nk_values = range(2,10)\n\nfor k in k_values:\n    kmeans = KMeans(n_clusters=k, random_state=42)\n    kmeans.fit(X_scaled)\n    inertia.append(kmeans.inertia_)\n\nplt.plot(k_values, inertia, marker=\"o\")\n\nplt.xlabel(\"Number of clusters\")\nplt.ylabel(\"Inertia\")\n\nplt.title(\"Elbow Method for Optimal Clusters\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:24:01.787203Z","iopub.execute_input":"2026-03-08T10:24:01.787537Z","iopub.status.idle":"2026-03-08T10:24:02.536202Z","shell.execute_reply.started":"2026-03-08T10:24:01.787498Z","shell.execute_reply":"2026-03-08T10:24:02.535462Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"!pip install umap-learn\n!pip install hdbscan","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:26:45.658087Z","iopub.execute_input":"2026-03-08T10:26:45.658599Z","iopub.status.idle":"2026-03-08T10:26:52.086422Z","shell.execute_reply.started":"2026-03-08T10:26:45.658572Z","shell.execute_reply":"2026-03-08T10:26:52.085490Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nfrom sklearn.preprocessing import StandardScaler\n\nimport umap\nimport hdbscan","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:27:35.233834Z","iopub.execute_input":"2026-03-08T10:27:35.234672Z","iopub.status.idle":"2026-03-08T10:28:13.958126Z","shell.execute_reply.started":"2026-03-08T10:27:35.234640Z","shell.execute_reply":"2026-03-08T10:28:13.957409Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/working/glomeruli_features_dataset.csv\")\n\nprint(df.shape)\ndf.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:28:40.584488Z","iopub.execute_input":"2026-03-08T10:28:40.585412Z","iopub.status.idle":"2026-03-08T10:28:40.625260Z","shell.execute_reply.started":"2026-03-08T10:28:40.585376Z","shell.execute_reply":"2026-03-08T10:28:40.624471Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X = df.drop(columns=[\"image\"], errors=\"ignore\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:28:52.348014Z","iopub.execute_input":"2026-03-08T10:28:52.348308Z","iopub.status.idle":"2026-03-08T10:28:52.353248Z","shell.execute_reply.started":"2026-03-08T10:28:52.348283Z","shell.execute_reply":"2026-03-08T10:28:52.352456Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"scaler = StandardScaler()\n\nX_scaled = scaler.fit_transform(X)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:29:05.914295Z","iopub.execute_input":"2026-03-08T10:29:05.914595Z","iopub.status.idle":"2026-03-08T10:29:05.923618Z","shell.execute_reply.started":"2026-03-08T10:29:05.914568Z","shell.execute_reply":"2026-03-08T10:29:05.922827Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"umap_model = umap.UMAP(\n    n_neighbors=15,\n    min_dist=0.1,\n    n_components=2,\n    random_state=42\n)\n\nX_umap = umap_model.fit_transform(X_scaled)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:29:20.942253Z","iopub.execute_input":"2026-03-08T10:29:20.942776Z","iopub.status.idle":"2026-03-08T10:29:47.633069Z","shell.execute_reply.started":"2026-03-08T10:29:20.942746Z","shell.execute_reply":"2026-03-08T10:29:47.632423Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"clusterer = hdbscan.HDBSCAN(\n    min_cluster_size=50,\n    metric=\"euclidean\"\n)\n\nclusters = clusterer.fit_predict(X_umap)\n\ndf[\"cluster\"] = clusters","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:30:12.955180Z","iopub.execute_input":"2026-03-08T10:30:12.956022Z","iopub.status.idle":"2026-03-08T10:30:13.099352Z","shell.execute_reply.started":"2026-03-08T10:30:12.955989Z","shell.execute_reply":"2026-03-08T10:30:13.098588Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(8,6))\n\nplt.scatter(\n    X_umap[:,0],\n    X_umap[:,1],\n    c=df[\"cluster\"],\n    cmap=\"Spectral\",\n    s=10\n)\n\nplt.title(\"UMAP + HDBSCAN Clustering of Glomeruli\")\n\nplt.xlabel(\"UMAP 1\")\nplt.ylabel(\"UMAP 2\")\n\nplt.colorbar(label=\"Cluster\")\n\nplt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:30:32.376483Z","iopub.execute_input":"2026-03-08T10:30:32.377091Z","iopub.status.idle":"2026-03-08T10:30:32.625750Z","shell.execute_reply.started":"2026-03-08T10:30:32.377060Z","shell.execute_reply":"2026-03-08T10:30:32.625115Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"print(df[\"cluster\"].value_counts())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:31:15.284356Z","iopub.execute_input":"2026-03-08T10:31:15.284952Z","iopub.status.idle":"2026-03-08T10:31:15.290132Z","shell.execute_reply.started":"2026-03-08T10:31:15.284926Z","shell.execute_reply":"2026-03-08T10:31:15.289551Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\n\npatch_folder = \"/kaggle/working/patches\"\n\nfor cluster_id in df[\"cluster\"].unique():\n\n    sample = df[df[\"cluster\"] == cluster_id].sample(6)\n\n    plt.figure(figsize=(10,6))\n    plt.suptitle(f\"Cluster {cluster_id}\")\n\n    for i,(idx,row) in enumerate(sample.iterrows()):\n\n        img = cv2.imread(f\"{patch_folder}/{row['image']}\")\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n\n        plt.subplot(2,3,i+1)\n        plt.imshow(img)\n        plt.axis(\"off\")\n\n    plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:35:08.844510Z","iopub.execute_input":"2026-03-08T10:35:08.845122Z","iopub.status.idle":"2026-03-08T10:35:09.609161Z","shell.execute_reply.started":"2026-03-08T10:35:08.845091Z","shell.execute_reply":"2026-03-08T10:35:09.608352Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"features = df.select_dtypes(include=['float64','int64'])\n\ncluster_summary = features.groupby(df[\"cluster\"]).mean()\n\nprint(cluster_summary)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-03-08T10:37:43.639776Z","iopub.execute_input":"2026-03-08T10:37:43.640507Z","iopub.status.idle":"2026-03-08T10:37:43.654103Z","shell.execute_reply.started":"2026-03-08T10:37:43.640478Z","shell.execute_reply":"2026-03-08T10:37:43.653342Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}