{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":114397,"databundleVersionId":13696770,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Clustering the BioTrove Dataset: BioCLIP  embeddings\n\nThis notebook embeds all 50000 images with BioCLIP and saves the embeddings, compressing the dataset from 5.5 GByte to 0.15 GByte. You can read the embeddings from another notebook and cluster them with the algorithm of your choice.\n\nReferences\n- [Competition](https://www.kaggle.com/competitions/biotrove-clustering)\n- https://imageomics.github.io/pybioclip/","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"print('Installing...')\n!pip install pybioclip","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T06:32:48.573437Z","iopub.execute_input":"2025-11-29T06:32:48.573697Z","iopub.status.idle":"2025-11-29T06:34:01.872214Z","shell.execute_reply.started":"2025-11-29T06:32:48.573672Z","shell.execute_reply":"2025-11-29T06:34:01.871286Z"},"_kg_hide-output":true,"collapsed":true,"jupyter":{"outputs_hidden":true}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport pickle\nimport glob\nfrom tqdm import trange\nimport torch\n\nimport PIL\nfrom bioclip import TreeOfLifeClassifier, Rank\n","metadata":{"trusted":true,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2025-11-29T06:40:13.395610Z","iopub.execute_input":"2025-11-29T06:40:13.396116Z","iopub.status.idle":"2025-11-29T06:40:13.399804Z","shell.execute_reply.started":"2025-11-29T06:40:13.396093Z","shell.execute_reply":"2025-11-29T06:40:13.399143Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Reading the metadata","metadata":{}},{"cell_type":"code","source":"metadata = pd.read_csv('/kaggle/input/biotrove-clustering/metadata.csv')\nfamilies = np.unique(metadata.family)\nmetadata","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T06:34:14.830794Z","iopub.execute_input":"2025-11-29T06:34:14.831055Z","iopub.status.idle":"2025-11-29T06:34:14.937126Z","shell.execute_reply.started":"2025-11-29T06:34:14.831032Z","shell.execute_reply":"2025-11-29T06:34:14.936508Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Embedding the images","metadata":{}},{"cell_type":"code","source":"device = 'cuda' if torch.cuda.is_available() else 'cpu'\nprint(device)\nclassifier = TreeOfLifeClassifier(device=device)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T06:40:19.797083Z","iopub.execute_input":"2025-11-29T06:40:19.797709Z","iopub.status.idle":"2025-11-29T06:40:33.063589Z","shell.execute_reply.started":"2025-11-29T06:40:19.797683Z","shell.execute_reply":"2025-11-29T06:40:33.062769Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Embed all images (this would take 20 hours on cpu)\nembedding_dim = 768\nembedding = np.zeros((len(metadata), embedding_dim), dtype=np.float32)\nbatch_size = 64\n\nfor batch_start in trange(0, len(metadata), batch_size):\n    batch_end = min(batch_start+batch_size, len(metadata))\n    batch_input = []\n    for i, bi in enumerate(range(batch_start, batch_end)):\n        hash_id = metadata.hash_id.iloc[bi]\n        batch_input.append(PIL.Image.open(glob.glob(f\"/kaggle/input/biotrove-clustering/images/images/{hash_id}.???\")[0]).convert('RGB'))\n    batch_output = classifier.create_image_features(batch_input)\n    embedding[batch_start:batch_end] = batch_output.cpu()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T06:41:21.637919Z","iopub.execute_input":"2025-11-29T06:41:21.638784Z","iopub.status.idle":"2025-11-29T06:44:08.281017Z","shell.execute_reply.started":"2025-11-29T06:41:21.638755Z","shell.execute_reply":"2025-11-29T06:44:08.280066Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the embeddings\nwith open('embedding.pickle', 'wb') as f:\n    pickle.dump(embedding, f) # 150 MByte","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-11-29T06:44:31.701667Z","iopub.execute_input":"2025-11-29T06:44:31.702134Z","iopub.status.idle":"2025-11-29T06:44:31.925075Z","shell.execute_reply.started":"2025-11-29T06:44:31.702114Z","shell.execute_reply":"2025-11-29T06:44:31.924507Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}