{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":94689,"databundleVersionId":11605086,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# # This Python 3 environment comes with many helpful analytics libraries installed\n# # It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# # For example, here's several helpful packages to load\n\n# import numpy as np # linear algebra\n# import pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# # Input data files are available in the read-only \"../input/\" directory\n# # For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\n# import os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# # You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# # You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-05-09T10:06:19.312927Z","iopub.execute_input":"2025-05-09T10:06:19.313260Z","iopub.status.idle":"2025-05-09T10:06:19.317290Z","shell.execute_reply.started":"2025-05-09T10:06:19.313225Z","shell.execute_reply":"2025-05-09T10:06:19.316455Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport os\n\n# Load the labeled data metadata\nbase_path=\"/kaggle/input/forams-classification-2025\"\nlabeled_df = pd.read_csv(os.path.join(base_path, 'labelled.csv'))\n\n# Display the first few rows\nprint(\"Labeled Data Metadata:\")\nprint(labeled_df.head())\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-09T10:06:19.318744Z","iopub.execute_input":"2025-05-09T10:06:19.318966Z","iopub.status.idle":"2025-05-09T10:06:19.341013Z","shell.execute_reply.started":"2025-05-09T10:06:19.318949Z","shell.execute_reply":"2025-05-09T10:06:19.340359Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\n# Path to the labeled volume folder\nvolumes_labelled_path = '/kaggle/input/forams-classification-2025/volumes/volumes/labelled'\n\n# List files in the labelled volumes folder\nlabeled_files = os.listdir(volumes_labelled_path)\n\n# Display the first few files to confirm structure\nprint(\"Labeled Volume Files:\")\nprint(labeled_files[:5])\n\n# Extract the numeric ID from the first row in labelled.csv\nsample_id = labeled_df.iloc[0]['id'].replace('labelled_', '')  # Remove 'labelled_' prefix\n\n# Look for the correct file format (adjust the scale factor as per your understanding)\nmatching_files = [f for f in labeled_files if f.startswith(f\"labelled_foram_{sample_id}\")]\n\n# Display matched files\nprint(f\"Matching files for id {sample_id}:\")\nprint(matching_files)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-09T10:06:19.341641Z","iopub.execute_input":"2025-05-09T10:06:19.341839Z","iopub.status.idle":"2025-05-09T10:06:19.348004Z","shell.execute_reply.started":"2025-05-09T10:06:19.341822Z","shell.execute_reply":"2025-05-09T10:06:19.347215Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tifffile\nimport matplotlib.pyplot as plt\n\n# Path to the sample file\nsample_file_path = '/kaggle/input/forams-classification-2025/volumes/volumes/labelled/labelled_foram_00000_sc_0_752.tif'\n\n# Load the TIFF file as a 3D NumPy array\nimg_array = tifffile.imread(sample_file_path)\n\n# Check the shape of the array to confirm it's 3D (128x128x128)\nprint(f\"Shape of the volume: {img_array.shape}\")\n\n# Visualize a few slices along the z-axis (we'll show slices at index 30, 60, and 90)\nfig, axes = plt.subplots(1, 3, figsize=(12, 4))\n\n# Slice indices to display (valid indices between 0 and 127)\nslice_indices = [30, 60, 90]\n\nfor i, slice_idx in enumerate(slice_indices):\n    axes[i].imshow(img_array[slice_idx], cmap='gray')\n    axes[i].set_title(f\"Slice {slice_idx}\")\n    axes[i].axis('off')\n\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-09T10:06:19.349492Z","iopub.execute_input":"2025-05-09T10:06:19.349697Z","iopub.status.idle":"2025-05-09T10:06:19.604898Z","shell.execute_reply.started":"2025-05-09T10:06:19.349682Z","shell.execute_reply":"2025-05-09T10:06:19.604090Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import numpy as np\n# import pandas as pd\n# import tifffile\n# from glob import glob\n# import os\n# from concurrent.futures import ThreadPoolExecutor\n# import gc\n# from tqdm import tqdm\n\n# # Paths\n# labeled_volume_path = '/kaggle/input/forams-classification-2025/volumes/volumes/labelled/'\n# unlabeled_volume_path = '/kaggle/input/forams-classification-2025/volumes/volumes/unlabelled/'\n\n# # Metadata\n# # labeled_metadata = pd.read_csv('/kaggle/input/forams-classification-2025/labelled.csv')\n\n\n# import os\n# import re\n# import pandas as pd\n# import torch\n# import torchvision.transforms as transforms\n# import numpy as np\n# import tifffile as tiff  # For loading .tif files\n\n# # Paths\n# volumes_path = \"/kaggle/input/forams-classification-2025/volumes/volumes/\"\n# labelled_csv_path = \"/kaggle/input/forams-classification-2025/labelled.csv\"\n\n# # Regex for filenames\n# labelled_regex = re.compile(r\"labelled_foram_(\\d{5})_sc_(\\d)_(\\d{3})\\.tif\")\n# unlabelled_regex = re.compile(r\"foram_(\\d{5})_sc_(\\d)_(\\d{3})\\.tif\")\n\n# # Parse labelled data\n# labelled_data = []\n# for filename in os.listdir(os.path.join(volumes_path, \"labelled\")):\n#     match = labelled_regex.match(filename)\n#     if match:\n#         file_id, int_scale, dec_scale = match.groups()\n#         scaling_factor = int(int_scale) + int(dec_scale) / 1000\n#         labelled_data.append({\"id\": int(file_id), \"scaling_factor\": scaling_factor, \"filename\": filename})\n\n# # Parse unlabelled data\n# unlabelled_data = []\n# for filename in os.listdir(os.path.join(volumes_path, \"unlabelled\")):\n#     match = unlabelled_regex.match(filename)\n#     if match:\n#         file_id, int_scale, dec_scale = match.groups()\n#         scaling_factor = int(int_scale) + int(dec_scale) / 1000\n#         unlabelled_data.append({\"id\": int(file_id), \"scaling_factor\": scaling_factor, \"filename\": filename})\n\n# # Load labelled CSV\n# # Read the CSV file, skipping the first row\n# labels_df = pd.read_csv(labelled_csv_path)\n\n# labels_df = labels_df[pd.to_numeric(labels_df['id'], errors='coerce').notnull()]\n\n# labelled_df = pd.DataFrame(labelled_data)\n# labelled_df[\"id\"] = labelled_df[\"id\"].astype(int)\n\n# labelled_df = labelled_df.merge(labels_df, on=\"id\", how=\"inner\")\n\n# print(\"Labelled Data:\")\n# print(labelled_df.head())\n\n# print(\"\\nUnlabelled Data:\")\n# print(pd.DataFrame(unlabelled_data).head())\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-09T10:06:19.605704Z","iopub.execute_input":"2025-05-09T10:06:19.605905Z","iopub.status.idle":"2025-05-09T10:06:19.610143Z","shell.execute_reply.started":"2025-05-09T10:06:19.605889Z","shell.execute_reply":"2025-05-09T10:06:19.609373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# import numpy as np\n# import tifffile\n# from glob import glob\n# import os\n# import gc\n# from tqdm import tqdm\n\n# # Path for unlabeled volumes\n# unlabeled_volume_path = '/kaggle/input/forams-classification-2025/volumes/volumes/unlabelled/'\n\n# # Function to load and normalize a single volume\n# def load_volume(file_path):\n#     try:\n#         volume = tifffile.imread(file_path)\n#         return volume / np.max(volume)  # Normalize to [0, 1]\n#     except Exception as e:\n#         print(f\"Error loading {file_path}: {e}\")\n#         return None\n\n# # Process unlabeled data one by one\n# def process_unlabeled_incrementally():\n#     unlabeled_files = sorted(glob(os.path.join(unlabeled_volume_path, \"*.tif\")))\n\n#     for idx, file in enumerate(tqdm(unlabeled_files, desc=\"Processing Unlabeled Volumes\")):\n#         volume = load_volume(file)\n#         if volume is not None:\n#             # Save processed volume immediately\n#             save_path = f'/kaggle/working/unlabeled_volume_{idx}.npz'\n#             np.savez_compressed(save_path, X=volume)\n#             print(f\"Saved {save_path}\")\n        \n#         # Explicitly release memory\n#         del volume\n#         gc.collect()\n\n#     print(\"Unlabeled data processing completed.\")\n\n# # Execute\n\n# process_unlabeled_incrementally()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-09T10:06:19.611485Z","iopub.execute_input":"2025-05-09T10:06:19.611815Z","iopub.status.idle":"2025-05-09T10:06:19.625882Z","shell.execute_reply.started":"2025-05-09T10:06:19.611793Z","shell.execute_reply":"2025-05-09T10:06:19.625325Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# print(f\"Training data shape: {X_train.shape}\")\n# print(f\"Training labels shape: {y_train.shape}\")\n\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-05-09T10:06:19.626535Z","iopub.execute_input":"2025-05-09T10:06:19.626738Z","iopub.status.idle":"2025-05-09T10:06:19.641474Z","shell.execute_reply.started":"2025-05-09T10:06:19.626724Z","shell.execute_reply":"2025-05-09T10:06:19.640748Z"}},"outputs":[],"execution_count":null}]}