{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt\nimport os\n\ndata_path = '/kaggle/input/histopathologic-cancer-detection'\n\ntrain_labels_df = pd.read_csv(os.path.join(data_path, 'train_labels.csv'))\n\nprint(f'Number of samples in training data: {len(train_labels)}')\n\nprint(train_labels_df.head())\n\nprint(\"\\nClass distribution:\")\nprint(train_labels_df['label'].value_counts())\n\n# Class distribution visualization\ntrain_labels_df['label'].value_counts().plot(kind='bar')\nplt.title('Class Distribution')\nplt.xlabel('Label')\nplt.ylabel('Number of Samples')\nplt.show()","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-04-18T18:28:00.131077Z","iopub.execute_input":"2025-04-18T18:28:00.131505Z","iopub.status.idle":"2025-04-18T18:28:00.716811Z","shell.execute_reply.started":"2025-04-18T18:28:00.131465Z","shell.execute_reply":"2025-04-18T18:28:00.716032Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_images_dir = os.path.join(data_dir, 'train')\n\nrandom_image_ids = random.sample(train_labels_df['id'].tolist(), 4)\n\nplt.figure(figsize=(10, 10))\n\nfor i, image_id in enumerate(random_image_ids):\n    image_path = os.path.join(train_images_dir, f'{image_id}.tif')\n\n    try:\n        img = Image.open(image_path)\n\n        label = train_labels_df[train_labels_df['id'] == image_id]['label'].iloc[0]\n        label_text = 'Tumor' if label == 1 else 'No Tumor'\n\n        plt.subplot(2, 2, i + 1)\n        plt.imshow(img)\n        plt.title(f'ID: {image_id}\\nLabel: {label_text}')\n        plt.axis('off')\n\n    except FileNotFoundError:\n        print(f\"Error: Image file not found at {image_path}\")\n    except Exception as e:\n        print(f\"An error occurred while processing image {image_id}: {e}\")\n\nplt.tight_layout()\nplt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}