{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport cv2\nimport numpy as np\nimport pandas as pd\nfrom PIL import Image\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import train_test_split\n\nfrom tensorflow import keras\nfrom keras.models import Sequential\nfrom keras.layers import Conv2D, MaxPooling2D, Dense, Flatten, Dropout\nfrom tensorflow.keras.layers import BatchNormalization\n\nprint(\"Loaded all libraries\")\n\n\n# 1. DATA LOADING AND EXPLORATION\n\n\n# Đường dẫn đến thư mục chứa ảnh và file labels\ntrain_path = \"/kaggle/input/histopathologic-cancer-detection/train\"\nlabels_path = \"/kaggle/input/histopathologic-cancer-detection/train_labels.csv\"\nrandom_seed = 42\n\n# Đọc file labels\ndf_labels = pd.read_csv(labels_path)\nprint(f\"Total images: {len(df_labels)}\")\nprint(f\"\\nLabel distribution:\\n{df_labels['label'].value_counts()}\")\nprint(f\"\\nFirst few rows:\\n{df_labels.head()}\")\n\n\n# 2. LOAD IMAGES AND LABELS\n\n\ndef load_images_and_labels(df, img_path, img_size=(227, 227), max_samples=None):\n    \"\"\"\n    Load images and labels from Histopathologic Cancer Detection dataset\n    \n    Parameters:\n    - df: DataFrame containing image ids and labels\n    - img_path: Path to image directory\n    - img_size: Target size for resizing (default: 227x227 for AlexNet)\n    - max_samples: Limit number of samples (None = load all)\n    \"\"\"\n    img_lst = []\n    labels = []\n    \n    # Limit samples if specified\n    if max_samples:\n        df = df.head(max_samples)\n    \n    for idx, row in df.iterrows():\n        img_id = row['id']\n        label = row['label']\n        \n        # Construct full image path\n        img_file = os.path.join(img_path, f\"{img_id}.tif\")\n        \n        # Check if file exists\n        if not os.path.exists(img_file):\n            continue\n            \n        # Read and process image\n        img = cv2.imread(img_file)\n        img = cv2.cvtColor(img, cv2.COLOR_BGR2RGB)\n        \n        # Resize to target size (227x227 for AlexNet)\n        img_array = Image.fromarray(img, 'RGB')\n        resized_img = img_array.resize(img_size)\n        \n        img_lst.append(np.array(resized_img))\n        labels.append(label)\n        \n        # Progress indicator\n        if (idx + 1) % 10000 == 0:\n            print(f\"Loaded {idx + 1} images...\")\n    \n    return np.array(img_lst), np.array(labels)\n\n# Load images (sử dụng max_samples để test nhanh, bỏ tham số này để load toàn bộ)\nprint(\"\\nLoading images...\")\nimages, labels = load_images_and_labels(df_labels, train_path, max_samples=5000)\n\nprint(f\"\\nImages shape: {images.shape}\")\nprint(f\"Labels shape: {labels.shape}\")\nprint(f\"Data types: {images.dtype}, {labels.dtype}\")\n\n\n# 3. VISUALIZE RANDOM SAMPLES\n\n\ndef display_rand_images(images, labels, title_prefix=\"Label\"):\n    \"\"\"Display 9 random images with their labels\"\"\"\n    plt.figure(figsize=(15, 10))\n    \n    for i in range(9):\n        plt.subplot(3, 3, i + 1)\n        \n        # Random index\n        idx = np.random.randint(0, len(images))\n        \n        plt.imshow(images[idx])\n        plt.title(f'{title_prefix}: {labels[idx]}')\n        plt.axis('off')\n    \n    plt.tight_layout()\n    plt.show()\n\nprint(\"\\nDisplaying sample images...\")\ndisplay_rand_images(images, labels)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-10-03T02:53:45.023815Z","iopub.execute_input":"2025-10-03T02:53:45.024127Z"}},"outputs":[],"execution_count":null}]}