{"metadata":{"kernelspec":{"name":"python3","display_name":"Python 3","language":"python"},"language_info":{"name":"python","version":"3.11.11","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31012,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pathlib import Path\nfrom PIL import Image\nimport os\n\n# Use HLS palette for more fun diagram colorings\nsns.set_palette(\"hls\")\n\n# Store the directory for easy access\nin_dir = '/kaggle/input/histopathologic-cancer-detection'\nout_dir = '/kaggle/working'\n\n# Load the labels\nlabels_df = pd.read_csv(f'{in_dir}/train_labels.csv')\n\n# Basic information about the dataset\nprint(\"\\n=== Dataset Overview ===\")\nprint(f\"Total number of images: {len(labels_df)}\")\nprint(\"\\nLabel distribution:\")\nprint(labels_df['label'].value_counts(normalize=True).round(3) * 100)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-04-16T19:03:10.826624Z","iopub.execute_input":"2025-04-16T19:03:10.826954Z","iopub.status.idle":"2025-04-16T19:03:11.276578Z","shell.execute_reply.started":"2025-04-16T19:03:10.826914Z","shell.execute_reply":"2025-04-16T19:03:11.275640Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Basic information about the dataset\nprint(\"\\n=== Dataset Overview ===\")\nprint(f\"Total number of images: {len(labels_df)}\")\nprint(\"\\nLabel distribution:\")\nprint(labels_df['label'].value_counts(normalize=True).round(3) * 100)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Create a pie chart of the distribution\nplt.figure(figsize=(8, 6))\nlabels_df['label'].value_counts().plot(kind='pie', autopct='%1.1f%%')\nplt.title('Distribution of Cancer vs Non-Cancer Cases')\nplt.ylabel('')\nplt.savefig('label_distribution.png')\nplt.plot()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Basic statistics\nprint(\"\\n=== Basic Statistics ===\")\nprint(labels_df.describe())\n\n# Create a bar plot of the distribution\nplt.figure(figsize=(8, 6))\nsns.countplot(data=labels_df, x='label')\nplt.title('Count of Cancer vs Non-Cancer Cases')\nplt.xlabel('Label (0: No Cancer, 1: Cancer)')\nplt.ylabel('Count')\nplt.savefig('label_counts.png')\nplt.plot()\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\n# Checking for any missing values\nprint(\"\\n=== Missing Values ===\")\nprint(labels_df.isnull().sum())","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_image_sizes(directory):\n    sizes = []\n    for img_path in Path(directory).glob('*.tif'):\n        try:\n            size = os.path.getsize(img_path)\n            sizes.append(size)\n        except:\n            continue\n    return sizes\n\nimage_sizes = get_image_sizes(f'{dir}/test')\nprint(\"\\n=== Image Size Statistics ===\")\nprint(pd.Series(image_sizes).describe())","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display images as a grid of\nn_samples = 25  # number of images to display\nn_cols = 5     # number of columns in the grid\nn_rows = (n_samples + n_cols - 1) // n_cols\n\n# Create a figure with subplots\nfig, axes = plt.subplots(n_rows, n_cols, figsize=(12, 6*n_rows))\naxes = axes.flatten()  # flatten the axes array for easier iteration\n\n# Get random sample of indices\nrandom_indices = np.random.choice(len(labels_df), n_samples, replace=False)\n\n# Display images\nfor i, ax in enumerate(axes):\n    if i < n_samples:\n        # Get the image path and label\n        img = random_indices[i]\n        img_path = f'{in_dir}/train/{labels_df.iloc[img][\"id\"]}.tif' \n        label = labels_df.iloc[img][\"label\"]\n        \n        try:\n            # Load and display the image\n            img = Image.open(img_path)\n            ax.imshow(img)\n            ax.set_title(f'Label: {label}\\n(Cancer: {label == 1})')\n            ax.axis('off')  # hide axes\n        except Exception as e:\n            ax.text(0.5, 0.5, f'Error loading image\\n{str(e)}', \n                   ha='center', va='center')\n            ax.axis('off')\n    else:\n        ax.axis('off')  # hide empty subplots if any\n\nplt.tight_layout()\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.metrics import accuracy_score\nfrom sklearn.preprocessing import StandardScaler\nimport cv2\nfrom tqdm import tqdm\nfrom sklearn.metrics import silhouette_score\n\ndef extract_features(image_path, n_bins=16):\n    \"\"\"Extract basic features from an image\"\"\"\n    # Read image\n    img = cv2.imread(image_path)\n    if img is None:\n        return None\n    \n    # Convert to grayscale\n    gray = cv2.cvtColor(img, cv2.COLOR_BGR2GRAY)\n    \n    # Basic statistics\n    mean_intensity = np.mean(gray)\n    std_intensity = np.std(gray)\n    \n    # Histogram features with variable number of bins\n    hist = cv2.calcHist([gray], [0], None, [n_bins], [0, 256])\n    hist = hist.flatten() / hist.sum()  # normalize histogram\n    \n    # Texture features (using Laplacian variance)\n    laplacian = cv2.Laplacian(gray, cv2.CV_64F)\n    texture = np.var(laplacian)\n    \n    # Combine all features\n    features = np.concatenate([\n        [mean_intensity, std_intensity, texture],\n        hist\n    ])\n    \n    return features\n\n# Test different numbers of bins\nn_bins_list = [8, 16, 32, 64]\nresults = []\n\n# Determine the best number of bins to use for the model\nfor n_bins in n_bins_list:\n    print(f\"\\nTesting with {n_bins} histogram bins...\")\n    \n    # Extract features\n    features_list = []\n    valid_indices = []\n    \n    print(f\"Extracting features for {len(labels_df)} images...\")\n    # Use tqdm to show progress for feature extraction\n    for idx, row in tqdm(labels_df.iterrows(), total=len(labels_df)):\n        img_path = f'{in_dir}/train/{row[\"id\"]}.tif'\n        features = extract_features(img_path, n_bins)\n        if features is not None:\n            features_list.append(features)\n            valid_indices.append(idx)\n    \n    print(f\"Converting to numpy array...\")\n    # Convert to numpy array\n    X = np.array(features_list)\n    y = labels_df.iloc[valid_indices]['label'].values\n    \n    print(f\"Splitting and scaling data...\")\n    # Split and scale\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n    scaler = StandardScaler()\n    X_train_scaled = scaler.fit_transform(X_train)\n    X_test_scaled = scaler.transform(X_test)\n    \n    # Use a subset for silhouette score calculation\n    sample_size = min(5000, len(X_train_scaled))\n    sample_indices = np.random.choice(len(X_train_scaled), sample_size, replace=False)\n    X_sample = X_train_scaled[sample_indices]\n    y_sample = y_train[sample_indices]\n\n    print(\"Calculating silhouette score on subset of data...\")\n    silhouette = silhouette_score(X_sample, y_sample)\n    \n    print(f\"Training model...\") \n    # Train model\n    model = LogisticRegression(random_state=42)\n    model.fit(X_train_scaled, y_train)\n    \n    print(f\"Getting predictions...\")\n    # Get predictions\n    y_pred = model.predict(X_test_scaled)\n    accuracy = accuracy_score(y_test, y_pred)\n    \n    print(f\"Evaluating model...\")\n    results.append({\n        'n_bins': n_bins,\n        'accuracy': accuracy,\n        'silhouette': silhouette,\n        'n_features': X.shape[1]\n    })\n    \n    print(f\"Accuracy: {accuracy:.3f}\")\n    print(f\"Silhouette Score: {silhouette:.3f}\")\n    print(f\"Number of features: {X.shape[1]}\")\n\n# Create DataFrame and find best bin size\nresults_df = pd.DataFrame(results)\nprint(\"\\nSummary of Results:\")\nprint(results_df.to_string(index=False))\n\n# Find best bin size based on both metrics\nbest_accuracy_bins = results_df.loc[results_df['accuracy'].idxmax(), 'n_bins']\nbest_silhouette_bins = results_df.loc[results_df['silhouette'].idxmax(), 'n_bins']\n\nprint(f\"\\nBest bin size for accuracy: {best_accuracy_bins}\")\nprint(f\"Best bin size for silhouette score: {best_silhouette_bins}\")\n\n# Plot results\nplt.figure(figsize=(12, 5))\n\n# Plot accuracy\nplt.subplot(1, 2, 1)\nplt.plot(results_df['n_bins'], results_df['accuracy'], 'bo-')\nplt.xlabel('Number of Histogram Bins')\nplt.ylabel('Accuracy')\nplt.title('Accuracy vs Number of Bins')\nplt.grid(True, alpha=0.3)\n\n# Plot silhouette score\nplt.subplot(1, 2, 2)\nplt.plot(results_df['n_bins'], results_df['silhouette'], 'ro-')\nplt.xlabel('Number of Histogram Bins')\nplt.ylabel('Silhouette Score')\nplt.title('Silhouette Score vs Number of Bins')\nplt.grid(True, alpha=0.3)\n\nplt.tight_layout()\nplt.show()","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Set the optimal number of bins\nn_bins = 16\nprint(f\"Building final model with {n_bins} histogram bins...\")\n\n# Extract features\nfeatures_list = []\nvalid_indices = []\n\nfor idx, row in tqdm(labels_df.iterrows(), total=len(labels_df)):\n    img_path = f'{in_dir}/train/{row[\"id\"]}.tif'\n    features = extract_features(img_path, n_bins)\n    if features is not None:\n        features_list.append(features)\n        valid_indices.append(idx)\n\n# Convert to numpy array\nX = np.array(features_list)\ny = labels_df.iloc[valid_indices]['label'].values\n\n# Split and scale\nX_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\nscaler = StandardScaler()\nX_train_scaled = scaler.fit_transform(X_train)\nX_test_scaled = scaler.transform(X_test)\n\n# Train final model\nfinal_model = LogisticRegression(random_state=42)\nfinal_model.fit(X_train_scaled, y_train)\n\n# Get predictions\ny_pred = final_model.predict(X_test_scaled)\naccuracy = accuracy_score(y_test, y_pred)\n\nprint(f\"\\nFinal Model Performance:\")\nprint(f\"Accuracy: {accuracy:.3f}\")\nprint(f\"Number of features: {X.shape[1]}\")","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from sklearn.decomposition import PCA\nimport matplotlib.pyplot as plt\n\n# Reduce features to 2D using PCA\npca = PCA(n_components=2)\nX_train_2d = pca.fit_transform(X_train_scaled)\nX_test_2d = pca.transform(X_test_scaled)\n\n# Create a mesh grid for the decision boundary\nx_min, x_max = X_train_2d[:, 0].min() - 1, X_train_2d[:, 0].max() + 1\ny_min, y_max = X_train_2d[:, 1].min() - 1, X_train_2d[:, 1].max() + 1\nxx, yy = np.meshgrid(np.arange(x_min, x_max, 0.1),\n                     np.arange(y_min, y_max, 0.1))\n\n# Fit logistic regression on 2D data\nlog_reg_2d = LogisticRegression(random_state=42)\nlog_reg_2d.fit(X_train_2d, y_train)\n\n# Get predictions for the mesh grid\nZ = log_reg_2d.predict_proba(np.c_[xx.ravel(), yy.ravel()])[:, 1]\nZ = Z.reshape(xx.shape)\n\n# Create the plot\nplt.figure(figsize=(10, 8))\nplt.contourf(xx, yy, Z, alpha=0.4)\nplt.scatter(X_train_2d[:, 0], X_train_2d[:, 1], c=y_train, alpha=0.8)\nplt.colorbar(label='Probability of Cancer')\nplt.xlabel('First Principal Component')\nplt.ylabel('Second Principal Component')\nplt.title('Logistic Regression Decision Boundary (2D PCA projection)')\nplt.show()\n\n# Print explained variance ratio\nprint(\"\\nExplained variance ratio of first two components:\")\nprint(pca.explained_variance_ratio_)","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import torch\nimport torch.nn as nn\nimport torch.optim as optim\nfrom torch.utils.data import Dataset, DataLoader\nfrom torchvision import transforms","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Define the CNN architecture\nclass CancerCNN(nn.Module):\n    def __init__(self):\n        super(CancerCNN, self).__init__()\n        # Convolutional layers\n        self.conv1 = nn.Conv2d(3, 32, kernel_size=3, padding=1)\n        self.conv2 = nn.Conv2d(32, 64, kernel_size=3, padding=1)\n        self.conv3 = nn.Conv2d(64, 128, kernel_size=3, padding=1)\n        \n        # Pooling and dropout\n        self.pool = nn.MaxPool2d(2, 2)\n        self.dropout = nn.Dropout(0.5)\n        \n        # Fully connected layers\n        self.fc1 = nn.Linear(128 * 8 * 8, 512)\n        self.fc2 = nn.Linear(512, 1)\n        \n        # Activation functions\n        self.relu = nn.ReLU()\n        \n    def forward(self, x):\n        # First conv block\n        x = self.pool(self.relu(self.conv1(x)))\n        \n        # Second conv block\n        x = self.pool(self.relu(self.conv2(x)))\n        \n        # Third conv block\n        x = self.pool(self.relu(self.conv3(x)))\n        \n        # Flatten\n        x = x.view(-1, 128 * 8 * 8)\n        \n        # Fully connected layers\n        x = self.dropout(self.relu(self.fc1(x)))\n        x = torch.sigmoid(self.fc2(x))\n        \n        return x\n\n# Custom Dataset class\nclass CancerDataset(Dataset):\n    def __init__(self, csv_file, img_dir, transform=None):\n        self.data = pd.read_csv(csv_file)\n        self.img_dir = img_dir\n        self.transform = transform\n\n    def __len__(self):\n        return len(self.data)\n\n    def __getitem__(self, idx):\n        img_name = f\"{self.img_dir}/{self.data.iloc[idx]['id']}.tif\"\n        image = Image.open(img_name).convert('RGB')\n        label = self.data.iloc[idx]['label']\n\n        if self.transform:\n            image = self.transform(image)\n\n        return image, label\n\n# Define data transforms\ntransform = transforms.Compose([\n    transforms.Resize((64, 64)),  # Resize images to 64x64\n    transforms.ToTensor(),\n    transforms.Normalize(mean=[0.485, 0.456, 0.406],\n                        std=[0.229, 0.224, 0.225])\n])\n\n\n# Create datasets\ntrain_dataset = CancerDataset(in_dir + '/train_labels.csv', in_dir + '/train', transform=transform)\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True)\n\n# Split into train and validation\ntrain_size = int(0.8 * len(train_dataset))\nval_size = len(train_dataset) - train_size\ntrain_dataset, val_dataset = torch.utils.data.random_split(train_dataset, [train_size, val_size])\n\n# Create data loaders\ntrain_loader = DataLoader(train_dataset, batch_size=32, shuffle=True)\nval_loader = DataLoader(val_dataset, batch_size=32, shuffle=False)\n\n# Initialize model, loss function, and optimizer\n# Use MPS (Apple Silicon) if available, otherwise use GPU if available, otherwise use CPU\nif torch.backends.mps.is_available():\n    device = torch.device(\"mps\")\nelse:\n    device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\ncnn_model = CancerCNN().to(device)\ncriterion = nn.BCELoss()\noptimizer = optim.Adam(cnn_model.parameters(), lr=0.001)\n\n# Training loop\ndef train_model(model, train_loader, val_loader, criterion, optimizer, num_epochs=10):\n    train_losses = []\n    val_losses = []\n    best_val_loss = float('inf')\n    \n    for epoch in range(num_epochs):\n        # Training phase\n        model.train()\n        running_loss = 0.0\n        correct = 0\n        total = 0\n        \n        for images, labels in train_loader:\n            images = images.to(device)\n            labels = labels.float().to(device)\n            \n            optimizer.zero_grad()\n            outputs = model(images)\n            loss = criterion(outputs.squeeze(), labels)\n            loss.backward()\n            optimizer.step()\n            \n            # Detach the loss before adding to running_loss\n            running_loss += loss.detach().item()\n            predicted = (outputs.squeeze() > 0.5).float()\n            total += labels.size(0)\n            correct += (predicted == labels).sum().item()\n        \n        # Calculate loss and accuracy\n        train_loss = running_loss / len(train_loader)\n        train_acc = 100 * correct / total\n        train_losses.append(train_loss)\n        \n        # Validation phase\n        model.eval()\n        val_running_loss = 0.0\n        val_correct = 0\n        val_total = 0\n        \n        with torch.no_grad():\n            for images, labels in val_loader:\n                images = images.to(device)\n                labels = labels.float().to(device)\n                \n                outputs = model(images)\n                loss = criterion(outputs.squeeze(), labels)\n                \n                val_running_loss += loss.item()\n                predicted = (outputs.squeeze() > 0.5).float()\n                val_total += labels.size(0)\n                val_correct += (predicted == labels).sum().item()\n        \n        val_loss = val_running_loss / len(val_loader)\n        val_acc = 100 * val_correct / val_total\n        val_losses.append(val_loss)\n        \n        print(f'Epoch [{epoch+1}/{num_epochs}]')\n        print(f'Train Loss: {train_loss:.4f}, Train Acc: {train_acc:.2f}%')\n        print(f'Val Loss: {val_loss:.4f}, Val Acc: {val_acc:.2f}%')\n        \n        # Save best model\n        if val_loss < best_val_loss:\n            best_val_loss = val_loss\n            torch.save(model.state_dict(), 'best_model.pth')\n    \n    return train_losses, val_losses\n\n# Train the model\ntrain_losses, val_losses = train_model(cnn_model, train_loader, val_loader, criterion, optimizer)\n\n# Plot training history\nplt.figure(figsize=(10, 5))\nplt.plot(train_losses, label='Training Loss')\nplt.plot(val_losses, label='Validation Loss')\nplt.title('Training and Validation Loss')\nplt.xlabel('Epoch')\nplt.ylabel('Loss')\nplt.legend()\nplt.grid(True)\nplt.show()\n","metadata":{},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to predict on new images\ndef predict_image(model, image_path, transform):\n    model.eval()\n    with torch.no_grad():\n        image = Image.open(image_path).convert('RGB')\n        image = transform(image).unsqueeze(0)\n        image = image.to(device)\n        output = model(image)\n        prediction = (output.squeeze() > 0.5).float()\n        probability = output.squeeze().item()\n        return prediction.item(), probability\n\n# Create a submission for the CNN model using  predict_image\ndef create_cnn_submission(model):\n    # Load the best model\n    model.load_state_dict(torch.load('best_model.pth'))\n    model.eval()\n    \n    # Create a list to store predictions\n    predictions = []\n    \n    # Get all test images\n    test_images = [f for f in os.listdir(f'{in_dir}/test') if f.endswith('.tif')]\n    print(f\"Found {len(test_images)} test images\")\n    \n    # Process each test image\n    for img in tqdm(test_images):\n        image_path = f'{in_dir}/test/{img}'\n        try:\n            # Get prediction\n            prediction, probability = predict_image(model, image_path, transform)\n            \n            # Extract image ID (remove .tif extension)\n            image_id = img.split('.')[0]\n            \n            # Add to predictions\n            predictions.append({\n                'id': image_id,\n                'label': prediction\n            })\n            \n        except Exception as e:\n            print(f\"Error processing {img}: {str(e)}\")\n    \n    # Create DataFrame and save to csv\n    submission_df = pd.DataFrame(predictions)\n    submission_df.to_csv(f'{out_dir}/submission.csv', index=False)\n    print(f\"\\nCreated submission.csv with {len(submission_df)} predictions\")\n    \n    # Display first few rows to verify format\n    print(\"\\nFirst few rows of submission file:\")\n    print(submission_df.head())\n    \n# Run and done\ncreate_cnn_submission(cnn_model)","metadata":{},"outputs":[],"execution_count":null}]}