{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install pandas\n!pip install numpy\n!pip install matplotlib\n!pip install scikit-learn\n!pip install keras\n!pip install tensorflow\n!pip install scikit-image","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom collections import Counter\nimport matplotlib.pyplot as plt\nimport os\nimport shutil\nfrom sklearn.model_selection import train_test_split\nfrom keras.models import Sequential, load_model\nfrom keras.layers import Dense, Dropout, Flatten, Conv2D, MaxPool2D, BatchNormalization\nfrom keras.optimizers import RMSprop, SGD\nfrom keras.preprocessing.image import ImageDataGenerator\nfrom skimage import io\nimport tensorflow as tf\nprint(\"Num GPUs Available: \", len(tf.config.experimental.list_physical_devices('GPU')))","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"EDA\nThe train_labels.csv data is quite clean and contains an image id along with the label. No null counts were found and no cleaning is required.","metadata":{}},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/histopathologic-cancer-detection/train_labels.csv\")\n\nprint(df.head())\n\nprint(\"\\n----- Null counts -----\")\nnull_counts = df.isna().sum()\nprint(null_counts)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels, counts = zip(*Counter(df[\"label\"]).items())\nplt.figure(figsize=(10, 6))\nplt.bar(labels, counts, color=[\"green\", \"red\"])\nplt.xlabel(\"Label (0 = No Tumor, 1 = Tumor)\")\nplt.ylabel(\"Image Count\")\nplt.title(\"Distribution of Labels in the Data\")\nplt.xticks(labels)\nplt.grid(axis='y')\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"num_images_to_display = 3\n\ndf[\"path\"] = df[\"id\"].apply(lambda x: os.path.join(\"/kaggle/input/histopathologic-cancer-detection/train\", str(x) + \".tif\"))\nimages_with_label_0 = df[df[\"label\"] == 0]\nimages_with_label_1 = df[df[\"label\"] == 1]\n\nfor i in range(num_images_to_display):\n    image = plt.imread(images_with_label_0[\"path\"].iloc[i])\n\n    plt.subplot(2, num_images_to_display, i+1)\n    plt.imshow(image)\n    plt.axis('off')\n    plt.title(\"Label 0\")\n\nfor i in range(num_images_to_display):\n    image = plt.imread(images_with_label_1[\"path\"].iloc[i])\n\n    plt.subplot(2, num_images_to_display, num_images_to_display + i + 1)\n    plt.imshow(image)\n    plt.axis('off')\n    plt.title(\"Label 1\")\n\n\nplt.show()\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"unique_dimensions = set()\nrandom_indices = np.random.randint(0, len(df), 1000)\nfor index in random_indices:\n    image = io.imread(df[\"path\"].iloc[index])\n    image_height, image_width, image_channels = image.shape\n    unique_dimensions.add((image_height, image_width, image_channels))\n\nprint(f\"Unique dimension: {unique_dimensions}\")\nprint(f\"Image Height: {image_height} pixels\")\nprint(f\"Image Width: {image_width} pixels\")\nprint(f\"Number of Channels (Depth): {image_channels}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[\"label\"] = df[\"label\"].astype(str)\ndf_0 = df[df[\"label\"] == \"0\"].sample(10000, random_state=42)\ndf_1 = df[df[\"label\"] == \"1\"].sample(10000, random_state=42)\ndf_subset = pd.concat([df_0, df_1], ignore_index=True)\n\ntrain_file_paths, test_file_paths, train_labels, test_labels = train_test_split(df_subset[\"path\"], df_subset[\"label\"], test_size=0.2, random_state=42)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create directory called train_data and copy the training data into it\ntrain_dir = \"train_data\"\nif os.path.exists(train_dir):\n    shutil.rmtree(train_dir)\nos.makedirs(train_dir)\nos.makedirs(os.path.join(train_dir, \"0\"))\nos.makedirs(os.path.join(train_dir, \"1\"))\nfor file_path, label in zip(train_file_paths, train_labels):\n    name = file_path.split(\"/\")[-1]\n    if label == \"0\":\n        shutil.copy2(file_path, os.path.join(train_dir, \"0\", name))\n    else:\n        shutil.copy2(file_path, os.path.join(train_dir, \"1\", name))\n\n# Create directory called test_data and copy the test data into it\ntest_dir = \"test_data\"\nif os.path.exists(test_dir):\n    shutil.rmtree(test_dir)\nos.makedirs(test_dir)\nos.makedirs(os.path.join(test_dir, \"0\"))\nos.makedirs(os.path.join(test_dir, \"1\"))\nfor file_path, label in zip(test_file_paths, test_labels):\n    name = file_path.split(\"/\")[-1]\n    if label == \"0\":\n        shutil.copy2(file_path, os.path.join(test_dir, \"0\", name))\n    else:\n        shutil.copy2(file_path, os.path.join(test_dir, \"1\", name))\n\ntrain_datagen = ImageDataGenerator(\n    rescale=1.0 / 255,\n    rotation_range=20,\n    width_shift_range=0.2,\n    height_shift_range=0.2,\n    shear_range=0.2,\n    zoom_range=0.2,\n    horizontal_flip=True,\n    fill_mode=\"nearest\"\n)\ntest_datagen = ImageDataGenerator(rescale=1.0 / 255)\n\ntrain_generator = train_datagen.flow_from_directory(\n    directory=train_dir,\n    target_size=(image_width, image_height),\n    batch_size=32,\n    class_mode=\"binary\",\n    color_mode=\"rgb\",\n    shuffle=True,\n    seed=42\n)\ntest_generator = test_datagen.flow_from_directory(\n    directory=test_dir,\n    target_size=(image_width, image_height),\n    batch_size=32,\n    class_mode=\"binary\",\n    color_mode=\"rgb\",\n    shuffle=False,\n    seed=42\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hyperparameters\nlearning_rate = 0.001\noptimizer = RMSprop(learning_rate=learning_rate)\ntraining_epochs = 10\nbatch_size = 32\ndropout_rate = 0.5\n\nmodel1 = Sequential()\npool_size = (3, 3)\nfilter_size = (3, 3)\n\n# Convolutional layers\nmodel1.add(Conv2D(16, filter_size, activation='relu', input_shape=(image_width, image_height, image_channels)))\nmodel1.add(Conv2D(16, filter_size, activation='relu'))\nmodel1.add(BatchNormalization())\nmodel1.add(MaxPool2D(pool_size=pool_size))\n\nmodel1.add(Conv2D(32, filter_size, activation='relu'))\nmodel1.add(Conv2D(32, filter_size, activation='relu'))\nmodel1.add(BatchNormalization())\nmodel1.add(MaxPool2D(pool_size=pool_size))\n\nmodel1.add(Conv2D(64, filter_size, activation='relu'))\nmodel1.add(Conv2D(64, filter_size, activation='relu'))\nmodel1.add(BatchNormalization())\nmodel1.add(MaxPool2D(pool_size=pool_size))\n\n# Convert to 1D vector\nmodel1.add(Flatten())\n\n# Classification layers\nmodel1.add(Dense(64, activation='sigmoid'))\nmodel1.add(Dropout(dropout_rate))\nmodel1.add(Dense(1, activation='sigmoid'))\n\n# Compile the model\nmodel1.compile(optimizer=optimizer, loss='binary_crossentropy', metrics=['accuracy'])\n\n# Train the model\ntrain_steps_per_epoch=train_generator.n//train_generator.batch_size\nvalidation_steps_per_epoch=test_generator.n//test_generator.batch_size\nhistory1 = model1.fit(\n    train_generator,\n    steps_per_epoch=train_steps_per_epoch,\n    epochs=training_epochs,\n    validation_data=test_generator,\n    validation_steps=validation_steps_per_epoch\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Hyperparameters\nlearning_rate = 0.01\noptimizer = SGD(learning_rate=learning_rate)\ntraining_epochs = 10\nbatch_size = 32\ndropout_rate = 0.25\n\nmodel2 = Sequential()\npool_size = (3, 3)\nfilter_size = (3, 3)\n\n# Convolutional layers\nmodel2.add(Conv2D(16, filter_size, activation='relu', input_shape=(image_width, image_height, image_channels)))\nmodel2.add(Conv2D(16, filter_size, activation='relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(MaxPool2D(pool_size=pool_size))\n\nmodel2.add(Conv2D(32, filter_size, activation='relu'))\nmodel2.add(Conv2D(32, filter_size, activation='relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(MaxPool2D(pool_size=pool_size))\n\nmodel2.add(Conv2D(64, filter_size, activation='relu'))\nmodel2.add(Conv2D(64, filter_size, activation='relu'))\nmodel2.add(BatchNormalization())\nmodel2.add(MaxPool2D(pool_size=pool_size))\n\n# Convert to 1D vector\nmodel2.add(Flatten())\n\n# Classification layers\nmodel2.add(Dense(64, activation='sigmoid'))\nmodel2.add(Dropout(dropout_rate))\nmodel2.add(Dense(1, activation='sigmoid'))\n\n# Compile the model\nmodel2.compile(optimizer=optimizer, loss='binary_crossentropy', metrics=['accuracy'])\n\n# Train the model\ntrain_steps_per_epoch=train_generator.n//train_generator.batch_size\nvalidation_steps_per_epoch=test_generator.n//test_generator.batch_size\nhistory2 = model2.fit(\n    train_generator,\n    steps_per_epoch=train_steps_per_epoch,\n    epochs=training_epochs,\n    validation_data=test_generator,\n    validation_steps=validation_steps_per_epoch\n)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"","metadata":{}},{"cell_type":"code","source":"# Access the training history from the 'history' object of each model\ntraining_loss1 = model1.history.history['loss']\nvalidation_loss1 = model1.history.history['val_loss']\ntraining_loss2 = model2.history.history['loss']\nvalidation_loss2 = model2.history.history['val_loss']\nepochs = range(1, len(training_loss1) + 1)\n\n# Create a subplot for training loss for Model 1\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nplt.plot(epochs, training_loss1, label='Training Loss (Model 1)', color='blue')\nplt.plot(epochs, validation_loss1, label='Validation Loss (Model 1)', color='red')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.title('Training and Validation Loss for Model 1')\nplt.grid(True)\n\n# Create a subplot for training loss for Model 2\nplt.subplot(1, 2, 2)\nplt.plot(epochs, training_loss2, label='Training Loss (Model 2)', color='blue')\nplt.plot(epochs, validation_loss2, label='Validation Loss (Model 2)', color='red')\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend()\nplt.title('Training and Validation Loss for Model 2')\nplt.grid(True)\n\nplt.tight_layout()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training_acc1 = model1.history.history['accuracy']\nvalidation_acc1 = model1.history.history['val_accuracy']\ntraining_acc2 = model2.history.history['accuracy']\nvalidation_acc2 = model2.history.history['val_accuracy']\n\n# Create an array of epochs for the x-axis\nepochs = range(1, len(training_acc1) + 1)\n\n# Create a subplot for training accuracy\nplt.figure(figsize=(12, 5))\nplt.subplot(1, 2, 1)\nplt.plot(epochs, training_acc1, label='Training Accuracy (Model 1)', color='blue')\nplt.plot(epochs, validation_acc1, label='Validation Accuracy (Model 1)', color='red')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.title('Training and Validation Accuracy for Model 1')\nplt.grid(True)\n\n# Create a subplot for validation accuracy\nplt.subplot(1, 2, 2)\nplt.plot(epochs, training_acc2, label='Training Accuracy (Model 2)', color='blue')\nplt.plot(epochs, validation_acc2, label='Validation Accuracy (Model 2)', color='red')\nplt.xlabel('Epochs')\nplt.ylabel('Accuracy')\nplt.legend()\nplt.title('Training and Validation Accuracy for Model 2')\nplt.grid(True)\n\nplt.tight_layout()\nplt.show()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df = pd.DataFrame(columns=[\"id\", \"label\", \"prediction\"])\n\ntest_file_paths = []\nfor root, _, files in os.walk(\"/kaggle/input/histopathologic-cancer-detection/test\"):\n    for name in files:\n        test_file_paths.append(os.path.join(root, name))\n\nfor i in range(len(test_file_paths)):\n    image = io.imread(test_file_paths[i])\n    id = test_file_paths[i].split(\"/\")[-1].split(\".\")[0]\n    prediction = model1.predict(image.reshape(1, 96, 96, 3), verbose=0)\n    \n    # Use len(submission_df) to determine the index\n    row = {\"id\": id, \"prediction\": prediction[0], \"label\": 0 if prediction[0] < 0.5 else 1}\n    submission_df.loc[len(submission_df) + 1] = row\n    print(f\"Progress: {len(submission_df)}/{len(test_file_paths)}\", end=\"\\r\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df.to_csv(\"submission.csv\", index=False)\n","metadata":{},"execution_count":null,"outputs":[]}]}