{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    print(os.path.join(dirname))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T21:51:35.062734Z","iopub.execute_input":"2025-07-10T21:51:35.063358Z","iopub.status.idle":"2025-07-10T21:56:22.628283Z","shell.execute_reply.started":"2025-07-10T21:51:35.063328Z","shell.execute_reply":"2025-07-10T21:56:22.627285Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BASE_DIR = \"/kaggle/input/histopathologic-cancer-detection\"\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T21:56:22.629760Z","iopub.execute_input":"2025-07-10T21:56:22.630036Z","iopub.status.idle":"2025-07-10T21:56:22.633928Z","shell.execute_reply.started":"2025-07-10T21:56:22.630013Z","shell.execute_reply":"2025-07-10T21:56:22.633266Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# ==========================================\n# STEP 1: Import Libraries\n# ==========================================\nimport numpy as np\nimport pandas as pd\nimport os\nimport matplotlib.pyplot as plt\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Conv2D, MaxPooling2D, Flatten, Dense, Dropout\n\n# ==========================================\n# STEP 2: Set Paths for Dataset\n# ==========================================\n# Dataset is already mounted in Kaggle\nBASE_DIR = \"/kaggle/input/histopathologic-cancer-detection\"\nTRAIN_DIR = os.path.join(BASE_DIR, \"train\")\nTEST_DIR = os.path.join(BASE_DIR, \"test\")\nLABELS_PATH = os.path.join(BASE_DIR, \"train_labels.csv\")\n\n# ==========================================\n# STEP 3: Load Labels\n# ==========================================\nlabels_df = pd.read_csv(LABELS_PATH)\nlabels_df['id'] = labels_df['id'] + \".tif\"  # Add file extension\nlabels_df['label'] = labels_df['label'].astype(str)  # Convert to string for keras\nprint(\"Dataset shape:\", labels_df.shape)\nprint(labels_df.head())\n\n# ==========================================\n# STEP 4: Data Generators\n# ==========================================\ndatagen = ImageDataGenerator(\n    rescale=1./255,\n    validation_split=0.2\n)\n\ntrain_generator = datagen.flow_from_dataframe(\n    dataframe=labels_df,\n    directory=TRAIN_DIR,\n    x_col=\"id\",\n    y_col=\"label\",\n    subset=\"training\",\n    batch_size=32,\n    seed=42,\n    shuffle=True,\n    class_mode=\"binary\",\n    target_size=(96, 96)\n)\n\nval_generator = datagen.flow_from_dataframe(\n    dataframe=labels_df,\n    directory=TRAIN_DIR,\n    x_col=\"id\",\n    y_col=\"label\",\n    subset=\"validation\",\n    batch_size=32,\n    seed=42,\n    shuffle=True,\n    class_mode=\"binary\",\n    target_size=(96, 96)\n)\n\n# ==========================================\n# STEP 5: Build Lightweight CNN Model\n# ==========================================\nmodel = Sequential([\n    Conv2D(16, (3,3), activation='relu', input_shape=(96,96,3)),\n    MaxPooling2D(2,2),\n    Conv2D(32, (3,3), activation='relu'),\n    MaxPooling2D(2,2),\n    Flatten(),\n    Dense(64, activation='relu'),\n    Dropout(0.5),\n    Dense(1, activation='sigmoid')\n])\n\nmodel.compile(optimizer='adam',\n              loss='binary_crossentropy',\n              metrics=['accuracy'])\n\nmodel.summary()\n\n# ==========================================\n# STEP 6: Train Model\n# ==========================================\nepochs = 3  # Keep epochs low for faster Kaggle runtime\nhistory = model.fit(\n    train_generator,\n    validation_data=val_generator,\n    epochs=epochs\n)\n\n# ==========================================\n# STEP 7: Predict on Test Data\n# ==========================================\ntest_datagen = ImageDataGenerator(rescale=1./255)\n\ntest_generator = test_datagen.flow_from_directory(\n    directory=BASE_DIR,\n    classes=[\"test\"],\n    class_mode=None,\n    shuffle=False,\n    target_size=(96, 96),\n    batch_size=32\n)\n\npreds = model.predict(test_generator, verbose=1)\n\n# ==========================================\n# STEP 8: Create Submission File\n# ==========================================\ntest_filenames = test_generator.filenames\ntest_ids = [fname.split(\"/\")[1].replace(\".tif\", \"\") for fname in test_filenames]\n\nsubmission = pd.DataFrame({\n    \"id\": test_ids,\n    \"label\": preds.flatten()\n})\n\nsubmission.to_csv(\"/kaggle/working/submission.csv\", index=False)\nprint(\"submission.csv created at /kaggle/working/\")\n\n# ==========================================\n# Plot Predictions\n# ==========================================\nfor i in range(9):\n    img_path = os.path.join(TEST_DIR, test_filenames[i].split(\"/\")[1])\n    img = tf.keras.utils.load_img(img_path, target_size=(96, 96))\n    plt.subplot(3, 3, i+1)\n    plt.imshow(img)\n    plt.title(f\"Pred: {preds[i][0]:.2f}\")\n    plt.axis('off')\nplt.show()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-07-10T21:56:22.634857Z","iopub.execute_input":"2025-07-10T21:56:22.635140Z","iopub.status.idle":"2025-07-10T23:17:00.885346Z","shell.execute_reply.started":"2025-07-10T21:56:22.635113Z","shell.execute_reply":"2025-07-10T23:17:00.884257Z"}},"outputs":[],"execution_count":null}]}