{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceType":"competition","sourceId":11848,"databundleVersionId":862157}],"dockerImageVersionId":30732,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nfrom glob import glob\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.layers import Input, Conv2D, MaxPooling2D, Flatten, Dense, Dropout, GlobalMaxPooling2D, GlobalAveragePooling2D, Concatenate\nfrom tensorflow.keras.losses import binary_crossentropy\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ModelCheckpoint\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2026-05-16T13:41:54.08733Z","iopub.execute_input":"2026-05-16T13:41:54.087997Z","iopub.status.idle":"2026-05-16T13:42:12.247211Z","shell.execute_reply.started":"2026-05-16T13:41:54.087946Z","shell.execute_reply":"2026-05-16T13:42:12.245864Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow.keras.layers import Input, Conv2D, MaxPooling2D, Lambda, Dropout, Dense, GlobalMaxPooling2D, GlobalAveragePooling2D, Flatten, concatenate\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.losses import binary_crossentropy\nimport tensorflow.keras.backend as K","metadata":{"execution":{"iopub.status.busy":"2026-05-16T13:42:12.250302Z","iopub.execute_input":"2026-05-16T13:42:12.251153Z","iopub.status.idle":"2026-05-16T13:42:12.260158Z","shell.execute_reply.started":"2026-05-16T13:42:12.251091Z","shell.execute_reply":"2026-05-16T13:42:12.258587Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"<div align='center'><br><font size='70'><b>Histopatholgical cancer prediction</b></font></div>\n","metadata":{}},{"cell_type":"markdown","source":"<div align=\"center\">\n    <font size='5'><b><span style=\"color:blue\">I. Data preprocessing </span></b></font>\n</div>","metadata":{}},{"cell_type":"markdown","source":"### <span style=\"color:blue\"> Read data</span>\n","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/histopathologic-cancer-detection/train_labels.csv\")\ndf_train['file_path'] = df_train['id'].apply(lambda x: f\"/kaggle/input/histopathologic-cancer-detection/train/{x}.tif\")\ndf_train['label'] = df_train['label'].astype(str)\n\n# 将文件路径和标签分为训练集和验证集\ntrain_df, val_df = train_test_split(df_train, test_size=0.1, random_state=101010)\n","metadata":{"execution":{"iopub.status.busy":"2026-05-16T13:42:12.261503Z","iopub.execute_input":"2026-05-16T13:42:12.261874Z","iopub.status.idle":"2026-05-16T13:42:13.197856Z","shell.execute_reply.started":"2026-05-16T13:42:12.261848Z","shell.execute_reply":"2026-05-16T13:42:13.196655Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"raw","source":"train_df.head()","metadata":{}},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2026-05-16T13:42:13.199447Z","iopub.execute_input":"2026-05-16T13:42:13.200992Z","iopub.status.idle":"2026-05-16T13:42:13.223526Z","shell.execute_reply.started":"2026-05-16T13:42:13.200946Z","shell.execute_reply":"2026-05-16T13:42:13.222022Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2026-05-16T13:42:13.226674Z","iopub.execute_input":"2026-05-16T13:42:13.227622Z","iopub.status.idle":"2026-05-16T13:42:13.301908Z","shell.execute_reply.started":"2026-05-16T13:42:13.227584Z","shell.execute_reply":"2026-05-16T13:42:13.300463Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_df.shape","metadata":{"execution":{"iopub.status.busy":"2026-05-16T13:42:13.303495Z","iopub.execute_input":"2026-05-16T13:42:13.303931Z","iopub.status.idle":"2026-05-16T13:42:13.318649Z","shell.execute_reply.started":"2026-05-16T13:42:13.303878Z","shell.execute_reply":"2026-05-16T13:42:13.317005Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <span style=\"color:blue\"> data manifest</span>\n","metadata":{}},{"cell_type":"code","source":"import cv2\nimport matplotlib.pyplot as plt\nbatch_size=32\n\n# Function to read and display images in a subplot grid\ndef display_images(df, num_images, columns=5):\n    rows = (num_images + columns - 1) // columns  # Calculate the number of rows needed\n    fig, axes = plt.subplots(rows, columns, figsize=(20, rows * 4))\n\n    for i in range(num_images):\n        file_path = df.iloc[i]['file_path']\n        label = df.iloc[i]['label']\n        # Read the image using OpenCV\n        image = cv2.imread(file_path, cv2.IMREAD_UNCHANGED)\n        if image is None:\n            print(f\"Error reading {file_path}\")\n            continue\n        # Convert the image from BGR to RGB\n        image_rgb = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        # Determine the position in the grid\n        row = i // columns\n        col = i % columns\n        # Display the image\n        axes[row, col].imshow(image_rgb)\n        axes[row, col].set_title(f\"Label {label}\")\n        axes[row, col].axis('off')\n\n#     # Hide any empty subplots\n#     for j in range(num_images, rows * columns):\n#         row = j // columns\n#         col = j % columns\n#         axes[row, col].axis('off')\n\n    plt.tight_layout()\n    plt.show()\n\n# Display the first 10 images in a 5-column layout\ndisplay_images(train_df, 10, columns=5)","metadata":{"execution":{"iopub.status.busy":"2026-05-16T13:42:13.320472Z","iopub.execute_input":"2026-05-16T13:42:13.320826Z","iopub.status.idle":"2026-05-16T13:42:14.793279Z","shell.execute_reply.started":"2026-05-16T13:42:13.320799Z","shell.execute_reply":"2026-05-16T13:42:14.791842Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <span style=\"color:blue\"> data enhancement</span>\n","metadata":{}},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"batch_size=64\ndata_gen = ImageDataGenerator(\n        horizontal_flip=True,\n        vertical_flip=True,\n        rotation_range=20,\n        width_shift_range=0.2,\n        height_shift_range=0.2,\n        shear_range=0.2,\n        zoom_range=0.2,\n        fill_mode='nearest'\n    )\ntrain_gen_data = data_gen.flow_from_dataframe(\n        dataframe=train_df,\n        x_col='file_path',\n        y_col='label',\n        target_size=(96, 96),\n        batch_size=batch_size,\n        class_mode='binary'\n    )\n\n\nval_gen_data = data_gen.flow_from_dataframe(\n        dataframe=val_df,\n        x_col='file_path',\n        y_col='label',\n        target_size=(96, 96),\n        batch_size=batch_size,\n        class_mode='binary'\n    )","metadata":{"execution":{"iopub.status.busy":"2026-05-16T13:46:11.518429Z","iopub.execute_input":"2026-05-16T13:46:11.519662Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_gen_data[0][0].shape","metadata":{"execution":{"iopub.status.busy":"2026-05-16T13:45:49.809708Z","iopub.status.idle":"2026-05-16T13:45:49.810226Z","shell.execute_reply.started":"2026-05-16T13:45:49.809999Z","shell.execute_reply":"2026-05-16T13:45:49.810016Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"type(train_gen_data[0])","metadata":{"execution":{"iopub.status.busy":"2026-05-16T13:45:49.812426Z","iopub.status.idle":"2026-05-16T13:45:49.812785Z","shell.execute_reply.started":"2026-05-16T13:45:49.812625Z","shell.execute_reply":"2026-05-16T13:45:49.81264Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndef get_model():\n    inputs = Input(shape=(96, 96, 3))\n\n    x = Conv2D(32, (3, 3), activation='relu', padding='same')(inputs)\n    x = MaxPooling2D((2, 2))(x)\n\n    x = Conv2D(64, (3, 3), activation='relu', padding='same')(x)\n    x = MaxPooling2D((2, 2))(x)\n\n    x = Conv2D(128, (3, 3), activation='relu', padding='same')(x)\n    x = MaxPooling2D((2, 2))(x)\n\n    # Combining GlobalMaxPooling2D, GlobalAveragePooling2D, and Flatten\n    x1 = GlobalMaxPooling2D()(x)\n    x2 = GlobalAveragePooling2D()(x)\n    x3 = Flatten()(x)\n    x = concatenate([x1, x2, x3])\n\n    x = Dropout(0.5)(x)\n    outputs = Dense(1, activation=\"sigmoid\", name=\"3_\")(x)\n\n    model = Model(inputs, outputs)\n    \n    model.compile(optimizer=Adam(learning_rate=0.0001), loss=binary_crossentropy, metrics=['accuracy'])\n    model.summary()\n\n    return model\n\nmodel=get_model()\n","metadata":{"execution":{"iopub.status.busy":"2026-05-16T13:45:49.81446Z","iopub.status.idle":"2026-05-16T13:45:49.814859Z","shell.execute_reply.started":"2026-05-16T13:45:49.81469Z","shell.execute_reply":"2026-05-16T13:45:49.814706Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.compile(optimizer=Adam(learning_rate=0.0001),\n              loss='binary_crossentropy',\n              metrics=['accuracy'])","metadata":{"execution":{"iopub.status.busy":"2026-05-16T13:45:49.816488Z","iopub.status.idle":"2026-05-16T13:45:49.816821Z","shell.execute_reply.started":"2026-05-16T13:45:49.81667Z","shell.execute_reply":"2026-05-16T13:45:49.816684Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"epochs = 20\nhistory = model.fit(\n    train_gen_data,\n    epochs=epochs,\n    validation_data=val_gen_data\n)","metadata":{"execution":{"iopub.status.busy":"2026-05-16T13:45:49.819656Z","iopub.status.idle":"2026-05-16T13:45:49.82024Z","shell.execute_reply.started":"2026-05-16T13:45:49.81994Z","shell.execute_reply":"2026-05-16T13:45:49.819962Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### <span style=\"color:blue\"> data prediction submission</span>\n","metadata":{}},{"cell_type":"code","source":"test_dir = '/kaggle/input/histopathologic-cancer-detection/test'\ntest_id = [f.split(\".\")[0] for f in os.listdir(test_dir) if f.endswith('.tif')]\ntest_file_path = [os.path.join(test_dir,f) for f in os.listdir(test_dir) if f.endswith('.tif')]\n# Create a DataFrame from the test filenames\ntest_df = pd.DataFrame({\n    'id': test_id,\n    'file_path': test_file_path\n})","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-16T13:45:49.821503Z","iopub.status.idle":"2026-05-16T13:45:49.821992Z","shell.execute_reply.started":"2026-05-16T13:45:49.82175Z","shell.execute_reply":"2026-05-16T13:45:49.821772Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_df['file_path'][0]","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-16T13:45:49.823995Z","iopub.status.idle":"2026-05-16T13:45:49.824514Z","shell.execute_reply.started":"2026-05-16T13:45:49.82427Z","shell.execute_reply":"2026-05-16T13:45:49.824292Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image = cv2.imread(test_df['file_path'][0])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-16T13:45:49.825955Z","iopub.status.idle":"2026-05-16T13:45:49.826648Z","shell.execute_reply.started":"2026-05-16T13:45:49.826349Z","shell.execute_reply":"2026-05-16T13:45:49.826388Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.imshow(image)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-16T13:45:49.828727Z","iopub.status.idle":"2026-05-16T13:45:49.829264Z","shell.execute_reply.started":"2026-05-16T13:45:49.828986Z","shell.execute_reply":"2026-05-16T13:45:49.829008Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to read and preprocess images\ndef read_and_preprocess_image(file_path):\n    image = cv2.imread(file_path)\n    if image is not None:\n        image = cv2.resize(image, (96, 96))  # Resize to the target size\n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-16T13:45:49.831227Z","iopub.status.idle":"2026-05-16T13:45:49.831736Z","shell.execute_reply.started":"2026-05-16T13:45:49.831487Z","shell.execute_reply":"2026-05-16T13:45:49.83151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Read and preprocess all test images\ntest_images = np.array([read_and_preprocess_image(fp) for fp in test_df['file_path']])\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-16T13:45:49.833144Z","iopub.status.idle":"2026-05-16T13:45:49.833668Z","shell.execute_reply.started":"2026-05-16T13:45:49.833404Z","shell.execute_reply":"2026-05-16T13:45:49.833424Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_pred=model.predict(test_images)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-16T13:45:49.835738Z","iopub.status.idle":"2026-05-16T13:45:49.836259Z","shell.execute_reply.started":"2026-05-16T13:45:49.83599Z","shell.execute_reply":"2026-05-16T13:45:49.83601Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"submission=pd.read_csv(\"/kaggle/input/histopathologic-cancer-detection/sample_submission.csv\")\nsubmission['id']=test_df[\"id\"]\nsubmission['label']=test_pred\nsubmission.to_csv(\"submission.csv\",index=False)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-05-16T13:45:49.837605Z","iopub.status.idle":"2026-05-16T13:45:49.838112Z","shell.execute_reply.started":"2026-05-16T13:45:49.837855Z","shell.execute_reply":"2026-05-16T13:45:49.837876Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}