{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"}],"dockerImageVersionId":31090,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.applications import ResNet50\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.layers import Dense\nimport os\nfrom tensorflow.keras.metrics import AUC\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.layers import GlobalAveragePooling2D\nfrom tensorflow.keras.callbacks import EarlyStopping\n\nfrom tensorflow.keras.applications import EfficientNetB0\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau\nfrom sklearn.utils.class_weight import compute_class_weight\nfrom tensorflow.keras.callbacks import ModelCheckpoint\nfrom tensorflow.keras.models import load_model\n\n\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport os\nfrom fastai.vision.all import *\nfrom sklearn.metrics import roc_auc_score\nfrom torchvision.models import resnet50, densenet201, efficientnet_b3\nfrom fastai.tabular.all import *\nfrom sklearn.metrics import roc_auc_score\nimport torch\nimport warnings\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2025-09-14T21:40:23.385540Z","iopub.execute_input":"2025-09-14T21:40:23.385840Z","iopub.status.idle":"2025-09-14T21:40:53.685109Z","shell.execute_reply.started":"2025-09-14T21:40:23.385814Z","shell.execute_reply":"2025-09-14T21:40:53.684196Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/histopathologic-cancer-detection/train_labels.csv')\ntrain_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T21:40:57.902729Z","iopub.execute_input":"2025-09-14T21:40:57.903002Z","iopub.status.idle":"2025-09-14T21:40:58.297827Z","shell.execute_reply.started":"2025-09-14T21:40:57.902983Z","shell.execute_reply":"2025-09-14T21:40:58.297087Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Data Cleaning and EDA","metadata":{}},{"cell_type":"code","source":"train_df.info()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T21:41:01.500441Z","iopub.execute_input":"2025-09-14T21:41:01.501163Z","iopub.status.idle":"2025-09-14T21:41:01.533610Z","shell.execute_reply.started":"2025-09-14T21:41:01.501136Z","shell.execute_reply":"2025-09-14T21:41:01.532833Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df.duplicated().sum()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T21:41:03.961326Z","iopub.execute_input":"2025-09-14T21:41:03.961833Z","iopub.status.idle":"2025-09-14T21:41:04.032490Z","shell.execute_reply.started":"2025-09-14T21:41:03.961808Z","shell.execute_reply":"2025-09-14T21:41:04.031695Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_df['img_path'] = train_df['id'].apply(lambda x: os.path.join(\n    '/kaggle/input/histopathologic-cancer-detection/train',f\"{x}.tif\"\n))\ndef view_sample_images(df, no_sample =None,label_col =None, img_col='id'):\n    sample_df = df.sample(no_sample)  \n    plt.figure(figsize=(10, 6))\n    \n    for i, (_, row) in enumerate(sample_df.iterrows()):\n        img = mpimg.imread(row[img_col])\n        plt.subplot(2, 3, i + 1)\n        plt.imshow(img)\n        if label_col:\n            plt.title(str(row[label_col]))\n        plt.axis(\"off\")\n    \n    plt.tight_layout()\n    plt.show()\nview_sample_images(train_df, no_sample =6,label_col ='label',img_col='img_path')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T21:41:07.258386Z","iopub.execute_input":"2025-09-14T21:41:07.259111Z","iopub.status.idle":"2025-09-14T21:41:08.747189Z","shell.execute_reply.started":"2025-09-14T21:41:07.259078Z","shell.execute_reply":"2025-09-14T21:41:08.746055Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"#Get image shape \nimg = mpimg.imread(train_df['img_path'][0])\nimg.shape","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T21:41:15.256847Z","iopub.execute_input":"2025-09-14T21:41:15.257515Z","iopub.status.idle":"2025-09-14T21:41:15.275633Z","shell.execute_reply.started":"2025-09-14T21:41:15.257482Z","shell.execute_reply":"2025-09-14T21:41:15.274810Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Check class distribution \nlabels = train_df['label'].value_counts(normalize=True)\nplt.pie(labels,autopct='%1.2f%%',labels=labels.index)\nplt.title('Label Distribution');","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T21:41:17.678323Z","iopub.execute_input":"2025-09-14T21:41:17.678604Z","iopub.status.idle":"2025-09-14T21:41:17.791359Z","shell.execute_reply.started":"2025-09-14T21:41:17.678582Z","shell.execute_reply":"2025-09-14T21:41:17.790655Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train Base Models (ResNet50, DenseNet201, EfficientNetB3)","metadata":{}},{"cell_type":"code","source":"def build_model(model_arch, batch_size=64, image_size=96, \n                validation_pct=0.2, \n                test_path='/kaggle/input/histopathologic-cancer-detection/test', \n                model_name='model'):\n    \"\"\"\n    Train a CNN on train_df and save validation + test predictions for stacking.\n\n    Args:\n        model_arch: Fastai model architecture (resnet50, densenet201, efficientnet_b3)\n        batch_size: batch size for training\n        image_size: resize dimension for images\n        validation_pct: fraction of data used for validation\n        test_path: path to test images\n        model_name: string used to save CSVs\n    \"\"\"\n    # DataBlock & DataLoaders\n    dblock = DataBlock(\n        blocks=(ImageBlock, CategoryBlock),\n        get_x=ColReader('img_path'),\n        get_y=ColReader('label'),\n        splitter=RandomSplitter(valid_pct=validation_pct, seed=42),\n        item_tfms=Resize(image_size),\n        batch_tfms=[*aug_transforms(size=image_size), Normalize.from_stats(*imagenet_stats)]\n    )\n    dls = dblock.dataloaders(train_df, bs=batch_size)\n\n    # Learner\n    learn = vision_learner(dls, model_arch, metrics=[accuracy, RocAucBinary()])\n    learn.to_fp16()  \n\n    # Check if checkpoint exists: load if available\n    model_path = f'{model_name}_checkpoint'\n    try:\n        learn = learn.load(model_path)\n        print(f\" Loaded checkpoint: {model_path}\")\n    except Exception as e:\n        print(\"No checkpoint found, starting fresh training.\")\n\n    # model checkpoint callback\n    cbs = [\n        SaveModelCallback(monitor='roc_auc_score', fname=model_path),  # save best model\n        CSVLogger(fname=f'{model_name}_history.csv')  # log training history\n    ]\n    learn.fine_tune(5, cbs=cbs)\n\n    # Save validation predictions\n    val_preds, val_targs = learn.get_preds(ds_idx=1)\n    pd.DataFrame({\n        'val_0': val_preds[:,0].numpy(),\n        'val_1': val_preds[:,1].numpy(),\n        'ground_truth_label': val_targs.numpy()\n    }).to_csv(f'{model_name}_val_preds.csv', index=False)\n\n    # Save test predictions\n    test_files = get_image_files(test_path)\n    test_dl = dls.test_dl(test_files)\n    test_preds, _ = learn.get_preds(dl=test_dl)\n    pd.DataFrame({\n        'id': [f.stem for f in test_files],\n        'pred_0': test_preds[:,0].numpy(),\n        'pred_1': test_preds[:,1].numpy()\n    }).to_csv(f'{model_name}_test_preds.csv', index=False)\n\n    print(f\"{model_name} training & predictions saved\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-14T21:44:44.178923Z","iopub.execute_input":"2025-09-14T21:44:44.179811Z","iopub.status.idle":"2025-09-14T21:44:44.188206Z","shell.execute_reply.started":"2025-09-14T21:44:44.179785Z","shell.execute_reply":"2025-09-14T21:44:44.187276Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Train ResNet50\nbuild_model(resnet50, model_name='resnet50')\n\n# Train DenseNet201\nbuild_model(densenet201, model_name='densenet201')\n\n# Train EfficientNetB3\nbuild_model(efficientnet_b3, model_name='efficientnetb3', image_size=224)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T00:11:22.848712Z","iopub.execute_input":"2025-09-15T00:11:22.849510Z","iopub.status.idle":"2025-09-15T04:02:40.160451Z","shell.execute_reply.started":"2025-09-15T00:11:22.849477Z","shell.execute_reply":"2025-09-15T04:02:40.159414Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Load validation preds\nres50_val = pd.read_csv('resnet50_val_preds.csv')\ndense201_val = pd.read_csv('densenet201_val_preds.csv')\neffb3_val = pd.read_csv('efficientnetb3_val_preds.csv')\n\n# Load test preds\nres50_test = pd.read_csv('resnet50_test_preds.csv')\ndense201_test = pd.read_csv('densenet201_test_preds.csv')\neffb3_test = pd.read_csv('efficientnetb3_test_preds.csv')\n\n# Train stacking dataframe\ntrain_stack = pd.DataFrame({\n    'res50_0': res50_val.val_0, 'res50_1': res50_val.val_1,\n    'dense201_0': dense201_val.val_0, 'dense201_1': dense201_val.val_1,\n    'effb3_0': effb3_val.val_0, 'effb3_1': effb3_val.val_1,\n    'y': res50_val.ground_truth_label\n})\n\n# Test stacking dataframe\ntest_stack = pd.DataFrame({\n    'res50_0': res50_test.pred_0, 'res50_1': res50_test.pred_1,\n    'dense201_0': dense201_test.pred_0, 'dense201_1': dense201_test.pred_1,\n    'effb3_0': effb3_test.pred_0, 'effb3_1': effb3_test.pred_1\n})\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T04:02:40.400353Z","iopub.execute_input":"2025-09-15T04:02:40.400633Z","iopub.status.idle":"2025-09-15T04:02:40.614539Z","shell.execute_reply.started":"2025-09-15T04:02:40.400609Z","shell.execute_reply":"2025-09-15T04:02:40.613910Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Train Meta-Learner ","metadata":{}},{"cell_type":"code","source":"cont_names = list(train_stack.columns[:-1])  # model outputs\ndep_var = 'y'\n\n# Make y categorical\ntrain_stack[dep_var] = train_stack[dep_var].astype(str)\n\n# DataLoaders\nsplits = RandomSplitter(seed=42)(range_of(train_stack))\nto = TabularPandas(train_stack, procs=[],\n                   cont_names=cont_names,\n                   y_names=dep_var,\n                   y_block=CategoryBlock(),  \n                   splits=splits)\ndls = to.dataloaders(bs=64)\n\n# Custom ROC AUC metric\ndef roc_score(inp, targ):\n    # Convert to probabilities\n    probs = inp.softmax(dim=1)[:,1].cpu().numpy()\n    return roc_auc_score(targ.cpu().numpy(), probs)\n\n# Meta-learner\nlearn = tabular_learner(dls, layers=[20,10],\n                        metrics=[accuracy, roc_score],\n                        wd=1e-2)\n\n# Train\nlearn.fit_one_cycle(20, 1e-3)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T04:07:08.427304Z","iopub.execute_input":"2025-09-15T04:07:08.427632Z","iopub.status.idle":"2025-09-15T04:08:31.932852Z","shell.execute_reply.started":"2025-09-15T04:07:08.427603Z","shell.execute_reply":"2025-09-15T04:08:31.932251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Predict on test set\ntest_dl = learn.dls.test_dl(test_stack)\npreds, _ = learn.get_preds(dl=test_dl)\n\n# Submission (probability for cancer class = column 1)\nsubmission = pd.DataFrame({\n    'id': res50_test.id,  # all test sets have same order\n    'label': preds[:,1].numpy()\n})\n\nsubmission.to_csv('stacked_submission.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-15T04:08:39.602399Z","iopub.execute_input":"2025-09-15T04:08:39.602697Z","iopub.status.idle":"2025-09-15T04:08:41.960220Z","shell.execute_reply.started":"2025-09-15T04:08:39.602677Z","shell.execute_reply":"2025-09-15T04:08:41.959543Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## ImageDataGenerator","metadata":{}},{"cell_type":"code","source":"# #Extract image width and height info\n# h,w = img.shape[:2]\n# IMG_SIZE= (h,w)\n# #Use ImageDataGenerator to rescale and split images \n# train_gen = ImageDataGenerator(\n#     rescale =1./255,\n#     validation_split = 0.25\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-13T17:21:40.395319Z","iopub.execute_input":"2025-09-13T17:21:40.395583Z","iopub.status.idle":"2025-09-13T17:21:40.399711Z","shell.execute_reply.started":"2025-09-13T17:21:40.395565Z","shell.execute_reply":"2025-09-13T17:21:40.398935Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# # Use flow_from_dataframe to link image path to labels, \n# # split data into training and validation samples \n# # and generate training batches during training \n# train_generator = train_gen.flow_from_dataframe(\n#     dataframe=train_df,\n#     x_col='img_path',\n#     y_col='label',\n#     target_size=(96,96),\n#     class_mode='raw',\n#     batch_size=32,\n#     shuffle=True,\n#     seed=42,\n#     directory=None ,\n#     subset='training',\n#     vertical_flip=True,\n#     rotation_range=15,\n#     width_shift_range=0.1,\n#     height_shift_range=0.1,\n#     zoom_range=0.2,\n#     horizontal_flip=True,\n#     brightness_range=[0.9, 1.1]\n# )\n\n# valid_generator = train_gen.flow_from_dataframe(\n#     dataframe=train_df,\n#     x_col='img_path',\n#     y_col='label',\n#     target_size=(96,96),\n#     class_mode='raw',\n#     batch_size=32,\n#     shuffle=False,\n#     seed=42,\n#     directory=None,\n#     subset = 'validation'\n# )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-13T17:21:42.542601Z","iopub.execute_input":"2025-09-13T17:21:42.543166Z","iopub.status.idle":"2025-09-13T17:30:57.256224Z","shell.execute_reply.started":"2025-09-13T17:21:42.543142Z","shell.execute_reply":"2025-09-13T17:30:57.255446Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# base_model = EfficientNetB0(\n#     input_shape=(96, 96, 3), \n#     weights='imagenet', \n#     include_top=False\n# )\n# base_model.trainable = True \n\n# # Build custom head\n# x = base_model.output\n# x = GlobalAveragePooling2D()(x)\n# x = BatchNormalization()(x)\n# x = Dropout(0.5)(x)\n# output = Dense(1, activation='sigmoid')(x)\n\n# model = Model(inputs=base_model.input, outputs=output)\n\n# # Compile model\n# optimizer = Adam(learning_rate=1e-4)\n# model.compile(\n#     optimizer=optimizer,\n#     loss='binary_crossentropy',\n#     metrics=[AUC(name='auc')]\n# )\n\n# # Callbacks\n# early_stop = EarlyStopping(monitor='val_auc',\n#                            patience=5, \n#                            mode='max', \n#                            restore_best_weights=True)\n# lr_reduce = ReduceLROnPlateau(monitor='val_auc', \n#                               factor=0.2, \n#                               patience=2, \n#                               min_lr=1e-6, \n#                               mode='max')\n\n# class_weights = compute_class_weight(class_weight='balanced',\n#                                      classes=np.unique(train_df['label']),\n#                                      y=train_df['label'])\n\n# class_weights = dict(enumerate(class_weights))\n\n# checkpoint = ModelCheckpoint(\n#     'efficientnet_best.h5',   # file to save model\n#     monitor='val_auc',        \n#     mode='max',\n#     save_best_only=True,\n#     save_weights_only=False   # saves full model (weights + optimizer state)\n# )\n\n# callbacks = [early_stop, lr_reduce, checkpoint]\n# # Train model\n# history = model.fit(\n#     train_generator,\n#     validation_data=valid_generator,\n#     epochs=50,\n#     callbacks=[early_stop, lr_reduce],\n#     class_weight=class_weights\n# )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-13T17:31:48.600525Z","iopub.execute_input":"2025-09-13T17:31:48.601309Z","iopub.status.idle":"2025-09-13T18:57:53.937343Z","shell.execute_reply.started":"2025-09-13T17:31:48.601277Z","shell.execute_reply":"2025-09-13T18:57:53.936765Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# #Extract training auc \n# auc = history.history['auc']\n# #get validation auc\n# val_auc = history.history['val_auc']\n\n# epochs_range = range(len(auc))\n\n# #plot training and validation auc curve \n# plt.plot(epochs_range, auc, 'r', label=\"Training AUC\")\n# plt.plot(epochs_range, val_auc, 'b', label=\"Validation AUC\")\n# plt.title(\"Training and Validation AUC\")\n# plt.legend()\n# plt.show()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-09-13T19:13:44.166660Z","iopub.execute_input":"2025-09-13T19:13:44.166974Z","iopub.status.idle":"2025-09-13T19:13:44.337363Z","shell.execute_reply.started":"2025-09-13T19:13:44.166952Z","shell.execute_reply":"2025-09-13T19:13:44.336568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}