{"metadata": {"kernelspec": {"language": "python", "display_name": "Python 3", "name": "python3"}, "language_info": {"name": "python", "version": "3.6.6", "mimetype": "text/x-python", "codemirror_mode": {"name": "ipython", "version": 3}, "pygments_lexer": "ipython3", "nbconvert_exporter": "python", "file_extension": ".py"}, "kaggle": {"accelerator": "none", "dataSources": [], "dockerImageVersionId": 28755, "isInternetEnabled": false, "language": "python", "sourceType": "notebook", "isGpuEnabled": false}}, "nbformat_minor": 4, "nbformat": 4, "cells": [{"metadata": {}, "cell_type": "markdown", "source": "# Submit notebook\n\nSubmit notebook for attending [Histopathologic Cancer Detection](https://www.kaggle.com/competitions/histopathologic-cancer-detection/overview) competition.\n\nBest score:\n\nPrivate score \u2014 0.9696\n\nPublic score \u2014 0.9631\n\n- - -\nRelated links:\n\nNotebook used for model training, can be found \u2014 [here](https://www.kaggle.com/code/pavlozelinskiy/hcd-convnext-tiny-train-v1).\n\nBest model dataset, can be found \u2014 [here](https://www.kaggle.com/datasets/pavlozelinskiy/hcd-convnext-tiny-weights-v1/data).\n\nCustom training dataset used in training, can be found \u2014 [here](https://www.kaggle.com/datasets/pavlozelinskiy/hcd-convnext-tiny-train-dataset-with-wsi-column).\n"}, {"metadata": {"ExecuteTime": {"end_time": "2026-09-13T10:58:31.295845400Z", "start_time": "2026-09-13T10:58:28.318363900Z"}}, "cell_type": "code", "source": "# Core\nimport numpy as np\nimport torch\nimport timm\nfrom torchvision import transforms as T\n\n# Data related\nimport os\nfrom pathlib import Path\nimport pandas as pd\nfrom torch.utils.data import DataLoader\n# from histopathologic_cancer_detection.dataset import ImageDataset\n\n# Constants\nDEVICE = 'cuda'\n\nIS_KAGGLE = bool(os.environ.get('KAGGLE_KERNEL_RUN_TYPE'))\nKAGGLE_DIR = Path('/kaggle') if IS_KAGGLE else Path.cwd().parent / 'kaggle'\n\nINPUT_DIR  = KAGGLE_DIR / 'input' / 'competitions' / 'histopathologic-cancer-detection'\nOUTPUT_DIR = KAGGLE_DIR / 'working'\nMODEL_DIR  = (KAGGLE_DIR / 'input' / 'hcd-convnext-tiny-weights-v1') if IS_KAGGLE else OUTPUT_DIR\n\nBEST_MODEL_PATH = MODEL_DIR / 'best_model.pth'\n\nfor label, d in [('INPUT_DIR', INPUT_DIR), ('MODEL_DIR', MODEL_DIR)]:\n  if not d.is_dir():\n      raise FileNotFoundError(\n          f\"{label} missing: {d}\\nmounted: {[p.name for p in Path('/kaggle/input').iterdir()]}\"\n          if IS_KAGGLE else f\"{label} missing: {d}\"\n      )\nassert BEST_MODEL_PATH.is_file(), f\"{BEST_MODEL_PATH}\\nfound: {list(MODEL_DIR.iterdir())}\"", "outputs": [], "execution_count": 1}, {"metadata": {}, "cell_type": "markdown", "source": "# Define class for PyTorch Dataset\nto run training locally directly on Windows .py version of this class is used"}, {"metadata": {"ExecuteTime": {"end_time": "2026-09-09T18:18:11.082931600Z", "start_time": "2026-09-09T18:18:11.071896200Z"}}, "cell_type": "code", "outputs": [], "execution_count": 2, "source": "# Copy pasted from dataset.py\nfrom torch.utils.data import Dataset\nfrom PIL import Image\n\nclass ImageDataset(Dataset):\n    def __init__(self, df, img_dir, tf, has_label=True):\n        self.ids = df['id'].values\n        self.y = df['label'].values.astype('float32') if has_label else None\n        self.dir = img_dir\n        self.tf = tf\n\n    def __len__(self):\n        return len(self.ids)\n\n    def __getitem__(self, i):\n        img = Image.open(f'{self.dir}/{self.ids[i]}.tif').convert('RGB')\n        img = self.tf(img)\n        return (img, self.y[i] if self.y is not None else img) # TODO: Investigate more about the problem with getItem"}, {"metadata": {}, "cell_type": "markdown", "source": "# Create submission.csv file\nuses previously trained model from train-notebook"}, {"metadata": {"_uuid": "8f2839f25d086af736a60e9eeb907d3b93b6e0e5", "_cell_guid": "b1076dfc-b9ad-4769-8c92-a6c4dae69d19", "trusted": true, "ExecuteTime": {"end_time": "2026-09-13T10:58:58.118035500Z", "start_time": "2026-09-13T10:58:38.194840600Z"}}, "cell_type": "code", "source": "def init_test_loader():\n    transform_validation = T.Compose([\n        T.Resize(config['image_size'], interpolation=T.InterpolationMode.BICUBIC),\n        T.ToTensor(),\n        T.Normalize(config['mean'], config['std'])\n    ])\n    test_dataset = ImageDataset(submission, f'{INPUT_DIR}/test', transform_validation, has_label=False)\n    return DataLoader(test_dataset, batch_size=512, shuffle=False, num_workers=4, pin_memory=True)\n\nsubmission = pd.read_csv(f'{INPUT_DIR}/sample_submission.csv')\nprint(submission.columns.tolist(), len(submission))\n\ncheckpoint = torch.load(BEST_MODEL_PATH, map_location=DEVICE, weights_only=False)\nconfig = checkpoint['config']\nmodel = timm.create_model(\n    config['model_name'],\n    pretrained=False,\n    num_classes=config['num_classes']\n)\nmodel.load_state_dict(checkpoint['model'])\nmodel.to(DEVICE).eval()\nprint(f\"epoch {checkpoint['epoch']}, validation auc {checkpoint['validation_auc']:.4f}\")\n\ntest_loader = init_test_loader()\n\npredictions = []\nwith torch.inference_mode():\n    for x, y in test_loader:\n        x = x.to(DEVICE, non_blocking=True)\n        with torch.amp.autocast('cuda'):\n            logits = model(x).squeeze(1)\n        predictions.append(torch.sigmoid(logits.float()))\n\npredictions = torch.cat(predictions).cpu().numpy()\n\nsubmission['label'] = predictions\nsubmission.to_csv(f'{OUTPUT_DIR}/submission.csv', index=False)\n\n# sanity check\nprint(len(submission), submission.isna().sum().sum())\nprint(predictions.min(), predictions.max(), predictions.mean())\nprint(submission.head())", "outputs": [], "execution_count": 2}]}