{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"gpu","dataSources":[{"sourceId":11848,"databundleVersionId":862157,"sourceType":"competition"},{"sourceId":13261609,"sourceType":"datasetVersion","datasetId":8403706}],"dockerImageVersionId":31089,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Histopathologic Cancer Detection","metadata":{}},{"cell_type":"markdown","source":"# 1. Import libraries","metadata":{}},{"cell_type":"code","source":"!pip uninstall -y fastai\n!pip install fastai==1.0.61 \nimport fastai\nprint(fastai.__version__)\nimport warnings, torch\nwarnings.filterwarnings(\"ignore\", category=FutureWarning)\nimport fastai\nprint(\"Fastai version:\", fastai.__version__)\nprint(\"Torch version:\", torch.__version__) ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T12:54:47.523277Z","iopub.execute_input":"2025-10-06T12:54:47.523467Z","iopub.status.idle":"2025-10-06T12:56:07.784893Z","shell.execute_reply.started":"2025-10-06T12:54:47.523433Z","shell.execute_reply":"2025-10-06T12:56:07.784047Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os\nimport cv2\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nimport random\nfrom sklearn.utils import shuffle\nfrom tqdm import tqdm_notebook\n\nimport os\nbase_dir = '../input/histopathologic-cancer-detection/'\nprint(os.listdir(base_dir))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T12:56:07.787025Z","iopub.execute_input":"2025-10-06T12:56:07.787367Z","iopub.status.idle":"2025-10-06T12:56:08.770094Z","shell.execute_reply.started":"2025-10-06T12:56:07.787348Z","shell.execute_reply":"2025-10-06T12:56:08.769492Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Dataset","metadata":{}},{"cell_type":"code","source":"# Data paths\ntrain_path = '/kaggle/input/histopathologic-cancer-detection/train/'\ntest_path = '/kaggle/input/histopathologic-cancer-detection/test/'\n# Read CSV file\ntrain = pd.read_csv(\"/kaggle/input/histopathologic-cancer-train-test-index/train_split.csv\")\ntest = pd.read_csv(\"/kaggle/input/histopathologic-cancer-train-test-index/test_split.csv\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T12:56:08.770822Z","iopub.execute_input":"2025-10-06T12:56:08.771177Z","iopub.status.idle":"2025-10-06T12:56:09.254381Z","shell.execute_reply.started":"2025-10-06T12:56:08.771159Z","shell.execute_reply":"2025-10-06T12:56:09.253617Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 2. Data Preprocessing and augmentation","metadata":{}},{"cell_type":"code","source":"import random\nORIGINAL_SIZE = 96      # original size of the images - do not change\n\n# AUGMENTATION VARIABLES\nCROP_SIZE = 90          # final size after crop\nRANDOM_ROTATION = 3    # range (0-180), 180 allows all rotation variations, 0=no change\nRANDOM_SHIFT = 2        # center crop shift in x and y axes, 0=no change. This cannot be more than (ORIGINAL_SIZE - CROP_SIZE)//2 \nRANDOM_BRIGHTNESS = 7  # range (0-100), 0=no change\nRANDOM_CONTRAST = 5    # range (0-100), 0=no change\nRANDOM_90_DEG_TURN = 1  # 0 or 1= random turn to left or right\n\ndef readCroppedImage(path, augmentations = True):\n    # augmentations parameter is included for counting statistics from images, where we don't want augmentations\n    \n    # OpenCV reads the image in bgr format by default\n    bgr_img = cv2.imread(path)\n    # We flip it to rgb for visualization purposes\n    b,g,r = cv2.split(bgr_img)\n    rgb_img = cv2.merge([r,g,b])\n    \n    if(not augmentations):\n        return rgb_img / 255\n    \n    #random rotation\n    rotation = random.randint(-RANDOM_ROTATION,RANDOM_ROTATION)\n    if(RANDOM_90_DEG_TURN == 1):\n        rotation += random.randint(-1,1) * 90\n    M = cv2.getRotationMatrix2D((48,48),rotation,1)   # the center point is the rotation anchor\n    rgb_img = cv2.warpAffine(rgb_img,M,(96,96))\n    \n    #random x,y-shift\n    x = random.randint(-RANDOM_SHIFT, RANDOM_SHIFT)\n    y = random.randint(-RANDOM_SHIFT, RANDOM_SHIFT)\n    \n    # crop to center and normalize to 0-1 range\n    start_crop = (ORIGINAL_SIZE - CROP_SIZE) // 2\n    end_crop = start_crop + CROP_SIZE\n    rgb_img = rgb_img[(start_crop + x):(end_crop + x), (start_crop + y):(end_crop + y)] / 255\n    \n    # Random flip\n    flip_hor = bool(random.getrandbits(1))\n    flip_ver = bool(random.getrandbits(1))\n    if(flip_hor):\n        rgb_img = rgb_img[:, ::-1]\n    if(flip_ver):\n        rgb_img = rgb_img[::-1, :]\n        \n    # Random brightness\n    br = random.randint(-RANDOM_BRIGHTNESS, RANDOM_BRIGHTNESS) / 100.\n    rgb_img = rgb_img + br\n    \n    # Random contrast\n    cr = 1.0 + random.randint(-RANDOM_CONTRAST, RANDOM_CONTRAST) / 100.\n    rgb_img = rgb_img * cr\n    \n    # clip values to 0-1 range\n    rgb_img = np.clip(rgb_img, 0, 1.0)\n    \n    return rgb_img","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T12:56:09.255337Z","iopub.execute_input":"2025-10-06T12:56:09.255606Z","iopub.status.idle":"2025-10-06T12:56:09.264891Z","shell.execute_reply.started":"2025-10-06T12:56:09.255587Z","shell.execute_reply":"2025-10-06T12:56:09.264259Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 3. Baseline model (Fastai v1)","metadata":{}},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\n\n# we read the csv file earlier to pandas dataframe, now we set index to id so we can perform\ntrain_df = train.set_index('id')\n\ntrain_names = train_df.index.values\ntrain_labels = np.asarray(train_df['label'].values)\n\n# split, this function returns more than we need as we only need the validation indexes for fastai\ntr_n, tr_idx, val_n, val_idx = train_test_split(train_names, range(len(train_names)), test_size=0.1, stratify=train_labels, random_state=123)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T12:56:09.265600Z","iopub.execute_input":"2025-10-06T12:56:09.265806Z","iopub.status.idle":"2025-10-06T12:56:09.470930Z","shell.execute_reply.started":"2025-10-06T12:56:09.265790Z","shell.execute_reply":"2025-10-06T12:56:09.470362Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# fastai 1.0\nfrom fastai import *\nfrom fastai.vision import *\nfrom torchvision.models import *    # import *=all the models from torchvision  \n\narch = densenet169                  # specify model architecture, densenet169 seems to perform well for this data but you could experiment\nBATCH_SIZE = 128                    # specify batch size, hardware restrics this one. Large batch sizes may run out of GPU memory\nsz = CROP_SIZE                      # input size is the crop size\nMODEL_PATH = str(arch).split()[1]   # this will extrat the model name as the model file name ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T12:56:09.471678Z","iopub.execute_input":"2025-10-06T12:56:09.471984Z","iopub.status.idle":"2025-10-06T12:56:13.496688Z","shell.execute_reply.started":"2025-10-06T12:56:09.471959Z","shell.execute_reply":"2025-10-06T12:56:13.495991Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# create dataframe for the fastai loader\ntrain_dict = {'name': train_path + train_names, 'label': train_labels}\ndf = pd.DataFrame(data=train_dict)\n# create test dataframe\ntest_names = []\nfor f in os.listdir(test_path):\n    test_names.append(test_path + f)\ndf_test = pd.DataFrame(np.asarray(test_names), columns=['name'])\n# Subclass ImageList to use our own image opening function\nclass MyImageItemList(ImageList):\n    def open(self, fn:PathOrStr)->Image:\n        img = readCroppedImage(fn.replace('/./','').replace('//','/'))\n        # This ndarray image has to be converted to tensor before passing on as fastai Image, we can use pil2tensor\n        return vision.Image(px=pil2tensor(img, np.float32))\n    \n# Create ImageDataBunch using fastai data block API\nimgDataBunch = (MyImageItemList.from_df(path='/', df=df, suffix='.tif')\n        #Where to find the data?\n        .split_by_idx(val_idx)\n        #How to split in train/valid?\n        .label_from_df(cols='label')\n        #Where are the labels?\n        .add_test(MyImageItemList.from_df(path='/', df=df_test))\n        #dataframe pointing to the test set?\n        .transform(tfms=[[],[]], size=sz)\n        # We have our custom transformations implemented in the image loader but we could apply transformations also here\n        # Even though we don't apply transformations here, we set two empty lists to tfms. Train and Validation augmentations\n        .databunch(bs=BATCH_SIZE)\n        # convert to databunch\n        .normalize([tensor([0.702447, 0.546243, 0.696453]), tensor([0.238893, 0.282094, 0.216251])])\n        # Normalize with training set stats. These are means and std's of each three channel and we calculated these previously in the stats step.\n       )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T12:56:13.498900Z","iopub.execute_input":"2025-10-06T12:56:13.499284Z","iopub.status.idle":"2025-10-06T12:56:17.825296Z","shell.execute_reply.started":"2025-10-06T12:56:13.499266Z","shell.execute_reply":"2025-10-06T12:56:17.824761Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 4. Training ","metadata":{}},{"cell_type":"code","source":"# Next, we create a convnet learner object\n# ps = dropout percentage (0-1) in the final layer\ndef getLearner():\n    return create_cnn(imgDataBunch, arch, pretrained=True, path='.', metrics=accuracy, ps=0.5, callback_fns=ShowGraph)\n\nlearner = getLearner()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T12:56:17.825920Z","iopub.execute_input":"2025-10-06T12:56:17.826132Z","iopub.status.idle":"2025-10-06T12:56:19.291588Z","shell.execute_reply.started":"2025-10-06T12:56:17.826115Z","shell.execute_reply":"2025-10-06T12:56:19.291011Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### 1cycle policy","metadata":{}},{"cell_type":"code","source":"# We can use lr_find with different weight decays and record all losses so that we can plot them on the same graph\n# Number of iterations is by default 100, but at this low number of itrations, there might be too much variance\n# from random sampling that makes it difficult to compare WD's. I recommend using an iteration count of at least 300 for more consistent results.\nlrs = []\nlosses = []\nwds = []\niter_count = 600\n\n# WEIGHT DECAY = 1e-6\nlearner.lr_find(wd=1e-6, num_it=iter_count)\nlrs.append(learner.recorder.lrs)\nlosses.append(learner.recorder.losses)\nwds.append('1e-6')\nlearner = getLearner() #reset learner - this gets more consistent starting conditions\n\n# WEIGHT DECAY = 1e-4\nlearner.lr_find(wd=1e-4, num_it=iter_count)\nlrs.append(learner.recorder.lrs)\nlosses.append(learner.recorder.losses)\nwds.append('1e-4')\nlearner = getLearner() #reset learner - this gets more consistent starting conditions\n\n# WEIGHT DECAY = 1e-2\nlearner.lr_find(wd=1e-2, num_it=iter_count)\nlrs.append(learner.recorder.lrs)\nlosses.append(learner.recorder.losses)\nwds.append('1e-2')\nlearner = getLearner() #reset learner","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T12:56:19.292342Z","iopub.execute_input":"2025-10-06T12:56:19.292624Z","iopub.status.idle":"2025-10-06T13:02:00.480807Z","shell.execute_reply.started":"2025-10-06T12:56:19.292597Z","shell.execute_reply":"2025-10-06T13:02:00.480134Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot weight decays\n_, ax = plt.subplots(1,1)\nmin_y = 0.5\nmax_y = 0.55\nfor i in range(len(losses)):\n    ax.plot(lrs[i], losses[i])\n    min_y = min(np.asarray(losses[i]).min(), min_y)\nax.set_ylabel(\"Loss\")\nax.set_xlabel(\"Learning Rate\")\nax.set_xscale('log')\n#ax ranges may need some tuning with different model architectures \nax.set_xlim((1e-3,3e-1))\nax.set_ylim((min_y - 0.02,max_y))\nax.legend(wds)\nax.xaxis.set_major_formatter(plt.FormatStrFormatter('%.0e'))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T13:02:00.482110Z","iopub.execute_input":"2025-10-06T13:02:00.482380Z","iopub.status.idle":"2025-10-06T13:02:00.807569Z","shell.execute_reply.started":"2025-10-06T13:02:00.482358Z","shell.execute_reply":"2025-10-06T13:02:00.806620Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"max_lr = 2e-2\nwd = 1e-4\n# 1cycle policy\nlearner.fit_one_cycle(cyc_len=8, max_lr=max_lr, wd=wd)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T13:02:00.808536Z","iopub.execute_input":"2025-10-06T13:02:00.808787Z","iopub.status.idle":"2025-10-06T13:25:07.152594Z","shell.execute_reply.started":"2025-10-06T13:02:00.808771Z","shell.execute_reply":"2025-10-06T13:25:07.151937Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# plot learning rate of the one cycle\nlearner.recorder.plot_lr()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T13:25:07.153700Z","iopub.execute_input":"2025-10-06T13:25:07.154433Z","iopub.status.idle":"2025-10-06T13:25:07.300100Z","shell.execute_reply.started":"2025-10-06T13:25:07.154407Z","shell.execute_reply":"2025-10-06T13:25:07.299289Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# predict the validation set with our model\ninterp = ClassificationInterpretation.from_learner(learner)\ninterp.plot_confusion_matrix(title='Confusion matrix')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T13:25:07.300962Z","iopub.execute_input":"2025-10-06T13:25:07.301235Z","iopub.status.idle":"2025-10-06T13:25:18.197991Z","shell.execute_reply.started":"2025-10-06T13:25:07.301211Z","shell.execute_reply":"2025-10-06T13:25:18.197211Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# before we continue, lets save the model at this stage\nlearner.save(MODEL_PATH + '_stage1')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T13:25:18.199556Z","iopub.execute_input":"2025-10-06T13:25:18.199770Z","iopub.status.idle":"2025-10-06T13:25:18.443991Z","shell.execute_reply.started":"2025-10-06T13:25:18.199749Z","shell.execute_reply":"2025-10-06T13:25:18.443384Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import os\n\nprint(\"Current working directory:\", os.getcwd())\nprint(\"Model will be saved to:\", MODEL_PATH + '_stage1.pth')\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T13:25:18.445558Z","iopub.execute_input":"2025-10-06T13:25:18.445756Z","iopub.status.idle":"2025-10-06T13:25:18.450676Z","shell.execute_reply.started":"2025-10-06T13:25:18.445740Z","shell.execute_reply":"2025-10-06T13:25:18.449887Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 5. Finetuning the baseline model","metadata":{}},{"cell_type":"code","source":"import torch\n\nstate = torch.load(learner.path / 'models' / f'{MODEL_PATH}_stage1.pth', weights_only=False)\nlearner.model.load_state_dict(state['model'])\n\nlearner.unfreeze()\nlearner.lr_find(wd=wd)\nlearner.recorder.plot()\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T14:19:39.153124Z","iopub.execute_input":"2025-10-06T14:19:39.153626Z","iopub.status.idle":"2025-10-06T14:19:55.162023Z","shell.execute_reply.started":"2025-10-06T14:19:39.153602Z","shell.execute_reply":"2025-10-06T14:19:55.161198Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Now, smaller learning rates. This time we define the min and max lr of the cycle\nlearner.fit_one_cycle(cyc_len=12, max_lr=slice(4e-5,4e-4))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T15:43:54.283386Z","iopub.execute_input":"2025-10-06T15:43:54.284225Z","iopub.status.idle":"2025-10-06T16:24:58.406837Z","shell.execute_reply.started":"2025-10-06T15:43:54.284176Z","shell.execute_reply":"2025-10-06T16:24:58.406101Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfrom scipy.signal import savgol_filter\nimport pandas as pd\n\ndef plot_loss_vs_epoch(learner, smooth_window=5, polyorder=2):\n    recorder = learner.recorder\n\n    # Get data \n    train_loss = [float(l) for l in recorder.losses]       # theo batch\n    val_loss = [float(l) for l in recorder.val_losses]     # theo epoch\n    n_epoch = len(val_loss)\n    \n\n    # Compute train loss \n    batch_per_epoch = len(train_loss) // n_epoch\n    train_loss_epoch = [\n        sum(train_loss[i*batch_per_epoch:(i+1)*batch_per_epoch]) / batch_per_epoch\n        for i in range(n_epoch)\n    ]\n\n    # (smooth)\n    if n_epoch >= smooth_window:  # chỉ làm mượt khi dữ liệu đủ dài\n        train_smooth = savgol_filter(train_loss_epoch, smooth_window, polyorder)\n        val_smooth = savgol_filter(val_loss, smooth_window, polyorder)\n    else:\n        train_smooth, val_smooth = train_loss_epoch, val_loss\n\n    # DataFrame\n    df = pd.DataFrame({\n        'epoch': range(n_epoch),\n        'train_loss': train_smooth,\n        'valid_loss': val_smooth\n    })\n\n\n    plt.figure(figsize=(8,5))\n    plt.plot(df['epoch'], df['train_loss'], label='Train Loss', color='steelblue', linewidth=2.5)\n    plt.plot(df['epoch'], df['valid_loss'], label='Validation Loss', color='darkorange', linewidth=2.5)\n    plt.title('Train vs Validation Loss per Epoch', fontsize=14)\n    plt.xlabel('Epoch', fontsize=12)\n    plt.ylabel('Loss', fontsize=12)\n    plt.legend()\n    plt.grid(True, linestyle='--', alpha=0.6)\n    plt.show()\n    \nplot_loss_vs_epoch(learner)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T16:30:20.511507Z","iopub.execute_input":"2025-10-06T16:30:20.511777Z","iopub.status.idle":"2025-10-06T16:30:20.700275Z","shell.execute_reply.started":"2025-10-06T16:30:20.511758Z","shell.execute_reply":"2025-10-06T16:30:20.699503Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"learner.recorder.plot_losses()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T14:06:52.852676Z","iopub.execute_input":"2025-10-06T14:06:52.852943Z","iopub.status.idle":"2025-10-06T14:06:53.379794Z","shell.execute_reply.started":"2025-10-06T14:06:52.852918Z","shell.execute_reply":"2025-10-06T14:06:53.378900Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# lets take a second look at the confusion matrix. See if how much we improved.\ninterp = ClassificationInterpretation.from_learner(learner)\ninterp.plot_confusion_matrix(title='Confusion matrix')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T14:06:53.380732Z","iopub.execute_input":"2025-10-06T14:06:53.381052Z","iopub.status.idle":"2025-10-06T14:07:04.161333Z","shell.execute_reply.started":"2025-10-06T14:06:53.381023Z","shell.execute_reply":"2025-10-06T14:07:04.160331Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save the finetuned model\nlearner.save(MODEL_PATH + '_stage2')\n\n# if the model was better before finetuning, uncomment this to load the previous stage\n#learner.load(MODEL_PATH + '_stage1')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T14:07:04.162890Z","iopub.execute_input":"2025-10-06T14:07:04.163261Z","iopub.status.idle":"2025-10-06T14:07:04.571055Z","shell.execute_reply.started":"2025-10-06T14:07:04.163234Z","shell.execute_reply":"2025-10-06T14:07:04.570495Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# 6. Validation and analysis","metadata":{}},{"cell_type":"code","source":"from sklearn.metrics import accuracy_score, precision_score, recall_score, f1_score, roc_curve, auc, classification_report, confusion_matrix\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nimport numpy as np\nimport torch\n\n# Get predictions from learner \npreds, y_true, loss = learner.get_preds(with_loss=True)\nprobs = preds[:, 1].numpy()  # Probability of class 1\ny_true = y_true.numpy()\n\n\ndef evaluate_model_fastai(y_true, probs, min_tpr=0.95, tune_threshold=True):\n    \"\"\"\n    Evaluate the model based on the output of learner.get_preds().\n    Can automatically tune the threshold prioritizing Recall (TPR).\n    \"\"\"\n    # Tune threshold \n    threshold = 0.5\n    if tune_threshold:\n        fpr, tpr, thresholds = roc_curve(y_true, probs)\n        valid_idx = np.where(tpr >= min_tpr)[0]\n\n        if len(valid_idx) == 0:\n            print(f\"No threshold achieves TPR >= {min_tpr}. The threshold with the highest TPR will be chosen.\")\n            best_idx = np.argmax(tpr)\n        else:\n            best_idx = valid_idx[np.argmin(fpr[valid_idx])]\n\n        threshold = thresholds[best_idx]\n        print(f\"\\nTuning threshold (TPR priority): {threshold:.4f}\")\n        print(f\"TPR={tpr[best_idx]:.4f}, FPR={fpr[best_idx]:.4f}\")\n\n    # Apply threshold\n    y_pred = (probs >= threshold).astype(int)\n\n    # Metrics\n    acc  = accuracy_score(y_true, y_pred)\n    prec = precision_score(y_true, y_pred)\n    rec  = recall_score(y_true, y_pred)\n    f1   = f1_score(y_true, y_pred)\n    roc_auc = auc(*roc_curve(y_true, probs)[:2])\n\n    print(\"\\n=== Evaluation Report ===\")\n    print(f\"Accuracy : {acc:.4f}\")\n    print(f\"Precision: {prec:.4f}\")\n    print(f\"Recall   : {rec:.4f}\")\n    print(f\"F1-score : {f1:.4f}\")\n    print(f\"ROC AUC  : {roc_auc:.4f}\")\n    print(f\"Threshold: {threshold:.4f}\")\n\n    print(\"\\nClassification Report:\\n\", classification_report(y_true, y_pred, digits=4))\n\n    # Confusion Matrix\n    cm = confusion_matrix(y_true, y_pred)\n    plt.figure(figsize=(5, 4))\n    sns.heatmap(cm, annot=True, fmt=\"d\", cmap=\"Blues\", xticklabels=[0,1], yticklabels=[0,1])\n    plt.xlabel(\"Predicted\")\n    plt.ylabel(\"True\")\n    plt.title(f\"Confusion Matrix (Threshold={threshold:.3f})\")\n    plt.show()\n\n    # ROC Curve\n    fpr, tpr, _ = roc_curve(y_true, probs)\n    plt.figure(figsize=(6,6))\n    plt.plot(fpr, tpr, label=f'ROC Curve (AUC = {roc_auc:.3f})')\n    plt.plot([0,1],[0,1],'k--')\n    plt.xlabel('False Positive Rate')\n    plt.ylabel('True Positive Rate')\n    plt.title('Receiver Operating Characteristic (ROC) Curve')\n    plt.legend(loc='lower right')\n    plt.show()\n\n    return {\n        \"accuracy\": acc,\n        \"precision\": prec,\n        \"recall\": rec,\n        \"f1\": f1,\n        \"auc\": roc_auc,\n        \"threshold\": threshold,\n        \"confusion_matrix\": cm\n    }\n\nresults = evaluate_model_fastai(y_true, probs, tune_threshold=True, min_tpr=0.95)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-10-06T14:15:06.165556Z","iopub.execute_input":"2025-10-06T14:15:06.166123Z","iopub.status.idle":"2025-10-06T14:15:16.436869Z","shell.execute_reply.started":"2025-10-06T14:15:06.166100Z","shell.execute_reply":"2025-10-06T14:15:16.436007Z"}},"outputs":[],"execution_count":null}]}