{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ====================================================\n# Library\n# ====================================================\nimport sys\nsys.path.append('../input/pytorch-image-models/pytorch-image-models-master')\n\nimport os\nimport math\nimport time\nimport random\nimport shutil\nfrom pathlib import Path\nfrom contextlib import contextmanager\nfrom collections import defaultdict, Counter\n\nimport scipy as sp\nimport numpy as np\nimport pandas as pd\n\nfrom sklearn import preprocessing\nfrom sklearn.metrics import roc_auc_score\nfrom sklearn.model_selection import StratifiedKFold, GroupKFold, KFold\n\nfrom tqdm.auto import tqdm\nfrom functools import partial\n\nimport cv2\nfrom PIL import Image\n\nimport torch\nimport torch.nn as nn\nimport torch.nn.functional as F\nfrom torch.optim import Adam, SGD\nimport torchvision.models as models\nfrom torch.nn.parameter import Parameter\nfrom torch.utils.data import DataLoader, Dataset\nfrom torch.optim.lr_scheduler import CosineAnnealingWarmRestarts, CosineAnnealingLR, ReduceLROnPlateau\n\nfrom albumentations import (\n    Compose, OneOf, Normalize, Resize, RandomResizedCrop, RandomCrop, HorizontalFlip, VerticalFlip, \n    RandomBrightness, RandomContrast, RandomBrightnessContrast, Rotate, ShiftScaleRotate, Cutout, \n    IAAAdditiveGaussianNoise, Transpose\n    )\nfrom albumentations.pytorch import ToTensorV2\nfrom albumentations import ImageOnlyTransform\n!pip install timm\nimport timm\n\nfrom torch.cuda.amp import autocast, GradScaler","metadata":{"execution":{"iopub.status.busy":"2022-03-06T02:11:56.305732Z","iopub.execute_input":"2022-03-06T02:11:56.306117Z","iopub.status.idle":"2022-03-06T02:12:09.65652Z","shell.execute_reply.started":"2022-03-06T02:11:56.30607Z","shell.execute_reply":"2022-03-06T02:12:09.655559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.listdir('../input/ranzcr-clip-catheter-line-classification')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/ranzcr-clip-catheter-line-classification/train.csv')\ntest = pd.read_csv('../input/ranzcr-clip-catheter-line-classification/sample_submission.csv')\ndisplay(train.head())\ndisplay(test.head())","metadata":{"execution":{"iopub.status.busy":"2022-03-06T02:12:27.431424Z","iopub.execute_input":"2022-03-06T02:12:27.431865Z","iopub.status.idle":"2022-03-06T02:12:27.733742Z","shell.execute_reply.started":"2022-03-06T02:12:27.431827Z","shell.execute_reply":"2022-03-06T02:12:27.733091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"target_cols = ['ETT - Abnormal', 'ETT - Borderline', 'ETT - Normal', 'NGT - Abnormal', \n               'NGT - Borderline', 'NGT - Incompletely Imaged', 'NGT - Normal', 'CVC - Abnormal',\n               'CVC - Borderline', 'CVC - Normal', 'Swan Ganz Catheter Present']\n\noutput_dir = './'\n\n","metadata":{"execution":{"iopub.status.busy":"2022-03-06T02:12:44.762823Z","iopub.execute_input":"2022-03-06T02:12:44.763164Z","iopub.status.idle":"2022-03-06T02:12:44.768384Z","shell.execute_reply.started":"2022-03-06T02:12:44.763131Z","shell.execute_reply":"2022-03-06T02:12:44.767056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device('cuda')\n\ndevice","metadata":{"execution":{"iopub.status.busy":"2022-03-06T02:12:43.041902Z","iopub.execute_input":"2022-03-06T02:12:43.042273Z","iopub.status.idle":"2022-03-06T02:12:43.051767Z","shell.execute_reply.started":"2022-03-06T02:12:43.042241Z","shell.execute_reply":"2022-03-06T02:12:43.050215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#### Get our final score\n\ndef get_score(y_true, y_pred):\n    scores = []\n    for i in range(y_true.shape[1]):\n        score = roc_auc_score(y_true[:,i], y_pred[:,i])\n        scores.append(score)\n    avg_score = np.mean(scores)\n    return avg_score, scores\n\n","metadata":{"execution":{"iopub.status.busy":"2022-03-06T02:12:47.212884Z","iopub.execute_input":"2022-03-06T02:12:47.213252Z","iopub.status.idle":"2022-03-06T02:12:47.219266Z","shell.execute_reply.started":"2022-03-06T02:12:47.21322Z","shell.execute_reply":"2022-03-06T02:12:47.218353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"####Get training data\nTRAIN_PATH = '/kaggle/input/ranzcr-clip-catheter-line-classification/train'\n\nclass TrainData(Dataset):\n    \n    def __init__(self, df, train_transforms=None):\n        \n        self.df = df\n        self.file_names = df['StudyInstanceUID'].values\n        self.labels = df[target_cols].values\n        self.train_transforms = train_transforms\n        \n    def __len__(self):\n        \n        return len(self.df)\n    \n    def __getitem__(self, idx):\n        \n        file_name = self.file_names[idx]\n        file_path = f'{TRAIN_PATH}/{file_name}.jpg'\n        image = cv2.imread(file_path)\n        \n        image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n        \n        if self.train_transforms:\n            \n            new_img = self.train_transforms(image=image)\n            \n        new_img = new_img['image']\n            \n        label = torch.tensor(self.labels[idx]).float()\n        #print(new_img.shape)\n        \n        return (new_img, label)\n        ","metadata":{"execution":{"iopub.status.busy":"2022-03-06T02:13:37.314041Z","iopub.execute_input":"2022-03-06T02:13:37.314425Z","iopub.status.idle":"2022-03-06T02:13:37.323358Z","shell.execute_reply.started":"2022-03-06T02:13:37.314393Z","shell.execute_reply":"2022-03-06T02:13:37.322488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"####Image Size\n\nsize = 600\n\n\n###Transforms for augmented images\n\ndef get_transforms(data):\n    \n    if data == 'train':\n        \n        return Compose([\n            RandomResizedCrop(384, 384),\n            HorizontalFlip(p=0.5),\n            Normalize(\n                mean=[0.485, 0.456, 0.406],\n                std=[0.229, 0.224, 0.225],\n            ),\n            ToTensorV2(),\n        ])\n\n    elif data == 'valid':\n        \n        return Compose([\n            Resize(256, 384),\n            Normalize(\n                mean=[0.485, 0.456, 0.406],\n                std=[0.229, 0.224, 0.225],\n            ),\n            ToTensorV2(),\n        ])\n        ","metadata":{"execution":{"iopub.status.busy":"2022-03-06T02:13:33.814117Z","iopub.execute_input":"2022-03-06T02:13:33.814466Z","iopub.status.idle":"2022-03-06T02:13:33.82261Z","shell.execute_reply.started":"2022-03-06T02:13:33.814437Z","shell.execute_reply":"2022-03-06T02:13:33.82133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class ResNet(nn.Module):\n    \n    def __init__(self, model_name='resnext50_32x4d', pretrained=False):\n        super(ResNet, self).__init__()\n        \n        self.model = timm.create_model(model_name, pretrained=pretrained)\n        in_features = self.model.fc.in_features\n        self.model.fc = nn.Linear(in_features, 11)\n        \n    def forward(self, x):\n        \n        x = self.model(x)\n        \n        return x","metadata":{"execution":{"iopub.status.busy":"2022-03-06T02:13:40.822929Z","iopub.execute_input":"2022-03-06T02:13:40.823297Z","iopub.status.idle":"2022-03-06T02:13:40.829677Z","shell.execute_reply.started":"2022-03-06T02:13:40.823265Z","shell.execute_reply":"2022-03-06T02:13:40.828854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train_fn(train_loader, model, criterion, optimizer, epoch, scheduler, device):\n    \n    scaler = GradScaler()\n    \n    # switch model to train mode\n    model.train()\n    global_step = 0\n    \n    start_time = time.time()\n    \n    \n    for step, (images, labels) in enumerate(train_loader):\n        \n        #get image and label\n        images = images.to(device)\n        labels = labels.to(device)\n        batch_size = labels.size(0)\n        \n       \n        output = model(images)\n        loss = criterion(output, labels)\n        # backpropogate the loss\n        scaler.scale(loss).backward()\n        grad_norm = torch.nn.utils.clip_grad_norm_(model.parameters(), 1000)\n            \n        scaler.step(optimizer)\n        scaler.update()\n        optimizer.zero_grad()\n        global_step += 1\n            \n        # measure elapsed time\n        if step % 100 == 0 or step == (len(train_loader)-1):\n            \n            print('Epoch: [{0}][{1}/{2}] '\n                  'Loss: {loss:.4f} '\n                  #'Grad: {grad_norm:.4f}  '\n                  'LR: {lr} '\n                  'Elapsed: {elapsed:.4f}'\n\n                  .format(\n                   epoch+1, step, len(train_loader),\n                   loss=loss,\n                   #grad_norm=grad_norm.item(),\n                   lr=scheduler.get_last_lr(),\n                   elapsed=time.time() - start_time\n                   ))\n            \n    return loss.item()\n\n\ndef valid_fn(valid_loader, model, criterion, device):\n\n    # switch to evaluation mode\n    model.eval()\n    preds = []\n    \n    for step, (images, labels) in enumerate(valid_loader):\n        \n        images = images.to(device)\n        labels = labels.to(device)\n        batch_size = labels.size(0)\n        # compute loss\n        with torch.no_grad():\n            y_preds = model(images)\n            \n        loss = criterion(y_preds, labels)\n        \n        # record accuracy\n        preds.append(y_preds.sigmoid().to('cpu').numpy())\n        \n        if step % 100 == 0 or step == (len(valid_loader)-1):\n            \n            print('Step: [{0}/{1}] '\n                  'Loss: {loss:.4f}'\n                  #'Grad: {grad_norm:.4f}  '\n\n                  .format(\n                   step, len(valid_loader),\n                   loss=loss.item(),\n                   #grad_norm=grad_norm,\n                   #lr=scheduler.get_lr()[0],\n                   ))\n            \n            \n            \n    predictions = np.concatenate(preds)\n    \n    return loss.item(), predictions","metadata":{"execution":{"iopub.status.busy":"2022-03-06T02:13:46.646305Z","iopub.execute_input":"2022-03-06T02:13:46.646626Z","iopub.status.idle":"2022-03-06T02:13:46.663608Z","shell.execute_reply.started":"2022-03-06T02:13:46.646593Z","shell.execute_reply":"2022-03-06T02:13:46.662608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"training = True\ntrn_fold = [0, 1, 2, 3]\n\n# ====================================================\n# Train loop\n# ====================================================\ndef train_loop(folds, fold):\n\n    # ====================================================\n    # loader\n    # ====================================================\n    trn_idx = folds[folds['fold'] != fold].index\n    val_idx = folds[folds['fold'] == fold].index\n\n    train_folds = folds.loc[trn_idx].reset_index(drop=True)\n    valid_folds = folds.loc[val_idx].reset_index(drop=True)\n    valid_labels = valid_folds[target_cols].values\n\n    train_dataset = TrainData(train_folds, \n                                 train_transforms=get_transforms(data='train'))\n    valid_dataset = TrainData(valid_folds, \n                                 train_transforms=get_transforms(data='valid'))\n\n    train_loader = DataLoader(train_dataset, \n                              batch_size=32, \n                              shuffle=True, \n                              pin_memory=True, drop_last=True)\n    \n    valid_loader = DataLoader(valid_dataset, \n                              batch_size=32 * 2, \n                              shuffle=False, \n                              pin_memory=True, drop_last=False)\n\n    # ====================================================\n    # model & optimizer\n    # ====================================================\n    model_name = 'resnext50_32x4d'\n    model = ResNet()\n    model.to(device)\n\n    optimizer = Adam(model.parameters(), lr=1e-4, weight_decay=1e-6, amsgrad=False)\n    scheduler = CosineAnnealingWarmRestarts(optimizer, T_0=6, T_mult=1, eta_min=1e-6, last_epoch=-1)\n\n    # ====================================================\n    # loop\n    # ====================================================\n    criterion = nn.BCEWithLogitsLoss().to(device)\n\n    best_score = 0.\n    best_loss = np.inf\n    \n    for epoch in range(6):\n        \n        start_time = time.time()\n        \n        # train\n        loss = train_fn(train_loader, model, criterion, optimizer, epoch, scheduler, device)\n\n        # eval\n        val_loss, preds = valid_fn(valid_loader, model, criterion, device)\n        \n        scheduler.step()\n\n        # scoring\n        score, scores = get_score(valid_labels, preds)\n\n        elapsed = time.time() - start_time\n\n        print(f\"Loss: {loss} Elapsed: {elapsed}\")\n        \n        \n        \n    \n    check_point = torch.load(\"/kaggle/output/\"+f'{model_name}_fold{fold}_best.pth')\n    for c in [f'pred_{c}' for c in target_cols]:\n        valid_folds[c] = np.nan\n    valid_folds[[f'pred_{c}' for c in target_cols]] = check_point['preds']\n\n    return valid_folds","metadata":{"execution":{"iopub.status.busy":"2022-03-06T02:18:47.171922Z","iopub.execute_input":"2022-03-06T02:18:47.172286Z","iopub.status.idle":"2022-03-06T02:18:47.189Z","shell.execute_reply.started":"2022-03-06T02:18:47.172255Z","shell.execute_reply":"2022-03-06T02:18:47.188142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folds = train.copy()\nFold = GroupKFold(n_splits=4)\ngroups = folds['PatientID'].values\nfor n, (train_index, val_index) in enumerate(Fold.split(folds, folds[target_cols], groups)):\n    folds.loc[val_index, 'fold'] = int(n)\nfolds['fold'] = folds['fold'].astype(int)\ndisplay(folds.groupby('fold').size())","metadata":{"execution":{"iopub.status.busy":"2022-03-06T02:18:47.513337Z","iopub.execute_input":"2022-03-06T02:18:47.513652Z","iopub.status.idle":"2022-03-06T02:18:47.589185Z","shell.execute_reply.started":"2022-03-06T02:18:47.513623Z","shell.execute_reply":"2022-03-06T02:18:47.588316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def main():\n    \n    oof_df = pd.DataFrame()\n    for fold in range(4):\n        if fold in trn_fold:\n\n            _oof_df = train_loop(folds, fold)\n            oof_df = pd.concat([oof_df, _oof_df])\n            print(f\"========== fold: {fold} result ==========\")\n            preds = oof_df[[f'pred_{c}' for c in target_cols]].values\n            labels = result_df[target_cols].values\n            score, scores = get_score(labels, preds)\n            print(f'Score: {score:<.4f}  Scores: {np.round(scores, decimals=4)}')\n                \n    # CV result\n    print(f\"========== CV ==========\")\n    get_result(oof_df)\n    # save result\n    oof_df.to_csv(OUTPUT_DIR+'oof_df.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-06T02:18:47.840694Z","iopub.execute_input":"2022-03-06T02:18:47.840981Z","iopub.status.idle":"2022-03-06T02:18:47.84845Z","shell.execute_reply.started":"2022-03-06T02:18:47.840953Z","shell.execute_reply":"2022-03-06T02:18:47.847611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"main()","metadata":{"execution":{"iopub.status.busy":"2022-03-06T02:19:04.867163Z","iopub.execute_input":"2022-03-06T02:19:04.867601Z","iopub.status.idle":"2022-03-06T02:19:05.359211Z","shell.execute_reply.started":"2022-03-06T02:19:04.867558Z","shell.execute_reply":"2022-03-06T02:19:05.357197Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}