{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install wtfml==0.0.2","metadata":{"execution":{"iopub.status.busy":"2022-02-23T07:39:23.861299Z","iopub.execute_input":"2022-02-23T07:39:23.861998Z","iopub.status.idle":"2022-02-23T07:39:36.378133Z","shell.execute_reply.started":"2022-02-23T07:39:23.861870Z","shell.execute_reply":"2022-02-23T07:39:36.377294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install efficientnet_pytorch\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nfrom PIL import Image\n\nfrom sklearn import model_selection\nfrom sklearn import metrics\n\nimport torch\nimport torch.nn as nn\nfrom torch.nn import functional as F\nimport torch.optim as optim\nimport efficientnet_pytorch\n\nimport albumentations as A\n\nfrom wtfml.utils import EarlyStopping\nfrom wtfml.engine import Engine\nfrom wtfml.data_loaders.image import ClassificationLoader\n\ndevice = torch.device(\"cuda\" if torch.cuda.is_available() else \"cpu\")","metadata":{"execution":{"iopub.status.busy":"2022-02-23T07:54:56.535572Z","iopub.execute_input":"2022-02-23T07:54:56.535911Z","iopub.status.idle":"2022-02-23T07:55:10.420214Z","shell.execute_reply.started":"2022-02-23T07:54:56.535877Z","shell.execute_reply":"2022-02-23T07:55:10.419177Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-02-23T07:58:25.162887Z","iopub.execute_input":"2022-02-23T07:58:25.163862Z","iopub.status.idle":"2022-02-23T07:58:25.167863Z","shell.execute_reply.started":"2022-02-23T07:58:25.163787Z","shell.execute_reply":"2022-02-23T07:58:25.167035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\nimport pandas as pd\nfrom sklearn import model_selection\n\n# Training data is in a csv file called train.csv\ndf = pd.read_csv(\"../input/siim-isic-melanoma-classification/train.csv\")\n# we create a new column called kfold and fill it with -1\ndf[\"kfold\"] = -1\n# the next step is to randomize the rows of the data\ndf = df.sample(frac=1).reset_index(drop=True)\n# fetch targets\ny = df.target.values\n# initiate the kfold class from model_selection module\nkf = model_selection.StratifiedKFold(n_splits=5)\n# fill the new kfold column\n\nfor f, (t_, v_) in enumerate(kf.split(X=df, y=y)):\n    df.loc[v_, 'kfold'] = f\n# save the new csv with kfold column\ndf.to_csv(\"train_folds.csv\", index=False)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-02-23T08:04:32.000494Z","iopub.execute_input":"2022-02-23T08:04:32.000798Z","iopub.status.idle":"2022-02-23T08:04:32.310539Z","shell.execute_reply.started":"2022-02-23T08:04:32.000761Z","shell.execute_reply":"2022-02-23T08:04:32.309297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_folds =pd.read_csv(\"train_folds.csv\")\ndf_train_folds","metadata":{"execution":{"iopub.status.busy":"2022-02-23T08:04:57.510323Z","iopub.execute_input":"2022-02-23T08:04:57.510627Z","iopub.status.idle":"2022-02-23T08:04:57.598693Z","shell.execute_reply.started":"2022-02-23T08:04:57.510592Z","shell.execute_reply":"2022-02-23T08:04:57.597725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = '../input/siim-isic-melanoma-classification/jpeg/train'\ntest_dir = '../input/siim-isic-melanoma-classification/jpeg/test'\nt = os.listdir(train_dir)\n\nt1 = os.listdir(test_dir)\nprint(len(t),len(t1),len(t)+len(t1))","metadata":{"execution":{"iopub.status.busy":"2022-02-23T08:19:20.632254Z","iopub.execute_input":"2022-02-23T08:19:20.632834Z","iopub.status.idle":"2022-02-23T08:19:21.963591Z","shell.execute_reply.started":"2022-02-23T08:19:20.632797Z","shell.execute_reply":"2022-02-23T08:19:21.962408Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''\nimg2np = np.array(img)\nimg2np.shape\n\nten = torch.from_numpy(img2np)\nten.shape\n'''","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def train(fold):\n    train_dir = train_dir\n    df = pd.read_csv('./train_folds.csv')\n    device = \"cuda\"\n    epochs = 50\n    train_bs = 32\n    valid_bs = 16\n    \n    df_train = df[df.kfold != fold].reset_index(drop=True)\n    df_valid = df[df.kfold == fold].reset_index(drop=True)\n    \n    # Normalize the images\n    mean = (0.485, 0.456, 0.406)\n    std = (0.229, 0.224, 0.225)\n    ","metadata":{"execution":{"iopub.status.busy":"2022-02-23T09:00:32.892080Z","iopub.execute_input":"2022-02-23T09:00:32.892733Z","iopub.status.idle":"2022-02-23T09:00:32.900425Z","shell.execute_reply.started":"2022-02-23T09:00:32.892690Z","shell.execute_reply":"2022-02-23T09:00:32.899441Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#train\ntrain_dir :train_dir\n\ntrain_aug = A.Compose(\n        [\n            A.Normalize(mean, std, max_pixel_value=255.0, always_apply=True),\n            A.ShiftScaleRotate(shift_limit=0.0625, scale_limit=0.1, rotate_limit=15),\n            A.Flip(p=0.5)  \n        ]          \n    )\n\ntrain_images = df_train.image_name.values.tolist()\ntrain_images = [os.path.join(train_dir,i +'.png') for i in train_images]\ntrain_targets = df_train.targets.values\n\ntrain_dataset = ClassificationLoader(\n    image_path = train_images,\n    targets = train_targets,\n    resize = 220*220,\n    augmentation = train_aug\n)\n\ntrain_loader = torch.utils.data.DataLoader(\n        train_dataset, batch_size=train_bs, shuffle=True, num_workers=4)","metadata":{"execution":{"iopub.status.busy":"2022-02-23T09:00:23.728932Z","iopub.execute_input":"2022-02-23T09:00:23.729214Z","iopub.status.idle":"2022-02-23T09:00:23.754968Z","shell.execute_reply.started":"2022-02-23T09:00:23.729170Z","shell.execute_reply":"2022-02-23T09:00:23.753957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Valid\nvalid_images = df_valid.image_name.values.tolist()\nvalid_images = [os.path.join(training_data_path, i + \".png\") for i in valid_images]\nvalid_targets = df_valid.target.values\n\nvalid_aug = A.Compose([\n    A.Normalize(mean, std, max_pixel_value=255.0, always_apply=True)\n])\n\nvalid_dataset = ClassificationLoader(\n    image_paths=valid_images,\n    targets=valid_targets,\n    resize=None,\n    augmentations=valid_aug,\n)\n\nvalid_loader = torch.utils.data.DataLoader(\n        valid_dataset, batch_size=valid_bs, shuffle=False, num_workers=4)","metadata":{"execution":{"iopub.status.busy":"2022-02-23T09:00:38.305140Z","iopub.execute_input":"2022-02-23T09:00:38.305457Z","iopub.status.idle":"2022-02-23T09:00:38.329205Z","shell.execute_reply.started":"2022-02-23T09:00:38.305425Z","shell.execute_reply":"2022-02-23T09:00:38.328126Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class EfficientNet(nn.Module):\n    def __init__(self):\n        super(EfficientNet, self).__init__()\n        self.base_model = efficientnet_pytorch.EfficientNet.from_pretrained(\n            'efficientnet-b4'\n        )\n        self.base_model._fc = nn.Linear(\n            in_features=1792, \n            out_features=1, \n            bias=True\n        )\n        \n    def forward(self, image, targets):\n        out = self.base_model(image)\n        loss = nn.BCEWithLogitsLoss()(out, targets.view(-1, 1).type_as(out))\n        return out, loss","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = EfficientNet()\nmodel.to(device)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"optimizer = torch.optim.Adam(model.parameters(), lr=1e-4)\nscheduler = torch.optim.lr_scheduler.ReduceLROnPlateau(\n        optimizer,\n        patience=3,\n        threshold=0.001,\n        mode=\"max\"\n    )\n\nes = EarlyStopping(patience=5, mode=\"max\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"epochs = 50\nfor epoch in range(epochs):\n        train_loss = Engine.train(train_loader, model, optimizer, device=device)\n        predictions, valid_loss = Engine.evaluate(\n            valid_loader, model, device=device\n        )\n        predictions = np.vstack((predictions)).ravel()\n        auc = metrics.roc_auc_score(valid_targets, predictions)\n        print(f\"Epoch = {epoch}, AUC = {auc}\")\n        scheduler.step(auc)\n\n        es(auc, model, model_path=f\"model_fold_{fold}.bin\")\n        if es.early_stop:\n            print(\"Early stopping\")\n            break","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# torch.save\ntorch.save(model.state_dict(), './modetor.pt')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#Test\ndef predict(fold):\n    print(f\"Generating Predictions for saved model, fold = {fold+1}\")\n    test_data_path = \"/kaggle/input/siic-isic-224x224-images/test\"\n    df_test = pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/test.csv\")\n    df_test.loc[:,'target'] = 0\n    \n    #model_path = \"f'/kaggle/working/model_fold{fold}'\"\n    #model_path = '/kaggle/working/model_fold0_epoch0.bin'\n    model_path = './model_fold_0.bin'\n    \n    device = 'cuda'\n    \n    test_bs = 16\n    \n    mean = (0.485, 0.456, 0.406)\n    std = (0.229, 0.224, 0.225)\n    \n    test_aug = A.Compose(\n        [\n            A.Normalize(mean, std, max_pixel_value=255.0, always_apply=True,p=1.0)\n        ]\n    )\n    test_images_list = df_test.image_name.values.tolist()\n    test_images = [os.path.join(test_data_path,i + '.png') for i in test_images_list]\n    test_targets = df_test.target.values\n    \n    test_dataset = ClassificationLoader(\n        image_paths = test_images,\n        targets= test_targets,\n        resize = None,\n        augmentations = test_aug\n    )\n    \n    test_loader = torch.utils.data.DataLoader(\n        test_dataset,\n        batch_size = test_bs,\n        shuffle = False,\n        num_workers=4\n    )\n    #Earlier defined class for model\n    model = EfficientNet()\n    model.load_state_dict(torch.load(model_path))\n    model.to(device)\n    \n    predictions_op = Engine.predict(\n        test_loader,\n        model,\n        device\n    )\n    return np.vstack((predictions_op)).ravel()\n","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# prediction\npred = predict(0)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = pred\nsample = pd.read_csv(\"../input/siim-isic-melanoma-classification/sample_submission.csv\")\nsample.loc[:, \"target\"] = predictions\nsample.to_csv(\"submission.csv\", index=False)","metadata":{},"execution_count":null,"outputs":[]}]}