{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# pip install efficientnet","metadata":{"execution":{"iopub.status.busy":"2021-07-16T21:48:03.683319Z","iopub.execute_input":"2021-07-16T21:48:03.683994Z","iopub.status.idle":"2021-07-16T21:48:03.688650Z","shell.execute_reply.started":"2021-07-16T21:48:03.683891Z","shell.execute_reply":"2021-07-16T21:48:03.687602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import efficientnet.keras as efn ","metadata":{"execution":{"iopub.status.busy":"2021-07-16T21:48:03.690050Z","iopub.execute_input":"2021-07-16T21:48:03.690330Z","iopub.status.idle":"2021-07-16T21:48:03.700793Z","shell.execute_reply.started":"2021-07-16T21:48:03.690303Z","shell.execute_reply":"2021-07-16T21:48:03.700134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install efficientnet -q","metadata":{"execution":{"iopub.status.busy":"2021-07-26T19:32:23.840175Z","iopub.execute_input":"2021-07-26T19:32:23.840701Z","iopub.status.idle":"2021-07-26T19:32:34.233937Z","shell.execute_reply.started":"2021-07-26T19:32:23.840593Z","shell.execute_reply":"2021-07-26T19:32:34.232787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport random\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom sklearn.model_selection import GroupKFold\nimport efficientnet.tfkeras as efn\nimport matplotlib.pyplot as plt\nimport tensorflow_addons as tfa\nimport keras\nimport tensorflow.keras.backend as K\nfrom imgaug import augmenters as iaa\nimport imgaug as ia\nfrom tensorflow.keras.applications.vgg16 import preprocess_input, VGG16","metadata":{"execution":{"iopub.status.busy":"2021-07-26T19:32:47.507614Z","iopub.execute_input":"2021-07-26T19:32:47.508004Z","iopub.status.idle":"2021-07-26T19:32:56.466870Z","shell.execute_reply.started":"2021-07-26T19:32:47.507968Z","shell.execute_reply":"2021-07-26T19:32:56.465696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config","metadata":{}},{"cell_type":"code","source":"ep = 38\nseed = 2344\nfolds = 5 \nextra_data = 0 # whether we include extra data","metadata":{"execution":{"iopub.status.busy":"2021-07-26T19:33:01.261052Z","iopub.execute_input":"2021-07-26T19:33:01.261447Z","iopub.status.idle":"2021-07-26T19:33:01.267073Z","shell.execute_reply.started":"2021-07-26T19:33:01.261413Z","shell.execute_reply":"2021-07-26T19:33:01.265260Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if extra_data:\n    extra = pd.read_csv(\"../input/ricord-covid19-xray-positive-tests/MIDRC-RICORD-meta.csv\")\n    extra = extra[[\"fname\", 'labels']]\n    extra = extra.dropna()\n    extra = extra.reset_index(drop = True)\n    df2 = pd.DataFrame({\"id\": extra.fname, \"label\": extra.labels, \"Negative for Pneumonia\":0,'Typical Appearance':0,'Indeterminate Appearance':0,'Atypical Appearance':0})\n    df2.label = (df2.label).apply(lambda x: x.split(\",\"))\n    for index, item in enumerate(df2.label):\n        for i in item:\n            if \"Negative\" in i:\n                df2.loc[index, \"Negative for Pneumonia\"] += 1\n            elif \"Typical\" in i:\n                df2.loc[index, \"Typical Appearance\"] += 1\n            elif \"Indeterminate\" in i:\n                df2.loc[index, \"Indeterminate Appearance\"] += 1\n            elif \"Atypical\" in i:\n                df2.loc[index, \"Atypical Appearance\"] += 1\n    ll = [\"Negative for Pneumonia\",\"Typical Appearance\",\"Indeterminate Appearance\", \"Atypical Appearance\"]\n    for i in list(range(extra.shape[0])):\n        if df2.loc[i,\"Negative for Pneumonia\"] == df2[ll].max(axis=1)[i]:\n            df2.loc[i,\"Negative for Pneumonia\"] = 1\n            df2.loc[i, [\"Typical Appearance\",\"Indeterminate Appearance\", \"Atypical Appearance\"]] = 0\n        elif df2.loc[i,\"Typical Appearance\"] == df2[ll].max(axis=1)[i]:\n            df2.loc[i,\"Typical Appearance\"] = 1\n            df2.loc[i, [\"Negative for Pneumonia\",\"Indeterminate Appearance\", \"Atypical Appearance\"]] = 0\n        elif df2.loc[i,\"Indeterminate Appearance\"] == df2[ll].max(axis=1)[i]:\n            df2.loc[i,\"Indeterminate Appearance\"] = 1\n            df2.loc[i, [\"Negative for Pneumonia\",\"Typical Appearance\", \"Atypical Appearance\"]] = 0\n        elif df2.loc[i,\"Atypical Appearance\"] == df2[ll].max(axis=1)[i]:\n            df2.loc[i,\"Atypical Appearance\"] = 1\n            df2.loc[i, [\"Negative for Pneumonia\",\"Typical Appearance\",\"Indeterminate Appearance\"]] = 0\n    df2 = df2[[\"id\"] + ll]","metadata":{"execution":{"iopub.status.busy":"2021-07-26T02:01:55.005498Z","iopub.execute_input":"2021-07-26T02:01:55.005846Z","iopub.status.idle":"2021-07-26T02:01:55.025867Z","shell.execute_reply.started":"2021-07-26T02:01:55.005819Z","shell.execute_reply":"2021-07-26T02:01:55.024351Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    return strategy\ndef build_decoder(with_labels=True, target_size=(256, 256), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")    \n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.image.resize(img, target_size)\n        return img\n    def decode_with_labels(path, label):\n        return decode(path), label\n    return decode_with_labels if with_labels else decode\n    def decode_with_labels(path, label):\n        return decode(path), label\n    return decode_with_labels if with_labels else decode\n\ndef build_augmenter(with_labels=True):\n    def augment(img):\n#         img = tf.image.random_flip_left_right(img, seed = seed) #  H Flip \n#         img = tf.image.random_flip_up_down(img, seed = seed)\n        img = tf.image.random_brightness(img, 0.1)\n        img = tf.image.random_saturation(img, 0.7, 1.3)\n#         random_number = random.randint(0, 2)\n#         if random_number == 1:    \n#             img = tf.image.adjust_brightness(img, 0.2)\n#         if random_number == 2:\n#             img = tf.image.adjust_brightness(img, -0.2)\n#         img = tf.image.random_contrast(img, 0.8, 1.2)\n        \n    \n# Flips + rotate + zoom + brightness || 36.8\n# H Flip + rotate + zoom + brightness + cutout        \n        return img    \n    \n    def augment_with_labels(img, label):\n        return augment(img), label\n    return augment_with_labels if with_labels else augment\n\n\ndef build_dataset(paths, labels=None, bsize=128, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n#     if augment:\n#         extra_dset = dset.map(augment_fn, num_parallel_calls=AUTO)\n#         dset = dset.concatenate(extra_dset)\n    if augment:\n        dset = dset.map(augment_fn, num_parallel_calls=AUTO)\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)\n    return dset\ndef set_seed(seed = 0):\n    np.random.seed(seed)\n    random_state = np.random.RandomState(seed)\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    tf.random.set_seed(seed)\n    ia.seed(seed)\n    return random_state\nstrategy = auto_select_accelerator()\nrandom_state = set_seed(seed)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T19:33:08.039079Z","iopub.execute_input":"2021-07-26T19:33:08.039432Z","iopub.status.idle":"2021-07-26T19:33:08.069342Z","shell.execute_reply.started":"2021-07-26T19:33:08.039402Z","shell.execute_reply":"2021-07-26T19:33:08.068095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!conda install '/kaggle/input/pydicom-conda-helper/libjpeg-turbo-2.1.0-h7f98852_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/libgcc-ng-9.3.0-h2828fa1_19.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/gdcm-2.8.9-py37h500ead1_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/conda-4.10.1-py37h89c1867_0.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/certifi-2020.12.5-py37h89c1867_1.tar.bz2' -c conda-forge -y\n!conda install '/kaggle/input/pydicom-conda-helper/openssl-1.1.1k-h7f98852_0.tar.bz2' -c conda-forge -y","metadata":{"execution":{"iopub.status.busy":"2021-07-26T19:48:09.934998Z","iopub.execute_input":"2021-07-26T19:48:09.935634Z","iopub.status.idle":"2021-07-26T19:49:28.968385Z","shell.execute_reply.started":"2021-07-26T19:48:09.935545Z","shell.execute_reply":"2021-07-26T19:49:28.967238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom PIL import Image\nimport pandas as pd\nfrom tqdm.auto import tqdm\nimport numpy as np\nimport pydicom\nfrom pydicom.pixel_data_handlers.util import apply_voi_lut","metadata":{"execution":{"iopub.status.busy":"2021-07-26T19:52:48.940991Z","iopub.execute_input":"2021-07-26T19:52:48.941355Z","iopub.status.idle":"2021-07-26T19:52:49.287512Z","shell.execute_reply.started":"2021-07-26T19:52:48.941326Z","shell.execute_reply":"2021-07-26T19:52:49.286739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv('../input/siim-covid19-detection/sample_submission.csv')\nfast_sub = False\ndef read_xray(path, voi_lut = True, fix_monochrome = True):\n    dicom = pydicom.read_file(path)\n    if voi_lut:\n        data = apply_voi_lut(dicom.pixel_array, dicom)\n    else:\n        data = dicom.pixel_array\n    if fix_monochrome and dicom.PhotometricInterpretation == \"MONOCHROME1\":\n        data = np.amax(data) - data   \n    data = data - np.min(data)\n    data = data / np.max(data)\n    data = (data * 255).astype(np.uint8)\n    return data\ndef resize(array, size, keep_ratio=False, resample=Image.LANCZOS):\n    im = Image.fromarray(array)  \n    if keep_ratio:\n        im.thumbnail((size, size), resample)\n    else:\n        im = im.resize((size, size), resample)    \n    return im\nsplit = 'test'    \nsave_dir = f'/kaggle/working/{split}/'    \nos.makedirs(save_dir, exist_ok=True)\nsave_dir = f'/kaggle/working/{split}/study/'\nos.makedirs(save_dir, exist_ok=True)\nif fast_sub: \n    print(\"fast\")\nelse:   \n    for dirname, _, filenames in tqdm(os.walk(f'../input/siim-covid19-detection/{split}')):\n        for file in filenames:\n            # set keep_ratio=True to have original aspect ratio\n            xray = read_xray(os.path.join(dirname, file))\n            im = resize(xray, size = 600)  \n            study = dirname.split('/')[-2] + '_study.png'\n            im.save(os.path.join(save_dir, study))","metadata":{"execution":{"iopub.status.busy":"2021-07-26T20:08:04.123946Z","iopub.execute_input":"2021-07-26T20:08:04.124407Z","iopub.status.idle":"2021-07-26T20:19:17.185756Z","shell.execute_reply.started":"2021-07-26T20:08:04.124369Z","shell.execute_reply":"2021-07-26T20:19:17.184489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COMPETITION_NAME = \"siimcovid19-512-img-png-600-study-png\"\nBATCH_SIZE = strategy.num_replicas_in_sync * 8\nGCS_DS_PATH = KaggleDatasets().get_gcs_path(COMPETITION_NAME)\nif extra_data:\n    GCS_EXTRA_PATH = KaggleDatasets().get_gcs_path('ricord-covid19-xray-positive-tests')\nload_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\ndf = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\nlabel_cols = df.columns[1:5]\ngkf  = GroupKFold(n_splits = folds)\ndf['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(df, groups = df.id.tolist())):\n    df.loc[val_idx, 'fold'] = fold\nif extra_data:\n    df2['fold'] = -1\n    for fold, (train_idx, val_idx) in enumerate(gkf.split(df2, groups = df2.id.tolist())):\n        df2.loc[val_idx, 'fold'] = fold    \nfor i in range(gkf.n_splits):\n    valid_paths = GCS_DS_PATH + '/study/' + df[df['fold'] == i]['id'] + '.png' #\"/train/\"\n    train_paths = GCS_DS_PATH + '/study/' + df[df['fold'] != i]['id'] + '.png' #\"/train/\" \n    valid_labels = df[df['fold'] == i][label_cols].values\n    train_labels = df[df['fold'] != i][label_cols].values\n    tf.cast(train_labels, tf.float32)\n    if extra_data:\n        valid_paths2 = GCS_EXTRA_PATH + '/MIDRC-RICORD/MIDRC-RICORD/' + df2[df2['fold'] == i]['id']\n        train_paths2 = GCS_EXTRA_PATH + '/MIDRC-RICORD/MIDRC-RICORD/' + df2[df2['fold'] != i]['id']\n        valid_labels2 = df2[df2['fold'] == i][label_cols].values\n        train_labels2 = df2[df2['fold'] != i][label_cols].values\n        tf.cast(train_labels2, tf.float32)\n    IMSIZE = (224, 240, 260, 300, 380, 456, 528, 600, 632)\n    IMS = 7\n    decoder = build_decoder(with_labels=True, target_size=(IMSIZE[IMS], IMSIZE[IMS]), ext='png')\n    test_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[IMS], IMSIZE[IMS]),ext='png')\n    train_dataset = build_dataset(\n        train_paths, train_labels, bsize = BATCH_SIZE, decode_fn = decoder, augment = True\n    )\n    valid_dataset = build_dataset(\n        valid_paths, valid_labels, bsize = BATCH_SIZE, decode_fn = decoder,\n        repeat=False, shuffle=False, augment = False\n    )\n    test_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[7], IMSIZE[7]), ext='png')\n    if extra_data:\n        decoder2 = build_decoder(with_labels=True, target_size=(IMSIZE[IMS], IMSIZE[IMS]), ext='jpg')\n        test_decoder2 = build_decoder(with_labels=False, target_size=(IMSIZE[IMS], IMSIZE[IMS]),ext='jpg')\n        train_dataset2 = build_dataset(\n            train_paths2, train_labels2, bsize=BATCH_SIZE, decode_fn=decoder2, augment = True\n        )\n        valid_dataset2 = build_dataset(\n            valid_paths2, valid_labels2, bsize=BATCH_SIZE, decode_fn=decoder2,\n            repeat=False, shuffle=False, augment= False\n        )   \n    #     train_dataset = train_dataset.concatenate(train_dataset2)\n    #     valid_dataset = valid_dataset.concatenate(valid_dataset2)\n        train_dataset = train_dataset.concatenate(train_dataset2)\n        train_dataset = train_dataset.concatenate(valid_dataset2)\n    print(\"data sets have been built\")\n    try:\n        n_labels = train_labels.shape[1]\n    except:\n        n_labels = 1\n    with strategy.scope():\n        base_model = efn.EfficientNetB7(\n                input_shape=(IMSIZE[IMS], IMSIZE[IMS], 3),\n                weights='noisy-student',  #'noisy-student', 'imagenet'\n                include_top = False)\n        #base_model.trainable = False\n        base_model.trainable = True\n        model = tf.keras.Sequential([\n            base_model,\n            tf.keras.layers.GlobalAveragePooling2D(),\n            tf.keras.layers.Dense(n_labels, activation='softmax')\n        ])\n        model.compile(\n            optimizer=tf.keras.optimizers.Adam(),\n            loss = 'categorical_crossentropy',    # 'categorical_crossentropy'\n            metrics = [tf.keras.metrics.AUC(multi_label = True, curve='PR', summation_method = 'majoring')]) # tf.keras.metrics.AUC(multi_label = True), \n    steps_per_epoch = train_paths.shape[0] // BATCH_SIZE + 20\n    cp = tf.keras.callbacks.ModelCheckpoint(f'model{i}.h5', save_best_only = True, monitor='val_auc', mode='max')  #'val_loss' , 'min' monitor=f'val_auc_{i + 1}\n    cp1 = tf.keras.callbacks.ModelCheckpoint(f'model{i}.h5', save_best_only = True, monitor = f'val_auc_{i}', mode='max')\n    early_stopping = tf.keras.callbacks.EarlyStopping(monitor = 'val_loss', patience = 6)\n    lr_reducer = tf.keras.callbacks.ReduceLROnPlateau(\n        monitor = 'val_loss', patience = 3, min_lr = 1e-6, mode = 'min')\n    model.fit(\n        train_dataset, \n        epochs= ep,\n        verbose = 1,\n        callbacks=[cp,  cp1, early_stopping,\n                  # get_lr_callback()],\n                   lr_reducer],\n        steps_per_epoch=steps_per_epoch,\n        validation_data=valid_dataset)\n#     base_model.trainable = True  \n#     fine_tune_at = 90\n#     # Freeze all the layers before the `fine_tune_at` layer\n#     for layer in base_model.layers[:fine_tune_at]:\n#         layer.trainable =  False\n#     print(\"second compile\")\n#     model.compile(\n#     optimizer=tf.keras.optimizers.Adam(),\n#     loss='categorical_crossentropy',\n#     metrics = [tf.keras.metrics.AUC(multi_label = True)])\n#     model.fit(\n#         train_dataset, \n#         epochs= ep,\n#         verbose = 1,\n#         callbacks=[cp,  early_stopping,\n#            get_lr_callback()],\n#             #lr_reducer],\n#         steps_per_epoch=steps_per_epoch,\n#         validation_data=valid_dataset)\nimport time\nprint(\"---------\")\nprint(\"done\")\ntime.sleep(10000)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T02:02:05.736426Z","iopub.execute_input":"2021-07-26T02:02:05.738767Z","iopub.status.idle":"2021-07-26T02:02:07.454505Z","shell.execute_reply.started":"2021-07-26T02:02:05.736941Z","shell.execute_reply":"2021-07-26T02:02:07.451707Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<a href=\"./model2.h5\"> Download File </a>","metadata":{}},{"cell_type":"code","source":"# val_loss: 0.8471 - val_auc_5: 0.8168 -> mAP 0.212\n# val_loss : 0.9244 - val_auc : 0.8046 -> mAP 0.202\n# val_loss: 0.8321 - val_auc_8: 0.8179 -> mAP 0.177\n# val_loss: 0.8683 - val_auc_4: 0.7999 -> mAP 0.203","metadata":{"execution":{"iopub.status.busy":"2021-07-16T23:28:19.521136Z","iopub.status.idle":"2021-07-16T23:28:19.521667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COMPETITION_NAME = \"siimcovid19-512-img-png-600-study-png\"\n\nBATCH_SIZE = strategy.num_replicas_in_sync * 16\n\nGCS_DS_PATH = KaggleDatasets().get_gcs_path(COMPETITION_NAME)\nif extra_data:\n    GCS_EXTRA_PATH = KaggleDatasets().get_gcs_path('ricord-covid19-xray-positive-tests')\nload_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\ndf = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\nlabel_cols = df.columns[1:5]\ngkf  = GroupKFold(n_splits = folds)\ndf['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(df, groups = df.id.tolist())):\n    df.loc[val_idx, 'fold'] = fold\nif extra_data:\n    df2['fold'] = -1\n    for fold, (train_idx, val_idx) in enumerate(gkf.split(df2, groups = df2.id.tolist())):\n        df2.loc[val_idx, 'fold'] = fold    \n\nfor i in range(gkf.n_splits):\n    valid_paths = GCS_DS_PATH + '/study/' + df[df['fold'] == i]['id'] + '.png' #\"/train/\"\n    train_paths = GCS_DS_PATH + '/study/' + df[df['fold'] != i]['id'] + '.png' #\"/train/\" \n    valid_labels = df[df['fold'] == i][label_cols].values\n    train_labels = df[df['fold'] != i][label_cols].values\n    tf.cast(train_labels, tf.float32)\n    if extra_data:\n        valid_paths2 = GCS_EXTRA_PATH + '/MIDRC-RICORD/MIDRC-RICORD/' + df2[df2['fold'] == i]['id']\n        train_paths2 = GCS_EXTRA_PATH + '/MIDRC-RICORD/MIDRC-RICORD/' + df2[df2['fold'] != i]['id']\n        valid_labels2 = df2[df2['fold'] == i][label_cols].values\n        train_labels2 = df2[df2['fold'] != i][label_cols].values\n        tf.cast(train_labels2, tf.float32)\n    IMSIZE = (224, 240, 260, 300, 380, 456, 528, 600, 632)\n    IMS = 7\n    decoder = build_decoder(with_labels=True, target_size=(IMSIZE[IMS], IMSIZE[IMS]), ext='png')\n    test_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[IMS], IMSIZE[IMS]),ext='png')\n    train_dataset = build_dataset(\n        train_paths, train_labels, bsize = BATCH_SIZE, decode_fn = decoder, augment = True\n    )\n    valid_dataset = build_dataset(\n        valid_paths, valid_labels, bsize = BATCH_SIZE, decode_fn = decoder,\n        repeat=False, shuffle=False, augment = False\n    )\n    \n    test_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[7], IMSIZE[7]), ext='png')\n    if extra_data:\n        decoder2 = build_decoder(with_labels=True, target_size=(IMSIZE[IMS], IMSIZE[IMS]), ext='jpg')\n        test_decoder2 = build_decoder(with_labels=False, target_size=(IMSIZE[IMS], IMSIZE[IMS]),ext='jpg')\n        train_dataset2 = build_dataset(\n            train_paths2, train_labels2, bsize=BATCH_SIZE, decode_fn=decoder2, augment = True\n        )\n        valid_dataset2 = build_dataset(\n            valid_paths2, valid_labels2, bsize=BATCH_SIZE, decode_fn=decoder2,\n            repeat=False, shuffle=False, augment= False\n        )   \n    #     train_dataset = train_dataset.concatenate(train_dataset2)\n    #     valid_dataset = valid_dataset.concatenate(valid_dataset2)\n        train_dataset = train_dataset.concatenate(train_dataset2)\n        train_dataset = train_dataset.concatenate(valid_dataset2)\n\n    print(\"data sets have been built\")\n    try:\n        n_labels = train_labels.shape[1]\n    except:\n        n_labels = 1\n    with strategy.scope():\n        base_model = efn.EfficientNetB7(\n                input_shape=(IMSIZE[IMS], IMSIZE[IMS], 3),\n                weights='noisy-student',  #'noisy-student', 'imagenet'\n                include_top = False)\n        base_model.trainable = False\n        model = tf.keras.Sequential([\n            base_model,\n            tf.keras.layers.GlobalAveragePooling2D(),\n            tf.keras.layers.Dense(n_labels, activation='softmax')\n        ])\n        model.compile(\n            optimizer=tf.keras.optimizers.Adam(),\n            loss='categorical_crossentropy')\n    steps_per_epoch = train_paths.shape[0] // BATCH_SIZE + 10\n    cp = tf.keras.callbacks.ModelCheckpoint(f'model{i}.h5', save_best_only=True, monitor='val_loss', mode='min')  #'val_loss' , 'min'\n    lr_reducer = tf.keras.callbacks.ReduceLROnPlateau(\n        monitor = \"val_loss\", patience = 2, min_lr = 1e-6, mode = 'min')\n    model.fit(\n        train_dataset, \n        epochs= 10,\n        verbose = 1,\n        callbacks=[cp,  \n                   #get_lr_callback(BATCH_SIZE)],\n                    lr_reducer],\n        steps_per_epoch=steps_per_epoch,\n        validation_data=valid_dataset)\n    base_model.trainable = True  \n    fine_tune_at = 90\n    # Freeze all the layers before the `fine_tune_at` layer\n    for layer in base_model.layers[:fine_tune_at]:\n        layer.trainable =  False\n    print(\"second compile\")\n    model.compile(\n            optimizer=tf.keras.optimizers.Adam(),\n            loss = 'categorical_crossentropy',    # 'categorical_crossentropy'\n            metrics = [tf.keras.metrics.AUC(multi_label = True, curve='PR', summation_method = 'majoring')])\n    steps_per_epoch = train_paths.shape[0] // BATCH_SIZE + 20\n    cp = tf.keras.callbacks.ModelCheckpoint(f'model{i}.h5', save_best_only = True, monitor='val_auc', mode='max')  #'val_loss' , 'min' monitor=f'val_auc_{i + 1}\n    cp1 = tf.keras.callbacks.ModelCheckpoint(f'model{i}.h5', save_best_only = True, monitor = f'val_auc_{i}', mode='max')\n    early_stopping = tf.keras.callbacks.EarlyStopping(monitor = 'val_loss', patience = 6)\n    lr_reducer = tf.keras.callbacks.ReduceLROnPlateau(\n        monitor = 'val_loss', patience = 2, min_lr = 1e-6, mode = 'min')\n    model.fit(\n        train_dataset, \n        epochs= ep,\n        verbose = 1,\n        callbacks=[cp,  cp1, early_stopping,\n                  # get_lr_callback()],\n                   lr_reducer],\n        steps_per_epoch=steps_per_epoch,\n        validation_data=valid_dataset)    \nimport time\nprint(\"---------\")\nprint(\"done\")\ntime.sleep(10000)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T02:02:16.922541Z","iopub.execute_input":"2021-07-26T02:02:16.922949Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}