{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install efficientnet -q","metadata":{"execution":{"iopub.status.busy":"2021-07-08T09:37:20.297676Z","iopub.execute_input":"2021-07-08T09:37:20.298016Z","iopub.status.idle":"2021-07-08T09:37:29.493335Z","shell.execute_reply.started":"2021-07-08T09:37:20.297987Z","shell.execute_reply":"2021-07-08T09:37:29.492267Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport random\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom sklearn.model_selection import GroupKFold\nimport efficientnet.tfkeras as efn\nimport matplotlib.pyplot as plt\nimport tensorflow_addons as tfa\nimport keras\nimport tensorflow.keras.backend as K\nfrom imgaug import augmenters as iaa\nimport imgaug as ia","metadata":{"execution":{"iopub.status.busy":"2021-07-08T23:07:31.068134Z","iopub.execute_input":"2021-07-08T23:07:31.069211Z","iopub.status.idle":"2021-07-08T23:07:38.266479Z","shell.execute_reply.started":"2021-07-08T23:07:31.069035Z","shell.execute_reply":"2021-07-08T23:07:38.26472Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Config","metadata":{}},{"cell_type":"code","source":"ep = 38\nseed = 26\nfolds = 5 ","metadata":{"execution":{"iopub.status.busy":"2021-07-08T09:37:56.23768Z","iopub.execute_input":"2021-07-08T09:37:56.238188Z","iopub.status.idle":"2021-07-08T09:37:56.242617Z","shell.execute_reply.started":"2021-07-08T09:37:56.238138Z","shell.execute_reply":"2021-07-08T09:37:56.241878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    return strategy\ndef build_decoder(with_labels=True, target_size=(256, 256), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")    \n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.image.resize(img, target_size)\n        return img\n    def decode_with_labels(path, label):\n        return decode(path), label\n    return decode_with_labels if with_labels else decode\n    def decode_with_labels(path, label):\n        return decode(path), label\n    return decode_with_labels if with_labels else decode\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img) #  H Flip \n        img = tf.image.random_flip_up_down(img)\n#         img = tf.image.random_saturation(img, 0.7, 1.3)\n#         img = tf.image.random_saturation(img, 0.7, 1.3)\n#         img = tf.image.random_contrast(img, 0.8, 1.2)      \n#         rand_rot = np.random.randn() * 45\n#         img = tfa.image.rotate(img, rand_rot)\n#         img = tf.image.random_saturation(img, 0.7, 1.3)\n\n#         img = tf.image.random_brightness(img, 0.1)\n#         img = tf.image.random_saturation(img, 0.7, 1.3)\n\n#         img = tf.image.random_contrast(img, 0.8, 1.2)\n        \n    \n# Flips + rotate + zoom + brightness || 36.8\n# H Flip + rotate + zoom + brightness + cutout        \n        return img    \n    \n    def augment_with_labels(img, label):\n        return augment(img), label\n    return augment_with_labels if with_labels else augment\ndef build_dataset(paths, labels=None, bsize=128, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    if augment:\n        dset = dset.map(augment_fn, num_parallel_calls=AUTO)\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)\n    return dset\ndef set_seed(seed = 0):\n    np.random.seed(seed)\n    random_state = np.random.RandomState(seed)\n    random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    tf.random.set_seed(seed)\n    ia.seed(seed)\n    return random_state\nstrategy = auto_select_accelerator()\nrandom_state = set_seed(seed)","metadata":{"execution":{"iopub.status.busy":"2021-07-08T09:37:36.607273Z","iopub.execute_input":"2021-07-08T09:37:36.607775Z","iopub.status.idle":"2021-07-08T09:37:42.726077Z","shell.execute_reply.started":"2021-07-08T09:37:36.607727Z","shell.execute_reply":"2021-07-08T09:37:42.725348Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COMPETITION_NAME = \"siimcovid19-512-img-png-600-study-png\"\n\nBATCH_SIZE = 100 #strategy.num_replicas_in_sync * 16 \n\nGCS_DS_PATH = KaggleDatasets().get_gcs_path(COMPETITION_NAME)\nif extra_data:\n    GCS_EXTRA_PATH = KaggleDatasets().get_gcs_path('ricord-covid19-xray-positive-tests')\nload_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\ndf = pd.read_csv('../input/siim-covid19-detection/train_study_level.csv')\nlabel_cols = df.columns[1:5]\ngkf  = GroupKFold(n_splits = folds)\ndf['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(df, groups = df.id.tolist())):\n    df.loc[val_idx, 'fold'] = fold\nif extra_data:\n    df2['fold'] = -1\n    for fold, (train_idx, val_idx) in enumerate(gkf.split(df2, groups = df2.id.tolist())):\n        df2.loc[val_idx, 'fold'] = fold    \n\nfor i in range(gkf.n_splits):\n    valid_paths = GCS_DS_PATH + '/study/' + df[df['fold'] == i]['id'] + '.png' #\"/train/\"\n    train_paths = GCS_DS_PATH + '/study/' + df[df['fold'] != i]['id'] + '.png' #\"/train/\" \n    valid_labels = df[df['fold'] == i][label_cols].values\n    train_labels = df[df['fold'] != i][label_cols].values\n    tf.cast(train_labels, tf.float32)\n    IMSIZE = (224, 240, 260, 300, 380, 456, 528, 600, 632)\n    IMS = 7\n    decoder = build_decoder(with_labels=True, target_size=(IMSIZE[IMS], IMSIZE[IMS]), ext='png')\n    test_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[IMS], IMSIZE[IMS]),ext='png')\n    train_dataset = build_dataset(\n        train_paths, train_labels, bsize = BATCH_SIZE, decode_fn = decoder, augment = True\n    )\n    valid_dataset = build_dataset(\n        valid_paths, valid_labels, bsize = BATCH_SIZE, decode_fn = decoder,\n        repeat=False, shuffle=False, augment = False\n    )\n    \n    test_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[7], IMSIZE[7]), ext='png')\n\n    print(\"data sets have been built\")\n    try:\n        n_labels = train_labels.shape[1]\n    except:\n        n_labels = 1\n    with strategy.scope():\n        base_model = efn.EfficientNetB7(\n                input_shape=(IMSIZE[IMS], IMSIZE[IMS], 3),\n                weights='noisy-student',  #'noisy-student', 'imagenet'\n                include_top = False)\n        print(\"Number of layers in the base model: \", len(base_model.layers))\n        base_model.trainable = False\n        model = tf.keras.Sequential([\n            base_model,\n            tf.keras.layers.GlobalAveragePooling2D(),\n            tf.keras.layers.Dense(n_labels, activation='softmax')\n        ])\n        model.compile(\n            optimizer=tf.keras.optimizers.Adam(),\n            loss='categorical_crossentropy',\n            metrics = [tf.keras.metrics.AUC(multi_label = True)])\n    steps_per_epoch = train_paths.shape[0] // BATCH_SIZE\n    cp = tf.keras.callbacks.ModelCheckpoint(f'model{i}.h5', save_best_only=True, monitor='val_loss', mode='min')  #'val_loss' , 'min'\n    lr_reducer = tf.keras.callbacks.ReduceLROnPlateau(\n        monitor = \"val_loss\", patience = 3, min_lr = 1e-6, mode = 'min')\n    model.fit(\n        train_dataset, \n        epochs= ep,\n        verbose = 1,\n        callbacks=[cp,  \n                   #get_lr_callback(BATCH_SIZE)],\n                    lr_reducer],\n        steps_per_epoch=steps_per_epoch,\n        validation_data=valid_dataset)\n    base_model.trainable = True  \n    fine_tune_at = 90\n    # Freeze all the layers before the `fine_tune_at` layer\n    for layer in base_model.layers[:fine_tune_at]:\n        layer.trainable =  False\n    print(\"second compile\")\n    model.compile(\n    optimizer=tf.keras.optimizers.Adam(),\n    loss='categorical_crossentropy',\n    metrics = [tf.keras.metrics.AUC(multi_label = True)])\n    model.fit(\n        train_dataset, \n        epochs= ep,\n        verbose = 1,\n        callbacks=[cp, lr_reducer],\n        steps_per_epoch=steps_per_epoch,\n        validation_data=valid_dataset)\nimport time\nprint(\"---------\")\nprint(\"done\")\ntime.sleep(10000)","metadata":{"execution":{"iopub.status.busy":"2021-07-08T09:37:59.834954Z","iopub.execute_input":"2021-07-08T09:37:59.835647Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}