{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install efficientnet -q","metadata":{"execution":{"iopub.execute_input":"2021-04-08T13:39:37.159015Z","iopub.status.busy":"2021-04-08T13:39:37.158365Z","iopub.status.idle":"2021-04-08T13:39:46.771507Z","shell.execute_reply":"2021-04-08T13:39:46.77039Z"},"papermill":{"duration":9.636471,"end_time":"2021-04-08T13:39:46.771719","exception":false,"start_time":"2021-04-08T13:39:37.135248","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nimport efficientnet.tfkeras as efn\nimport numpy as np\nimport pandas as pd\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom sklearn.model_selection import GroupKFold\nfrom matplotlib import pyplot as plt\n\nprint(\"TF Version:\", tf.__version__)","metadata":{"execution":{"iopub.execute_input":"2021-04-08T13:39:46.808415Z","iopub.status.busy":"2021-04-08T13:39:46.807206Z","iopub.status.idle":"2021-04-08T13:39:54.87598Z","shell.execute_reply":"2021-04-08T13:39:54.874926Z"},"papermill":{"duration":8.089979,"end_time":"2021-04-08T13:39:54.876206","exception":false,"start_time":"2021-04-08T13:39:46.786227","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"EPOCHS = 20","metadata":{"execution":{"iopub.execute_input":"2021-04-08T13:39:54.908417Z","iopub.status.busy":"2021-04-08T13:39:54.907662Z","iopub.status.idle":"2021-04-08T13:39:54.911988Z","shell.execute_reply":"2021-04-08T13:39:54.912959Z"},"papermill":{"duration":0.02248,"end_time":"2021-04-08T13:39:54.91333","exception":false,"start_time":"2021-04-08T13:39:54.89085","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    \n    return strategy\n\n\ndef build_decoder(with_labels=True, target_size=(256, 256), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.image.resize(img, target_size)\n        print(target_size)\n        return img\n    \n    def decode_with_labels(path, label):\n        return decode(path), label\n    \n    return decode_with_labels if with_labels else decode\n\n\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        return img\n    \n    def augment_with_labels(img, label):\n        return augment(img), label\n    \n    return augment_with_labels if with_labels else augment\n\n\ndef build_dataset(paths, labels=None, bsize=128, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    \n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n    \n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n    \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    \n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)\n    \n    return dset","metadata":{"execution":{"iopub.execute_input":"2021-04-08T13:39:54.946998Z","iopub.status.busy":"2021-04-08T13:39:54.946262Z","iopub.status.idle":"2021-04-08T13:39:54.966309Z","shell.execute_reply":"2021-04-08T13:39:54.966927Z"},"papermill":{"duration":0.039342,"end_time":"2021-04-08T13:39:54.967202","exception":false,"start_time":"2021-04-08T13:39:54.92786","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#../input/hpa-single-cell-image-classification\nCOMPETITION_NAME = \"hpa-single-cell-image-classification\" #\"hpa-768768\"\nGCS_DS_PATH = KaggleDatasets().get_gcs_path(COMPETITION_NAME)\nGCS_DS_PATH","metadata":{"_kg_hide-output":true,"execution":{"iopub.execute_input":"2021-04-08T13:39:55.014252Z","iopub.status.busy":"2021-04-08T13:39:55.013435Z","iopub.status.idle":"2021-04-08T13:39:55.403326Z","shell.execute_reply":"2021-04-08T13:39:55.402721Z"},"papermill":{"duration":0.421579,"end_time":"2021-04-08T13:39:55.403496","exception":false,"start_time":"2021-04-08T13:39:54.981917","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"strategy = auto_select_accelerator()\nBATCH_SIZE = strategy.num_replicas_in_sync * 16","metadata":{"execution":{"iopub.execute_input":"2021-04-08T13:39:55.505921Z","iopub.status.busy":"2021-04-08T13:39:55.451112Z","iopub.status.idle":"2021-04-08T13:40:01.321948Z","shell.execute_reply":"2021-04-08T13:40:01.322506Z"},"papermill":{"duration":5.905401,"end_time":"2021-04-08T13:40:01.322886","exception":false,"start_time":"2021-04-08T13:39:55.417485","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#green\n#../input/hpa-2020-16bit-training-set\nload_dir = f\"/kaggle/input/{COMPETITION_NAME}/train/\"\ndf = pd.read_csv('../input/classification-label-csv-green/df_green.csv')\nlabel_cols = df.columns[2:21]\npaths = GCS_DS_PATH + '/train/' + df['ID'] + '.png'\nlabels = df[label_cols].values","metadata":{"execution":{"iopub.execute_input":"2021-04-08T13:40:01.365637Z","iopub.status.busy":"2021-04-08T13:40:01.364875Z","iopub.status.idle":"2021-04-08T13:40:01.548133Z","shell.execute_reply":"2021-04-08T13:40:01.548742Z"},"papermill":{"duration":0.208459,"end_time":"2021-04-08T13:40:01.548952","exception":false,"start_time":"2021-04-08T13:40:01.340493","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"(\n    train_paths, valid_paths, \n    train_labels, valid_labels\n) = train_test_split(paths, labels, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.execute_input":"2021-04-08T13:40:01.593255Z","iopub.status.busy":"2021-04-08T13:40:01.592224Z","iopub.status.idle":"2021-04-08T13:40:01.606175Z","shell.execute_reply":"2021-04-08T13:40:01.605314Z"},"papermill":{"duration":0.039107,"end_time":"2021-04-08T13:40:01.606351","exception":false,"start_time":"2021-04-08T13:40:01.567244","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMSIZE = (224, 240, 260, 300, 380, 456, 528, 600)\nIMS = 7\n\ndecoder = build_decoder(with_labels=True, target_size=(IMSIZE[IMS], IMSIZE[IMS]))\ntest_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[IMS], IMSIZE[IMS]))\n\ntrain_dataset = build_dataset(\n    train_paths, train_labels, bsize=BATCH_SIZE, decode_fn=decoder,augment=True\n)\n\nvalid_dataset = build_dataset(\n    valid_paths, valid_labels, bsize=BATCH_SIZE, decode_fn=decoder,\n    repeat=False, shuffle=False, augment=False\n)\n\n# test_dataset = build_dataset(\n#     test_paths, cache=False, bsize=BATCH_SIZE, decode_fn=test_decoder,\n#     repeat=False, shuffle=False, augment=False\n# )","metadata":{"execution":{"iopub.execute_input":"2021-04-08T13:40:01.67789Z","iopub.status.busy":"2021-04-08T13:40:01.667934Z","iopub.status.idle":"2021-04-08T13:40:02.021912Z","shell.execute_reply":"2021-04-08T13:40:02.020234Z"},"papermill":{"duration":0.400562,"end_time":"2021-04-08T13:40:02.022245","exception":false,"start_time":"2021-04-08T13:40:01.621683","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    n_labels = train_labels.shape[1]\nexcept:\n    n_labels = 1\n    \nwith strategy.scope():\n\n    model = tf.keras.Sequential(\n    [\n        efn.EfficientNetB7(\n            input_shape=(IMSIZE[IMS], IMSIZE[IMS], 3),\n            weights='noisy-student',\n            include_top=False),\n        \n        tf.keras.layers.GlobalAveragePooling2D(),\n        \n        tf.keras.layers.Dropout(0.3),\n        \n        tf.keras.layers.Dense(\n            n_labels, \n            activation='sigmoid',\n            name='output'\n        )\n    ])\n    METRICS = [ \n          tf.keras.metrics.Precision(name='precision'),\n          tf.keras.metrics.AUC(multi_label=True, name='auc'),\n    ]\n    model.compile(\n        optimizer=tf.keras.optimizers.Adamax(),\n        loss='binary_crossentropy',\n        metrics=METRICS)\n        \n    model.summary()","metadata":{"execution":{"iopub.execute_input":"2021-04-08T13:40:02.069263Z","iopub.status.busy":"2021-04-08T13:40:02.068105Z","iopub.status.idle":"2021-04-08T13:40:49.708391Z","shell.execute_reply":"2021-04-08T13:40:49.706701Z"},"papermill":{"duration":47.668434,"end_time":"2021-04-08T13:40:49.708642","exception":false,"start_time":"2021-04-08T13:40:02.040208","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Learning Rate Schedule\nWe'll train this network with a special learning rate schedule","metadata":{"papermill":{"duration":0.031558,"end_time":"2021-04-08T13:40:49.773673","exception":false,"start_time":"2021-04-08T13:40:49.742115","status":"completed"},"tags":[]}},{"cell_type":"code","source":"colour = '_green'","metadata":{"execution":{"iopub.execute_input":"2021-04-08T13:40:49.843365Z","iopub.status.busy":"2021-04-08T13:40:49.842531Z","iopub.status.idle":"2021-04-08T13:40:49.845924Z","shell.execute_reply":"2021-04-08T13:40:49.845283Z"},"papermill":{"duration":0.040506,"end_time":"2021-04-08T13:40:49.846182","exception":false,"start_time":"2021-04-08T13:40:49.805676","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"steps_per_epoch = train_paths.shape[0] // BATCH_SIZE\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(\n    f'model{colour}.h5', save_best_only=True, monitor='val_loss', mode='min')\nlr_reducer = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor=\"val_loss\", patience=3, min_lr=1e-6, mode='min')","metadata":{"execution":{"iopub.execute_input":"2021-04-08T13:40:49.918855Z","iopub.status.busy":"2021-04-08T13:40:49.918147Z","iopub.status.idle":"2021-04-08T13:40:49.922084Z","shell.execute_reply":"2021-04-08T13:40:49.921508Z"},"papermill":{"duration":0.043026,"end_time":"2021-04-08T13:40:49.922236","exception":false,"start_time":"2021-04-08T13:40:49.87921","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    train_dataset, \n    epochs=EPOCHS,\n    verbose=1,\n    callbacks=[checkpoint, lr_reducer],\n    steps_per_epoch=steps_per_epoch,\n    validation_data=valid_dataset)","metadata":{"execution":{"iopub.execute_input":"2021-04-08T13:40:49.998979Z","iopub.status.busy":"2021-04-08T13:40:49.994716Z","iopub.status.idle":"2021-04-08T15:09:15.382808Z","shell.execute_reply":"2021-04-08T15:09:15.383393Z"},"papermill":{"duration":5305.429931,"end_time":"2021-04-08T15:09:15.383813","exception":false,"start_time":"2021-04-08T13:40:49.953882","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"hist_df = pd.DataFrame(history.history)\nhist_df.to_csv(f'history{colour}.csv')","metadata":{"execution":{"iopub.execute_input":"2021-04-08T15:09:17.186645Z","iopub.status.busy":"2021-04-08T15:09:17.182357Z","iopub.status.idle":"2021-04-08T15:09:17.20117Z","shell.execute_reply":"2021-04-08T15:09:17.200368Z"},"papermill":{"duration":0.929946,"end_time":"2021-04-08T15:09:17.201342","exception":false,"start_time":"2021-04-08T15:09:16.271396","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(13, 5))\nplt.plot(history.history['loss'])\nplt.plot(history.history['val_loss'])\nplt.title(\"Model Loss\")\nplt.xlabel('Epochs')\nplt.ylabel('Loss')\nplt.legend(['Train', 'Test'])\nplt.ylim(ymax = 2, ymin = 0)\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-04-08T15:09:19.068771Z","iopub.status.busy":"2021-04-08T15:09:19.039526Z","iopub.status.idle":"2021-04-08T15:09:19.347948Z","shell.execute_reply":"2021-04-08T15:09:19.347409Z"},"papermill":{"duration":1.263055,"end_time":"2021-04-08T15:09:19.348145","exception":false,"start_time":"2021-04-08T15:09:18.08509","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.figure(figsize=(13, 5))\nplt.plot(history.history['precision'])\nplt.plot(history.history['val_precision'])\nplt.title('Model Precision')\nplt.xlabel('Epochs')\nplt.ylabel('Precision')\nplt.legend(['Train','Test'])\nplt.grid()\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2021-04-08T15:09:21.191183Z","iopub.status.busy":"2021-04-08T15:09:21.183116Z","iopub.status.idle":"2021-04-08T15:09:21.372965Z","shell.execute_reply":"2021-04-08T15:09:21.372295Z"},"papermill":{"duration":1.130633,"end_time":"2021-04-08T15:09:21.373143","exception":false,"start_time":"2021-04-08T15:09:20.24251","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}