{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install efficientnet -q","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:04:07.233119Z","iopub.execute_input":"2022-04-30T07:04:07.233965Z","iopub.status.idle":"2022-04-30T07:04:17.998096Z","shell.execute_reply.started":"2022-04-30T07:04:07.233840Z","shell.execute_reply":"2022-04-30T07:04:17.996663Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport efficientnet.tfkeras as efn\nimport numpy as np\nimport pandas as pd\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf\nfrom sklearn.model_selection import GroupKFold","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:04:19.605690Z","iopub.execute_input":"2022-04-30T07:04:19.607074Z","iopub.status.idle":"2022-04-30T07:04:26.995724Z","shell.execute_reply.started":"2022-04-30T07:04:19.607006Z","shell.execute_reply":"2022-04-30T07:04:26.994848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    \n    return strategy","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:04:26.997374Z","iopub.execute_input":"2022-04-30T07:04:26.998236Z","iopub.status.idle":"2022-04-30T07:04:27.006025Z","shell.execute_reply.started":"2022-04-30T07:04:26.998185Z","shell.execute_reply":"2022-04-30T07:04:27.004735Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:32:32.312225Z","iopub.execute_input":"2022-04-30T07:32:32.312589Z","iopub.status.idle":"2022-04-30T07:32:32.347045Z","shell.execute_reply.started":"2022-04-30T07:32:32.312557Z","shell.execute_reply":"2022-04-30T07:32:32.345828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef build_decoder(with_labels=True, target_size=(256, 256), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n        img = tf.cast(img, tf.float32) / 255.0   \n        img = tf.image.resize(img, target_size)  \n\n        return img\n    \n    def decode_with_labels(path, label):\n        return decode(path), label\n    \n    return decode_with_labels if with_labels else decode","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:04:47.153245Z","iopub.execute_input":"2022-04-30T07:04:47.153909Z","iopub.status.idle":"2022-04-30T07:04:47.166298Z","shell.execute_reply.started":"2022-04-30T07:04:47.153847Z","shell.execute_reply":"2022-04-30T07:04:47.165016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        return img\n    \n    def augment_with_labels(img, label):\n        return augment(img), label\n    \n    return augment_with_labels if with_labels else augment\n","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:04:51.868593Z","iopub.execute_input":"2022-04-30T07:04:51.868916Z","iopub.status.idle":"2022-04-30T07:04:51.875536Z","shell.execute_reply.started":"2022-04-30T07:04:51.868886Z","shell.execute_reply":"2022-04-30T07:04:51.874608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_dataset(paths, labels=None, bsize=128, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)    \n    \n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)      \n    \n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n    \n    AUTO = tf.data.experimental.AUTOTUNE       \n    slices = paths if labels is None else (paths, labels)  \n    \n    dset = tf.data.Dataset.from_tensor_slices(slices)   \n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)   \n    dset = dset.cache(cache_dir) if cache else dset       \n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset              \n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)               \n    \n    return dset","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:05:01.952373Z","iopub.execute_input":"2022-04-30T07:05:01.953603Z","iopub.status.idle":"2022-04-30T07:05:01.965002Z","shell.execute_reply.started":"2022-04-30T07:05:01.953526Z","shell.execute_reply":"2022-04-30T07:05:01.963774Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"COMPETITION_NAME = \"deepweeds\"\nstrategy = auto_select_accelerator()\nBATCH_SIZE = strategy.num_replicas_in_sync * 16\nGCS_DS_PATH = KaggleDatasets().get_gcs_path(COMPETITION_NAME)","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:06:32.983707Z","iopub.execute_input":"2022-04-30T07:06:32.984756Z","iopub.status.idle":"2022-04-30T07:08:36.055399Z","shell.execute_reply.started":"2022-04-30T07:06:32.984721Z","shell.execute_reply":"2022-04-30T07:08:36.054240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\ndf = pd.read_csv('../input/deepweeds/labels/labels.csv')\n#df['opacity_label'] = np.where(df['boxes'].isnull(),0,1)\nlabel_cols = df.columns[1]","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:10:33.745353Z","iopub.execute_input":"2022-04-30T07:10:33.745745Z","iopub.status.idle":"2022-04-30T07:10:33.775055Z","shell.execute_reply.started":"2022-04-30T07:10:33.745710Z","shell.execute_reply":"2022-04-30T07:10:33.774372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"label_cols","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:08:50.614698Z","iopub.execute_input":"2022-04-30T07:08:50.615068Z","iopub.status.idle":"2022-04-30T07:08:50.625297Z","shell.execute_reply.started":"2022-04-30T07:08:50.615028Z","shell.execute_reply":"2022-04-30T07:08:50.624057Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:11:50.644817Z","iopub.execute_input":"2022-04-30T07:11:50.645168Z","iopub.status.idle":"2022-04-30T07:11:50.673942Z","shell.execute_reply.started":"2022-04-30T07:11:50.645134Z","shell.execute_reply":"2022-04-30T07:11:50.672838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"gkf  = GroupKFold(n_splits = 5)\ndf['fold'] = -1\nfor fold, (train_idx, val_idx) in enumerate(gkf.split(df, groups = df.Label.tolist())):\n    df.loc[val_idx, 'fold'] = fold","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:12:20.233799Z","iopub.execute_input":"2022-04-30T07:12:20.234441Z","iopub.status.idle":"2022-04-30T07:12:20.255260Z","shell.execute_reply.started":"2022-04-30T07:12:20.234403Z","shell.execute_reply":"2022-04-30T07:12:20.254140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(5):\n    \n    valid_paths = GCS_DS_PATH + '/train/' + df[df['fold'] == i]['Filename'] + '.jpg' #\"/train/\"\n    train_paths = GCS_DS_PATH + '/train/' + df[df['fold'] != i]['Filename'] + '.jpg' #\"/train/\" \n    valid_labels = df[df['fold'] == i][label_cols].values\n    train_labels = df[df['fold'] != i][label_cols].values\n    \n    IMSIZE = (224, 240, 260, 300, 380, 456, 528, 600, 512)\n    IMS = 3\n    \n    decoder = build_decoder(with_labels=True, target_size=(IMSIZE[IMS], IMSIZE[IMS]), ext='jpg')\n    test_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[IMS], IMSIZE[IMS]),ext='jpg')\n    \n    train_dataset = build_dataset(\n        train_paths, train_labels, bsize=BATCH_SIZE, decode_fn=decoder\n    )\n\n    valid_dataset = build_dataset(\n        valid_paths, valid_labels, bsize=BATCH_SIZE, decode_fn=decoder,\n        repeat=False, shuffle=False, augment=False\n    )\n    \n","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:28:19.560628Z","iopub.execute_input":"2022-04-30T07:28:19.560955Z","iopub.status.idle":"2022-04-30T07:28:20.032650Z","shell.execute_reply.started":"2022-04-30T07:28:19.560921Z","shell.execute_reply":"2022-04-30T07:28:20.031200Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:34:57.954166Z","iopub.execute_input":"2022-04-30T07:34:57.955308Z","iopub.status.idle":"2022-04-30T07:34:57.962943Z","shell.execute_reply.started":"2022-04-30T07:34:57.955239Z","shell.execute_reply":"2022-04-30T07:34:57.962215Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    try:\n        n_labels = train_labels.shape[1]\n    except:\n        n_labels = 8\n    \n    with strategy.scope():\n        model = tf.keras.Sequential([\n            efn.EfficientNetB7(\n                input_shape=(IMSIZE[IMS], IMSIZE[IMS], 3),\n                weights='imagenet',\n                include_top=False), \n            tf.keras.layers.GlobalAveragePooling2D(),\n            tf.keras.layers.Dense(n_labels, activation='sigmoid')\n        ])\n        model.compile(\n            optimizer=tf.keras.optimizers.Adam(),\n            loss='binary_crossentropy',\n            metrics=[tf.keras.metrics.AUC(multi_label=True)])\n\n        model.summary()\n        \n    steps_per_epoch = train_paths.shape[0] // BATCH_SIZE\n    checkpoint = tf.keras.callbacks.ModelCheckpoint(\n        f'model{i}.h5', save_best_only=True, monitor='val_loss', mode='min')\n    lr_reducer = tf.keras.callbacks.ReduceLROnPlateau(\n        monitor=\"val_loss\", patience=3, min_lr=1e-6, mode='min')\n    \n    history = model.fit(\n        train_dataset, \n        epochs=20,\n        verbose=1,\n        callbacks=[checkpoint, lr_reducer],\n        steps_per_epoch=steps_per_epoch,\n        validation_data=valid_dataset)\n\n    hist_df = pd.DataFrame(history.history)\n    hist_df.to_csv(f'history{i}.csv')","metadata":{"execution":{"iopub.status.busy":"2022-04-30T07:28:40.769330Z","iopub.execute_input":"2022-04-30T07:28:40.769669Z","iopub.status.idle":"2022-04-30T07:29:36.963720Z","shell.execute_reply.started":"2022-04-30T07:28:40.769639Z","shell.execute_reply":"2022-04-30T07:29:36.962295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}