{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install /kaggle/input/kerasapplications -q","metadata":{"execution":{"iopub.status.busy":"2022-03-20T14:11:31.299456Z","iopub.execute_input":"2022-03-20T14:11:31.299794Z","iopub.status.idle":"2022-03-20T14:12:00.435917Z","shell.execute_reply.started":"2022-03-20T14:11:31.299761Z","shell.execute_reply":"2022-03-20T14:12:00.434805Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%env SM_FRAMEWORK=tf.keras","metadata":{"execution":{"iopub.status.busy":"2022-03-20T14:12:00.438229Z","iopub.execute_input":"2022-03-20T14:12:00.438528Z","iopub.status.idle":"2022-03-20T14:12:00.446411Z","shell.execute_reply.started":"2022-03-20T14:12:00.438496Z","shell.execute_reply":"2022-03-20T14:12:00.44528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install /kaggle/input/efficientnet-100 -q\n!pip install /kaggle/input/image-classifiers -q","metadata":{"execution":{"iopub.status.busy":"2022-03-20T14:12:00.448379Z","iopub.execute_input":"2022-03-20T14:12:00.449296Z","iopub.status.idle":"2022-03-20T14:12:56.333205Z","shell.execute_reply.started":"2022-03-20T14:12:00.44925Z","shell.execute_reply":"2022-03-20T14:12:56.331942Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nfrom tqdm.notebook import tqdm\nimport gc\nimport pickle\n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras import models\n\nimport efficientnet.tfkeras\nfrom sklearn.preprocessing import StandardScaler","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-20T14:13:27.967152Z","iopub.execute_input":"2022-03-20T14:13:27.967465Z","iopub.status.idle":"2022-03-20T14:13:28.078501Z","shell.execute_reply.started":"2022-03-20T14:13:27.967435Z","shell.execute_reply":"2022-03-20T14:13:28.077777Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Helper functions","metadata":{}},{"cell_type":"code","source":"def auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    \n    return strategy\n\n\ndef build_decoder(with_labels=True, target_size=(300, 300), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n\n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.image.resize(img, target_size)\n\n        return img\n    \n    def decode_with_labels(path, label):\n        return decode(path), label\n    \n    return decode_with_labels if with_labels else decode\n\n\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        return img\n    \n    def augment_with_labels(img, label):\n        return augment(img), label\n    \n    return augment_with_labels if with_labels else augment\n\n\ndef build_dataset(paths, labels=None, bsize=32, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    \n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n    \n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n    \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    \n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)\n    \n    return dset","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_kg_hide-output":false,"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-03-20T14:13:32.478713Z","iopub.execute_input":"2022-03-20T14:13:32.479044Z","iopub.status.idle":"2022-03-20T14:13:32.49901Z","shell.execute_reply.started":"2022-03-20T14:13:32.479011Z","shell.execute_reply":"2022-03-20T14:13:32.497884Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Variables and configurations","metadata":{}},{"cell_type":"code","source":"# 変更の必要性あり\nCOMPETITION_NAME = \"ai-medical-contest-2022\"\nstrategy = auto_select_accelerator()\nBATCH_SIZE = strategy.num_replicas_in_sync * 16","metadata":{"execution":{"iopub.status.busy":"2022-03-20T14:13:38.055109Z","iopub.execute_input":"2022-03-20T14:13:38.055431Z","iopub.status.idle":"2022-03-20T14:13:38.066429Z","shell.execute_reply.started":"2022-03-20T14:13:38.055399Z","shell.execute_reply":"2022-03-20T14:13:38.065462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 変更の必要性あり\ntarget_cols=['pneumonia','source']","metadata":{"execution":{"iopub.status.busy":"2022-03-20T14:13:40.673494Z","iopub.execute_input":"2022-03-20T14:13:40.673969Z","iopub.status.idle":"2022-03-20T14:13:40.681833Z","shell.execute_reply.started":"2022-03-20T14:13:40.673925Z","shell.execute_reply":"2022-03-20T14:13:40.680599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"load_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\n\ndf = pd.read_csv(load_dir + 'train.csv')\npaths = load_dir + \"image/image/\" + df['id'] + '.png'\n\nsub_df = pd.read_csv(load_dir + 'sample_submission.csv')\ntest_paths = load_dir + \"test/\" + sub_df['id'] + '.png'\n\n# Get the multi-labels\nlabel_cols = sub_df.columns[1:]\nlabels = df[label_cols].values","metadata":{"execution":{"iopub.status.busy":"2022-03-20T14:13:43.11579Z","iopub.execute_input":"2022-03-20T14:13:43.116223Z","iopub.status.idle":"2022-03-20T14:13:43.181362Z","shell.execute_reply.started":"2022-03-20T14:13:43.116163Z","shell.execute_reply":"2022-03-20T14:13:43.179939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# predct test data with EfficientNet_first","metadata":{}},{"cell_type":"code","source":"# Build the tensorflow datasets\nIMSIZES = (224, 240, 256, 300, 380, 456, 528, 600)\n# index i corresponds to b-i\nsize = IMSIZES[2]\n\ndecoder = build_decoder(with_labels=True, target_size=(size, size))\ntest_decoder = build_decoder(with_labels=False, target_size=(size, size))\n\n# Build the tensorflow datasets\ndtrain = build_dataset(\n    train_paths, train_labels, bsize=BATCH_SIZE, \n    cache_dir='/kaggle/tf_cache', decode_fn=decoder\n)\n\ndvalid = build_dataset(\n    valid_paths, valid_labels, bsize=BATCH_SIZE, \n    repeat=False, shuffle=False, augment=False, \n    cache_dir='/kaggle/tf_cache', decode_fn=decoder\n)\n\ndtest = build_dataset(\n    test_paths, bsize=BATCH_SIZE, repeat=False, \n    shuffle=False, augment=False, cache=False, \n    decode_fn=test_decoder\n)","metadata":{"execution":{"iopub.status.busy":"2022-03-20T14:13:47.718539Z","iopub.execute_input":"2022-03-20T14:13:47.718916Z","iopub.status.idle":"2022-03-20T14:13:47.748755Z","shell.execute_reply.started":"2022-03-20T14:13:47.718883Z","shell.execute_reply":"2022-03-20T14:13:47.747538Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_path = '../input/tfkeras-efficientnet-weights/efficientnetb7_notop.h5'  # imagenet\nn_labels = labels.shape[1]\n\nwith strategy.scope():\n    model = tf.keras.Sequential([\n        tf.keras.applications.EfficientNetB7(\n            input_shape=(size, size, 3),\n            weights=model_path,\n            include_top=False,\n            drop_connect_rate=0.5),\n        tf.keras.layers.GlobalAveragePooling2D(),\n        tf.keras.layers.Dense(n_labels, activation='sigmoid')\n    ])\n    model.compile(\n        optimizer='adam',\n        loss='binary_crossentropy',\n        metrics=[tf.keras.metrics.AUC(multi_label=True)])\n    model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-03-20T14:13:52.894505Z","iopub.execute_input":"2022-03-20T14:13:52.894984Z","iopub.status.idle":"2022-03-20T14:13:52.923398Z","shell.execute_reply.started":"2022-03-20T14:13:52.894938Z","shell.execute_reply":"2022-03-20T14:13:52.921794Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ############### Train the model ###############\nsteps_per_epoch = train_paths.shape[0] // BATCH_SIZE\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(\n    'model.h5', save_best_only=True, monitor='val_auc', mode='max')\nlr_reducer = tf.keras.callbacks.ReduceLROnPlateau(\n    monitor=\"val_auc\", patience=3, min_lr=1e-6, mode='max')","metadata":{"execution":{"iopub.status.busy":"2022-03-20T14:13:58.416178Z","iopub.execute_input":"2022-03-20T14:13:58.416492Z","iopub.status.idle":"2022-03-20T14:13:58.441346Z","shell.execute_reply.started":"2022-03-20T14:13:58.416461Z","shell.execute_reply":"2022-03-20T14:13:58.439938Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model.fit(\n    dtrain, \n    epochs=20,\n    verbose=1,\n    callbacks=[checkpoint, lr_reducer],\n    steps_per_epoch=steps_per_epoch,\n    validation_data=dvalid)","metadata":{"execution":{"iopub.status.busy":"2022-03-20T14:13:59.039999Z","iopub.execute_input":"2022-03-20T14:13:59.040328Z","iopub.status.idle":"2022-03-20T14:13:59.065592Z","shell.execute_reply.started":"2022-03-20T14:13:59.040296Z","shell.execute_reply":"2022-03-20T14:13:59.063704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# save models","metadata":{}},{"cell_type":"code","source":"# モデルの保存\nopen('model.json',\"w\").write(model.to_json())\n\n# 学習済みの重みを保存\nmodel.save_weights('weight.hdf5')\n\n# モデルの読み込み\nmodel_from_json(open('model.json',\"w\").read(model.to_json()))","metadata":{"execution":{"iopub.status.busy":"2022-03-20T14:13:59.813185Z","iopub.execute_input":"2022-03-20T14:13:59.813508Z","iopub.status.idle":"2022-03-20T14:13:59.839346Z","shell.execute_reply.started":"2022-03-20T14:13:59.813477Z","shell.execute_reply":"2022-03-20T14:13:59.838029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# save history","metadata":{}},{"cell_type":"code","source":"hist_df = pd.DataFrame(history.history)\nhist_df.to_csv('history.csv')","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub_df[label_cols] = model.predict(dtest, verbose=1)\nsub_df.to_csv('submission.csv', index=False)\n\nsub_df.head()","metadata":{},"execution_count":null,"outputs":[]}]}