{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install efficientnet -q","metadata":{"execution":{"iopub.status.busy":"2023-03-11T18:01:20.504672Z","iopub.execute_input":"2023-03-11T18:01:20.505061Z","iopub.status.idle":"2023-03-11T18:01:31.563329Z","shell.execute_reply.started":"2023-03-11T18:01:20.505026Z","shell.execute_reply":"2023-03-11T18:01:31.562364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\n\nimport efficientnet.tfkeras as efn\nimport numpy as np\nimport pandas as pd\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.model_selection import train_test_split\nimport tensorflow as tf","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-03-11T18:01:34.294889Z","iopub.execute_input":"2023-03-11T18:01:34.295274Z","iopub.status.idle":"2023-03-11T18:01:43.508173Z","shell.execute_reply.started":"2023-03-11T18:01:34.295239Z","shell.execute_reply":"2023-03-11T18:01:43.506772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Helper functions","metadata":{}},{"cell_type":"markdown","source":"The following functions are defined below (unhide to see):\n```python\nauto_select_accelerator()\n\nbuild_decoder(with_labels=True, target_size=(256, 256), ext='jpg')\n\nbuild_augmenter(with_labels=True)\n\nbuild_dataset(paths, labels=None, bsize=32, cache=True,\n              decode_fn=None, augment_fn=None,\n              augment=True, repeat=True, shuffle=1024, \n              cache_dir=\"\")\n```","metadata":{}},{"cell_type":"code","source":"def auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    \n    return strategy\n\n\ndef build_decoder(with_labels=True, target_size=(256, 256), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path)\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3)\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3)\n        else:\n            raise ValueError(\"Image extension not supported\")\n\n        img = tf.cast(img, tf.float32) / 255.0\n        img = tf.image.resize(img, target_size)\n\n        return img\n    \n    def decode_with_labels(path, label):\n        return decode(path), label\n    \n    return decode_with_labels if with_labels else decode\n\n\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        return img\n    \n    def augment_with_labels(img, label):\n        return augment(img), label\n    \n    return augment_with_labels if with_labels else augment\n\n\ndef build_dataset(paths, labels=None, bsize=32, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    \n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n    \n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n    \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    \n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO)\n    \n    return dset","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","_kg_hide-input":true,"execution":{"iopub.status.busy":"2023-03-11T18:01:48.690653Z","iopub.execute_input":"2023-03-11T18:01:48.691396Z","iopub.status.idle":"2023-03-11T18:01:48.708610Z","shell.execute_reply.started":"2023-03-11T18:01:48.691350Z","shell.execute_reply":"2023-03-11T18:01:48.706901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Variables and configurations","metadata":{}},{"cell_type":"code","source":"COMPETITION_NAME = \"ranzcr-clip-catheter-line-classification\"\nstrategy = auto_select_accelerator()\nBATCH_SIZE = strategy.num_replicas_in_sync * 16","metadata":{"execution":{"iopub.status.busy":"2023-03-11T18:01:56.181046Z","iopub.execute_input":"2023-03-11T18:01:56.181394Z","iopub.status.idle":"2023-03-11T18:02:01.726282Z","shell.execute_reply.started":"2023-03-11T18:01:56.181362Z","shell.execute_reply":"2023-03-11T18:02:01.725120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GCS_DS_PATH = KaggleDatasets().get_gcs_path(COMPETITION_NAME)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T18:02:03.251697Z","iopub.execute_input":"2023-03-11T18:02:03.252060Z","iopub.status.idle":"2023-03-11T18:02:03.598985Z","shell.execute_reply.started":"2023-03-11T18:02:03.252030Z","shell.execute_reply":"2023-03-11T18:02:03.597300Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Preparing dataset","metadata":{}},{"cell_type":"markdown","source":"### Loading and preprocess CSVs","metadata":{}},{"cell_type":"code","source":"load_dir = f\"/kaggle/input/{COMPETITION_NAME}/\"\ndf = pd.read_csv(load_dir + 'train.csv')\n\n# paths = load_dir + \"train/\" + df['StudyInstanceUID'] + '.jpg'\npaths = GCS_DS_PATH + \"/train/\" + df['StudyInstanceUID'] + '.jpg'\n\nsub_df = pd.read_csv(load_dir + 'sample_submission.csv')\n\n# test_paths = load_dir + \"test/\" + sub_df['StudyInstanceUID'] + '.jpg'\ntest_paths = GCS_DS_PATH + \"/test/\" + sub_df['StudyInstanceUID'] + '.jpg'\n\n# Get the multi-labels\nlabel_cols = sub_df.columns[1:]\nlabels = df[label_cols].values","metadata":{"execution":{"iopub.status.busy":"2023-03-11T18:02:07.717481Z","iopub.execute_input":"2023-03-11T18:02:07.717887Z","iopub.status.idle":"2023-03-11T18:02:07.889625Z","shell.execute_reply.started":"2023-03-11T18:02:07.717851Z","shell.execute_reply":"2023-03-11T18:02:07.887957Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Train test split\n(\n    train_paths, valid_paths, \n    train_labels, valid_labels\n) = train_test_split(paths, labels, test_size=0.2, random_state=42)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T18:02:12.262512Z","iopub.execute_input":"2023-03-11T18:02:12.262900Z","iopub.status.idle":"2023-03-11T18:02:12.277184Z","shell.execute_reply.started":"2023-03-11T18:02:12.262867Z","shell.execute_reply":"2023-03-11T18:02:12.276209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm -rf /kaggle/tf_cache","metadata":{"execution":{"iopub.status.busy":"2023-03-09T11:57:56.098554Z","iopub.execute_input":"2023-03-09T11:57:56.099450Z","iopub.status.idle":"2023-03-09T11:57:56.567878Z","shell.execute_reply.started":"2023-03-09T11:57:56.099410Z","shell.execute_reply":"2023-03-09T11:57:56.566434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rm -rf /kaggle/tf_cache_0.lockfile","metadata":{"execution":{"iopub.status.busy":"2023-03-09T11:58:05.453427Z","iopub.execute_input":"2023-03-09T11:58:05.455056Z","iopub.status.idle":"2023-03-09T11:58:05.789824Z","shell.execute_reply.started":"2023-03-09T11:58:05.454976Z","shell.execute_reply":"2023-03-09T11:58:05.788438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build the tensorflow datasets\nIMSIZE = (224, 240, 260, 300, 380, 456, 528, 600)\n\ndecoder = build_decoder(with_labels=True, target_size=(IMSIZE[7], IMSIZE[7]))\ntest_decoder = build_decoder(with_labels=False, target_size=(IMSIZE[7], IMSIZE[7]))\n\ntrain_dataset = build_dataset(\n    train_paths, train_labels, bsize=BATCH_SIZE, decode_fn=decoder\n)\n\nvalid_dataset = build_dataset(\n    valid_paths, valid_labels, bsize=BATCH_SIZE, decode_fn=decoder,\n    repeat=False, shuffle=False, augment=False\n)\n\ntest_dataset = build_dataset(\n    test_paths, cache=False, bsize=BATCH_SIZE, decode_fn=test_decoder,\n    repeat=False, shuffle=False, augment=False\n)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T18:02:18.062072Z","iopub.execute_input":"2023-03-11T18:02:18.062438Z","iopub.status.idle":"2023-03-11T18:02:18.358686Z","shell.execute_reply.started":"2023-03-11T18:02:18.062406Z","shell.execute_reply":"2023-03-11T18:02:18.357013Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_test = tf.keras.models.load_model('/kaggle/input/trained-model/model (6).h5')","metadata":{"execution":{"iopub.status.busy":"2023-03-11T18:02:38.413344Z","iopub.execute_input":"2023-03-11T18:02:38.413738Z","iopub.status.idle":"2023-03-11T18:03:01.599721Z","shell.execute_reply.started":"2023-03-11T18:02:38.413697Z","shell.execute_reply":"2023-03-11T18:03:01.598129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model_test.predict(valid_dataset, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-03-11T18:03:07.339336Z","iopub.execute_input":"2023-03-11T18:03:07.339803Z","iopub.status.idle":"2023-03-11T18:31:44.999768Z","shell.execute_reply.started":"2023-03-11T18:03:07.339751Z","shell.execute_reply":"2023-03-11T18:31:44.997980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Convert predictions to binary class labels\nbinary_predictions = (predictions > 0.5).astype(int)\nbinary_predictions","metadata":{"execution":{"iopub.status.busy":"2023-03-11T18:31:49.035174Z","iopub.execute_input":"2023-03-11T18:31:49.035649Z","iopub.status.idle":"2023-03-11T18:31:49.050797Z","shell.execute_reply.started":"2023-03-11T18:31:49.035610Z","shell.execute_reply":"2023-03-11T18:31:49.048880Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import roc_auc_score,accuracy_score\nfrom sklearn.metrics import roc_curve,auc","metadata":{"execution":{"iopub.status.busy":"2023-03-11T18:31:59.196729Z","iopub.execute_input":"2023-03-11T18:31:59.197208Z","iopub.status.idle":"2023-03-11T18:31:59.203563Z","shell.execute_reply.started":"2023-03-11T18:31:59.197170Z","shell.execute_reply":"2023-03-11T18:31:59.201961Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"area_under_curve=roc_auc_score(binary_predictions,valid_labels,average='micro')\narea_under_curve","metadata":{"execution":{"iopub.status.busy":"2023-03-11T18:32:34.741172Z","iopub.execute_input":"2023-03-11T18:32:34.741594Z","iopub.status.idle":"2023-03-11T18:32:34.757748Z","shell.execute_reply.started":"2023-03-11T18:32:34.741557Z","shell.execute_reply":"2023-03-11T18:32:34.755900Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import f1_score\nf1=f1_score(binary_predictions,valid_labels,average='micro')\nf1","metadata":{"execution":{"iopub.status.busy":"2023-03-11T18:41:05.168403Z","iopub.execute_input":"2023-03-11T18:41:05.168773Z","iopub.status.idle":"2023-03-11T18:41:05.186513Z","shell.execute_reply.started":"2023-03-11T18:41:05.168740Z","shell.execute_reply":"2023-03-11T18:41:05.185118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}