{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nimport matplotlib.pyplot as plt\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")\n\n# ML tools \nimport tensorflow as tf\n# from kaggle_datasets import KaggleDatasets\n# from keras.models import Sequential\n# from keras.layers import Dense, Flatten, Activation, Conv2D, MaxPooling2D, Dropout, Conv2D,MaxPooling2D,GlobalAveragePooling2D\n# from keras.optimizers import Adam\n# from tensorflow.keras import Model\n# from tensorflow.keras.applications import Xception\nimport os\n# from keras import optimizers\n# from sklearn.model_selection import train_test_split\n# from tensorflow.keras.callbacks import ReduceLROnPlateau, ModelCheckpoint, EarlyStopping","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# df_target = pd.read_csv('../input/ranzcr-clip-catheter-line-classification/train.csv')\n# display(df_target.head(3))\n# print(df_target.shape)\n# df_sample = pd.read_csv('../input/ranzcr-clip-catheter-line-classification/sample_submission.csv')\n# display(df_sample.head(3))\n# print(df_sample.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# target_cols = df_target.drop(['StudyInstanceUID','PatientID'], axis=1).columns.to_list()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# n_classes = len(target_cols)\n# img_size = 600\n# n_epochs = 30","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# def auto_select_accelerator():\n#     try:\n#         tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n#         tf.config.experimental_connect_to_cluster(tpu)\n#         tf.tpu.experimental.initialize_tpu_system(tpu)\n#         strategy = tf.distribute.experimental.TPUStrategy(tpu)\n#         print(\"Running on TPU:\", tpu.master())\n#     except ValueError:\n#         strategy = tf.distribute.get_strategy()\n#     print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    \n#     return strategy\n\n# def build_decoder(with_labels=True, target_size=(img_size, img_size), ext='jpg'):\n#     def decode(path):\n#         file_bytes = tf.io.read_file(path) # Reads and outputs the entire contents of the input filename.\n\n#         if ext == 'png':\n#             img = tf.image.decode_png(file_bytes, channels=3) # Decode a PNG-encoded image to a uint8 or uint16 tensor\n#         elif ext in ['jpg', 'jpeg']:\n#             img = tf.image.decode_jpeg(file_bytes, channels=3) # Decode a JPEG-encoded image to a uint8 tensor\n#         else:\n#             raise ValueError(\"Image extension not supported\")\n\n#         img = tf.cast(img, tf.float32) / 255.0 # Casts a tensor to the type float32 and divides by 255.\n#         img = tf.image.resize(img, target_size) # Resizing to target size\n#         return img\n    \n#     def decode_with_labels(path, label):\n#         return decode(path), label\n    \n#     return decode_with_labels if with_labels else decode\n\n\n# def build_augmenter(with_labels=True):\n#     def augment(img):\n#         img = tf.image.random_flip_left_right(img)\n#         img = tf.image.random_flip_up_down(img)\n#         img = tf.image.random_saturation(img, 0.8, 1.2)\n#         img = tf.image.random_brightness(img, 0.2)\n#         img = tf.image.random_contrast(img, 0.8, 1.2)\n#         img = tf.image.random_hue(img, 0.2)\n#         return img\n    \n#     def augment_with_labels(img, label):\n#         return augment(img), label\n    \n#     return augment_with_labels if with_labels else augment\n\n# def build_dataset(paths, labels=None, bsize=32, cache=True,\n#                   decode_fn=None, augment_fn=None,\n#                   augment=True, repeat=True, shuffle=1024, \n#                   cache_dir=\"\"):\n#     if cache_dir != \"\" and cache is True:\n#         os.makedirs(cache_dir, exist_ok=True)\n    \n#     if decode_fn is None:\n#         decode_fn = build_decoder(labels is not None)\n    \n#     if augment_fn is None:\n#         augment_fn = build_augmenter(labels is not None)\n    \n#     AUTO = tf.data.experimental.AUTOTUNE\n#     slices = paths if labels is None else (paths, labels)\n    \n#     dset = tf.data.Dataset.from_tensor_slices(slices)\n#     dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n#     dset = dset.cache(cache_dir) if cache else dset\n#     dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n#     dset = dset.repeat() if repeat else dset\n#     dset = dset.shuffle(shuffle) if shuffle else dset\n#     dset = dset.batch(bsize).prefetch(AUTO) # overlaps data preprocessing and model execution while training\n#     return dset","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# COMPETITION_NAME = \"ranzcr-clip-catheter-line-classification\"\n# strategy = auto_select_accelerator()\n# batch_size = strategy.num_replicas_in_sync * 16\n# print('batch size', batch_size)\n# GCS_DS_PATH = KaggleDatasets().get_gcs_path(COMPETITION_NAME)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# load_dir = '/kaggle/input/ranzcr-clip-catheter-line-classification/'\n# df_train = pd.read_csv(load_dir + 'train.csv')\n# paths = GCS_DS_PATH + \"/train/\" + df_train['StudyInstanceUID'] + '.jpg'\n\n# df_sub = pd.read_csv(load_dir + 'sample_submission.csv')\n# test_paths = GCS_DS_PATH + \"/test/\" + df_sub['StudyInstanceUID'] + '.jpg'\n\n# # Get the multi-labels\n# label_cols = df_sub.columns[1:]\n# labels = df_train[label_cols].values","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# # Train test split\n# train_paths, valid_paths, train_labels, valid_labels = train_test_split(paths, labels, test_size=0.10, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# # Build the tensorflow datasets\n\n# decoder = build_decoder(with_labels=True, target_size=(img_size, img_size))\n\n# # Build the tensorflow datasets\n# dtrain = build_dataset(\n#     train_paths, train_labels, bsize=batch_size, decode_fn=decoder\n# )\n\n# dvalid = build_dataset(\n#     valid_paths, valid_labels, bsize=batch_size, \n#     repeat=False, shuffle=False, augment=False, decode_fn=decoder\n# )","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# with strategy.scope():\n#     net = Xception(include_top=False,input_shape=(img_size, img_size, 3), weights='imagenet')\n#     x = net.output\n#     x = GlobalAveragePooling2D()(x)\n#     x = Dropout(0.5)(x)\n#     output = Dense(n_classes, activation='sigmoid')(x)\n#     model = Model(inputs=net.input, outputs=output)\n#     model.compile(optimizers.Adam(lr=1e-3),loss='binary_crossentropy',metrics=[tf.keras.metrics.AUC(multi_label=True)])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# rlr = ReduceLROnPlateau(monitor = 'val_loss', factor = 0.1, patience = 2, verbose = 0, \n#                                 min_delta = 1e-4, min_lr = 1e-6, mode = 'min')\n        \n# ckp = ModelCheckpoint('model.h5',monitor = 'val_loss',\n#                       verbose = 0, save_best_only = True, mode = 'min')\n        \n# es = EarlyStopping(monitor = 'val_loss', min_delta = 1e-4, patience = 5, mode = 'min', \n#                     restore_best_weights = True, verbose = 0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# steps_per_epoch = train_paths.shape[0] // batch_size","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# history = model.fit(dtrain,                      \n#                     validation_data=dvalid,                                       \n#                     epochs=n_epochs,\n#                     callbacks=[rlr,es,ckp],\n#                     steps_per_epoch=steps_per_epoch,\n#                     verbose=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# plt.rcParams.update({'font.size': 16})\n# hist = pd.DataFrame(history.history)\n# fig, (ax1, ax2) = plt.subplots(figsize=(12,12),nrows=2, ncols=1)\n# hist['loss'].plot(ax=ax1,c='k',label='training loss')\n# hist['val_loss'].plot(ax=ax1,c='r',linestyle='--', label='validation loss')\n# ax1.legend()\n# hist['auc'].plot(ax=ax2,c='k',label='training AUC')\n# hist['val_auc'].plot(ax=ax2,c='r',linestyle='--',label='validation AUC')\n# ax2.legend()\n# plt.show()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"Inference Code Starts-"},{"metadata":{"trusted":true},"cell_type":"code","source":"img_size = 600\ndef auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    \n    return strategy\n\ndef build_decoder(with_labels=True, target_size=(img_size, img_size), ext='jpg'):\n    def decode(path):\n        file_bytes = tf.io.read_file(path) # Reads and outputs the entire contents of the input filename.\n\n        if ext == 'png':\n            img = tf.image.decode_png(file_bytes, channels=3) # Decode a PNG-encoded image to a uint8 or uint16 tensor\n        elif ext in ['jpg', 'jpeg']:\n            img = tf.image.decode_jpeg(file_bytes, channels=3) # Decode a JPEG-encoded image to a uint8 tensor\n        else:\n            raise ValueError(\"Image extension not supported\")\n\n        img = tf.cast(img, tf.float32) / 255.0 # Casts a tensor to the type float32 and divides by 255.\n        img = tf.image.resize(img, target_size) # Resizing to target size\n        return img\n    \n    def decode_with_labels(path, label):\n        return decode(path), label\n    \n    return decode_with_labels if with_labels else decode\n\n\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        img = tf.image.random_flip_up_down(img)\n        img = tf.image.random_saturation(img, 0.8, 1.2)\n        img = tf.image.random_brightness(img, 0.1)\n        img = tf.image.random_contrast(img, 0.9, 1.2)\n        return img\n    \n    def augment_with_labels(img, label):\n        return augment(img), label\n    \n    return augment_with_labels if with_labels else augment\n\ndef build_dataset(paths, labels=None, bsize=32, cache=True,\n                  decode_fn=None, augment_fn=None,\n                  augment=True, repeat=True, shuffle=1024, \n                  cache_dir=\"\"):\n    if cache_dir != \"\" and cache is True:\n        os.makedirs(cache_dir, exist_ok=True)\n    \n    if decode_fn is None:\n        decode_fn = build_decoder(labels is not None)\n    \n    if augment_fn is None:\n        augment_fn = build_augmenter(labels is not None)\n    \n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    \n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.cache(cache_dir) if cache else dset\n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(bsize).prefetch(AUTO) # overlaps data preprocessing and model execution while training\n    return dset\n\nCOMPETITION_NAME = \"ranzcr-clip-catheter-line-classification\"\nstrategy = auto_select_accelerator()\nbatch_size = strategy.num_replicas_in_sync * 16\nprint('batch size', batch_size)\n\nwith strategy.scope():\n    model = tf.keras.models.load_model('../input/ranzcr-tpu-weights/model.h5', custom_objects={'FixedDropout': tf.keras.layers.Dropout})\n    \n\n\ntest_decoder = build_decoder(with_labels=False, target_size=(img_size, img_size))\nload_dir = '/kaggle/input/ranzcr-clip-catheter-line-classification/'\ndf_sub = pd.read_csv(load_dir + 'sample_submission.csv')\ntest_paths = load_dir + \"test/\" + df_sub['StudyInstanceUID'] + '.jpg'\ndtest = build_dataset(\n    test_paths, bsize=batch_size, repeat=False, \n    shuffle=False, augment=False, cache=False, \n    decode_fn=test_decoder\n)\n\n\n\n\ny_preds = model.predict(dtest, verbose=1)\ndf_sub.iloc[:, 1:] = y_preds\ndisplay(df_sub)\n\n\ndf_sub.to_csv('submission.csv',index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}