{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceType":"competition","sourceId":20270,"databundleVersionId":1222630},{"sourceType":"datasetVersion","sourceId":1324385,"datasetId":762138,"databundleVersionId":1356663},{"sourceType":"datasetVersion","sourceId":1322552,"datasetId":688719,"databundleVersionId":1354820}],"dockerImageVersionId":30299,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Cancer classification","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:31:36.713126Z","iopub.execute_input":"2024-10-02T06:31:36.714028Z","iopub.status.idle":"2024-10-02T06:31:36.735770Z","shell.execute_reply.started":"2024-10-02T06:31:36.713942Z","shell.execute_reply":"2024-10-02T06:31:36.735037Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"This is for the TPU settings","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy() # default distribution strategy in Tensorflow. Works on CPU and single GPU.\n\n\nAUTO     = tf.data.experimental.AUTOTUNE\nREPLICAS = strategy.num_replicas_in_sync\n#REPLICAS = 8\nprint(f'REPLICAS: {REPLICAS}')","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:31:39.840563Z","iopub.execute_input":"2024-10-02T06:31:39.841254Z","iopub.status.idle":"2024-10-02T06:31:44.036128Z","shell.execute_reply.started":"2024-10-02T06:31:39.841199Z","shell.execute_reply":"2024-10-02T06:31:44.035152Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Load dataset","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/train.csv\")\ndf_test = pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/test.csv\")\nsample_sub = pd.read_csv(\"/kaggle/input/siim-isic-melanoma-classification/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:31:48.066335Z","iopub.execute_input":"2024-10-02T06:31:48.067498Z","iopub.status.idle":"2024-10-02T06:31:48.184930Z","shell.execute_reply.started":"2024-10-02T06:31:48.067456Z","shell.execute_reply":"2024-10-02T06:31:48.184102Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"from skimage.io import imread\nnumpy_array = imread(\"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/ISIC_0015719.jpg\")","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:31:49.681025Z","iopub.execute_input":"2024-10-02T06:31:49.681775Z","iopub.status.idle":"2024-10-02T06:31:50.503143Z","shell.execute_reply.started":"2024-10-02T06:31:49.681735Z","shell.execute_reply":"2024-10-02T06:31:50.502074Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"base = \"/kaggle/input/siim-isic-melanoma-classification/jpeg/train/\"","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:31:52.033725Z","iopub.execute_input":"2024-10-02T06:31:52.034572Z","iopub.status.idle":"2024-10-02T06:31:52.039274Z","shell.execute_reply.started":"2024-10-02T06:31:52.034531Z","shell.execute_reply":"2024-10-02T06:31:52.037996Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"image_name = list(df_train[\"image_name\"])\nimg_link = [base + x for x in image_name]","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:31:53.661113Z","iopub.execute_input":"2024-10-02T06:31:53.661841Z","iopub.status.idle":"2024-10-02T06:31:53.680579Z","shell.execute_reply.started":"2024-10-02T06:31:53.661802Z","shell.execute_reply":"2024-10-02T06:31:53.679506Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train[\"img_link\"] = img_link","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:31:56.768873Z","iopub.execute_input":"2024-10-02T06:31:56.769519Z","iopub.status.idle":"2024-10-02T06:31:56.778147Z","shell.execute_reply.started":"2024-10-02T06:31:56.769481Z","shell.execute_reply":"2024-10-02T06:31:56.777153Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train.head()","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:32:47.650859Z","iopub.execute_input":"2024-10-02T06:32:47.651739Z","iopub.status.idle":"2024-10-02T06:32:47.674058Z","shell.execute_reply.started":"2024-10-02T06:32:47.651703Z","shell.execute_reply":"2024-10-02T06:32:47.673127Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### This is for loading the images to train for the model","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nGCS_PATH = KaggleDatasets().get_gcs_path(\"siim-isic-melanoma-classification\")\ntrain_filenames = tf.io.gfile.glob(GCS_PATH + '/tfrecords/train*.tfrec')\ntest_filenames = tf.io.gfile.glob(GCS_PATH + '/tfrecords/test*.tfrec')","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:32:52.080217Z","iopub.execute_input":"2024-10-02T06:32:52.081173Z","iopub.status.idle":"2024-10-02T06:32:52.545210Z","shell.execute_reply.started":"2024-10-02T06:32:52.081138Z","shell.execute_reply":"2024-10-02T06:32:52.542698Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"BATCH_SIZE = 8 * strategy.num_replicas_in_sync\nIMAGE_SIZE = [1024,1024]\nAUTO = tf.data.experimental.AUTOTUNE\nimSize = 512","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:32:53.700107Z","iopub.execute_input":"2024-10-02T06:32:53.700501Z","iopub.status.idle":"2024-10-02T06:32:53.705574Z","shell.execute_reply.started":"2024-10-02T06:32:53.700464Z","shell.execute_reply":"2024-10-02T06:32:53.704532Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def decode_image(image_data):\n    # Decoding the images\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0  # convert image to floats in [0, 1] range\n    image = tf.reshape(image, [*IMAGE_SIZE, 3]) # explicit size needed for TPU\n    image = tf.image.resize(image, [imSize,imSize])\n    return image\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string means bytestring\n        \"target\": tf.io.FixedLenFeature([], tf.int64),  # shape [] means single element\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['target'], tf.int32)\n    return image, label # returns a dataset of (image, label) pairs\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string means bytestring\n        \"image_name\": tf.io.FixedLenFeature([], tf.string),  # shape [] means single element\n        # class is missing, this competitions's challenge is to predict flower classes for the test dataset\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['image_name']\n    return image, idnum # returns a dataset of image(s)\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    # Read from TFRecords. For optimal performance, reading from multiple files at once and\n    # disregarding data order. Order does not matter since we will be shuffling the data anyway.\n\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False # disable order, increase speed\n\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO) # automatically interleaves reads from multiple files\n    dataset = dataset.with_options(ignore_order) # uses data as soon as it streams in, rather than in its original order\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, num_parallel_calls=AUTO)\n    # returns a dataset of (image, label) pairs if labeled=True or (image, id) pairs if labeled=False\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:32:55.582303Z","iopub.execute_input":"2024-10-02T06:32:55.582703Z","iopub.status.idle":"2024-10-02T06:32:55.596284Z","shell.execute_reply.started":"2024-10-02T06:32:55.582671Z","shell.execute_reply":"2024-10-02T06:32:55.595187Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"### Data preprocessing functions","metadata":{}},{"cell_type":"code","source":"def data_augment(image, label):\n    # Change the orientation of images to make model train on more \n    # generalized dataset\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_flip_up_down(image)\n    image = tf.image.random_saturation(image, 0, 2)\n    image = tf.image.rot90(image)\n    return image, label   \ndef get_training_dataset():\n    dataset = load_dataset(train_filenames, labeled=True)\n    dataset = dataset.map(data_augment, num_parallel_calls=AUTO)\n    dataset = dataset.repeat() \n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO) \n    return dataset\n\ndef get_val_dataset():\n    dataset = load_dataset(valid_filenames, labeled=True)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTO) \n    return dataset\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(test_filenames, labeled=False,ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO) \n    return dataset\n\ntest_dataset = get_test_dataset(ordered=True)\ntest_images_ds = test_dataset.map(lambda image, idnum: image)","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:32:59.640928Z","iopub.execute_input":"2024-10-02T06:32:59.641604Z","iopub.status.idle":"2024-10-02T06:33:03.080367Z","shell.execute_reply.started":"2024-10-02T06:32:59.641556Z","shell.execute_reply":"2024-10-02T06:33:03.079537Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Creating the model","metadata":{}},{"cell_type":"code","source":"# The model stop training when accuracy hits 98.4%\nclass myCallbacks(tf.keras.callbacks.Callback):\n    def on_epoch_end(self, epoch, logs={}):\n        if (logs.get(\"val_accuracy\")>0.984):\n            self.model.stop_training = True","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:33:06.301199Z","iopub.execute_input":"2024-10-02T06:33:06.301671Z","iopub.status.idle":"2024-10-02T06:33:07.289589Z","shell.execute_reply.started":"2024-10-02T06:33:06.301620Z","shell.execute_reply":"2024-10-02T06:33:07.288635Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"callbacks = []","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:33:09.021014Z","iopub.execute_input":"2024-10-02T06:33:09.021580Z","iopub.status.idle":"2024-10-02T06:33:09.026360Z","shell.execute_reply.started":"2024-10-02T06:33:09.021526Z","shell.execute_reply":"2024-10-02T06:33:09.025290Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Customizing learning rate\ndef build_lrfn(lr_start=0.00001, lr_max=0.000075, \n               lr_min=0.000001, lr_rampup_epochs=20, \n               lr_sustain_epochs=0, lr_exp_decay=.8):\n    lr_max = lr_max * strategy.num_replicas_in_sync\n\n    def lrfn(epoch):\n        if epoch < lr_rampup_epochs:\n            lr = (lr_max - lr_start) / lr_rampup_epochs * epoch + lr_start\n        elif epoch < lr_rampup_epochs + lr_sustain_epochs:\n            lr = lr_max\n        else:\n            lr = (lr_max - lr_min) * lr_exp_decay**(epoch - lr_rampup_epochs - lr_sustain_epochs) + lr_min\n        return lr\n    \n    return lrfn\n\nlrfn = build_lrfn()\nlr_schedule = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=1)\ncallbacks.append(lr_schedule)","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:33:10.981265Z","iopub.execute_input":"2024-10-02T06:33:10.981648Z","iopub.status.idle":"2024-10-02T06:33:10.989733Z","shell.execute_reply.started":"2024-10-02T06:33:10.981614Z","shell.execute_reply":"2024-10-02T06:33:10.988646Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import re\ndef count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)\n\nNUM_TRAINING_IMAGES = count_data_items(train_filenames)\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE # Number of steps per epoch","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:33:13.111140Z","iopub.execute_input":"2024-10-02T06:33:13.112082Z","iopub.status.idle":"2024-10-02T06:33:13.117836Z","shell.execute_reply.started":"2024-10-02T06:33:13.112042Z","shell.execute_reply":"2024-10-02T06:33:13.116829Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Download efficient net\n!pip install yapl==0.1.2 efficientnet > /dev/null","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:33:14.872705Z","iopub.execute_input":"2024-10-02T06:33:14.873071Z","iopub.status.idle":"2024-10-02T06:33:27.726149Z","shell.execute_reply.started":"2024-10-02T06:33:14.873041Z","shell.execute_reply":"2024-10-02T06:33:27.724933Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import efficientnet.tfkeras as efn\nimport numpy as np\nimport pandas as pd\nfrom sklearn.model_selection import train_test_split\ninput_shape = (512, 512, 3)\ndef create_model(train, validation=None):\n    # Creating the model\n    model = tf.keras.Sequential([\n        efn.EfficientNetB0(\n                        input_shape=input_shape,\n                        weights='imagenet',\n                        include_top=False\n                    ), # Getting Efficient net's weights\n        tf.keras.layers.GlobalAveragePooling2D(),\n        \n        tf.keras.layers.Dense(512, activation='relu'),\n        tf.keras.layers.Dropout(0.3),\n        \n        tf.keras.layers.Dense(1, activation='sigmoid')\n    ])\n    optimizer = tf.keras.optimizers.Adam(lr=0.001)\n    model.compile(optimizer=optimizer, loss='binary_crossentropy', metrics=['accuracy'])\n    history = model.fit(x=train, \n                        validation_data=validation, \n                        steps_per_epoch=STEPS_PER_EPOCH, epochs=15, verbose=2).history\n    return model\nmembers = []\nn_members = 3\n# Ensemble learning to get better result\nwith strategy.scope():\n    for _ in range(n_members):\n        train_filenames, valid_filenames = train_test_split(train_filenames, test_size=0.2, shuffle=True)\n        train = get_training_dataset()\n        validation = get_val_dataset()\n        model = create_model(train, validation=validation)\n        members.append(model)\n","metadata":{"execution":{"iopub.status.busy":"2024-10-02T06:33:30.366376Z","iopub.execute_input":"2024-10-02T06:33:30.367224Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"yhats = [model.predict(test_images_ds) for model in members]\nyhats = np.array(yhats)/n_members","metadata":{"execution":{"iopub.status.busy":"2023-01-15T06:09:31.530549Z","iopub.execute_input":"2023-01-15T06:09:31.530818Z","iopub.status.idle":"2023-01-15T06:10:49.427839Z","shell.execute_reply.started":"2023-01-15T06:09:31.530778Z","shell.execute_reply":"2023-01-15T06:10:49.426778Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"pred = np.zeros(shape=10982)\nfor x in yhats:\n    pred += x.flatten()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T06:10:49.429545Z","iopub.execute_input":"2023-01-15T06:10:49.430516Z","iopub.status.idle":"2023-01-15T06:10:49.435894Z","shell.execute_reply.started":"2023-01-15T06:10:49.430469Z","shell.execute_reply":"2023-01-15T06:10:49.43481Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"## Submitting the result","metadata":{}},{"cell_type":"code","source":"test_imgs = test_dataset.map(lambda images, ids: images)\nimg_ids_ds = test_dataset.map(lambda images, ids: ids).unbatch()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T06:10:49.437614Z","iopub.execute_input":"2023-01-15T06:10:49.437845Z","iopub.status.idle":"2023-01-15T06:10:49.488036Z","shell.execute_reply.started":"2023-01-15T06:10:49.43782Z","shell.execute_reply":"2023-01-15T06:10:49.487038Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_ids = []\nfor coutner, ids in enumerate(img_ids_ds):\n    img_ids.append(ids.numpy())","metadata":{"execution":{"iopub.status.busy":"2023-01-15T06:10:49.489403Z","iopub.execute_input":"2023-01-15T06:10:49.489815Z","iopub.status.idle":"2023-01-15T06:11:11.06162Z","shell.execute_reply.started":"2023-01-15T06:10:49.489775Z","shell.execute_reply":"2023-01-15T06:11:11.060672Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"img_ids = np.array(img_ids).astype('U')","metadata":{"execution":{"iopub.status.busy":"2023-01-15T06:11:11.062992Z","iopub.execute_input":"2023-01-15T06:11:11.063846Z","iopub.status.idle":"2023-01-15T06:11:11.071994Z","shell.execute_reply.started":"2023-01-15T06:11:11.063796Z","shell.execute_reply":"2023-01-15T06:11:11.07103Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub = sample_sub.set_index(\"image_name\").reindex(list(img_ids))\nsample_sub[\"target\"] = pred","metadata":{"execution":{"iopub.status.busy":"2023-01-15T06:11:11.073399Z","iopub.execute_input":"2023-01-15T06:11:11.073712Z","iopub.status.idle":"2023-01-15T06:11:11.098422Z","shell.execute_reply.started":"2023-01-15T06:11:11.073681Z","shell.execute_reply":"2023-01-15T06:11:11.097525Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-01-15T06:11:11.099562Z","iopub.execute_input":"2023-01-15T06:11:11.099883Z","iopub.status.idle":"2023-01-15T06:11:11.138452Z","shell.execute_reply.started":"2023-01-15T06:11:11.099854Z","shell.execute_reply":"2023-01-15T06:11:11.137636Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"sample_sub.head()","metadata":{"execution":{"iopub.status.busy":"2023-01-15T06:11:11.139597Z","iopub.execute_input":"2023-01-15T06:11:11.139823Z","iopub.status.idle":"2023-01-15T06:11:11.1504Z","shell.execute_reply.started":"2023-01-15T06:11:11.139798Z","shell.execute_reply":"2023-01-15T06:11:11.149515Z"},"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{},"outputs":[],"execution_count":null}]}