{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.8.17"},"papermill":{"default_parameters":{},"duration":2465.626498,"end_time":"2023-09-04T23:17:43.617837","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2023-09-04T22:36:37.991339","version":"2.3.4"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":20270,"databundleVersionId":1222630,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ☀️ Imports and Setup","metadata":{"papermill":{"duration":0.016302,"end_time":"2023-09-04T22:36:40.759387","exception":false,"start_time":"2023-09-04T22:36:40.743085","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import re\nimport os\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport tempfile\nimport matplotlib as mlp\nimport matplotlib.pyplot as plt\nimport sklearn\nimport math\n\nfrom PIL import Image\nfrom functools import partial\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.model_selection import train_test_split\nfrom tensorflow import keras\nfrom keras import layers, callbacks\nfrom keras.losses import BinaryCrossentropy\nfrom keras.callbacks import ModelCheckpoint,EarlyStopping\nfrom keras import backend as K\n\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Device:', tpu.master())\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept:\n    strategy = tf.distribute.get_strategy()\nprint('Number of replicas:', strategy.num_replicas_in_sync)\n    \nprint(tf.__version__)","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.execute_input":"2023-09-04T22:36:40.790110Z","iopub.status.busy":"2023-09-04T22:36:40.789482Z","iopub.status.idle":"2023-09-04T22:37:33.949819Z","shell.execute_reply":"2023-09-04T22:37:33.945281Z"},"papermill":{"duration":53.187238,"end_time":"2023-09-04T22:37:33.961311","exception":false,"start_time":"2023-09-04T22:36:40.774073","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mlp.rcParams['figure.figsize'] = (12, 10)\ncolors = plt.rcParams['axes.prop_cycle'].by_key()['color']","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:33.997738Z","iopub.status.busy":"2023-09-04T22:37:33.997400Z","iopub.status.idle":"2023-09-04T22:37:34.002278Z","shell.execute_reply":"2023-09-04T22:37:34.001482Z"},"papermill":{"duration":0.025415,"end_time":"2023-09-04T22:37:34.004351","exception":false,"start_time":"2023-09-04T22:37:33.978936","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🦆 Hyperparameters","metadata":{"papermill":{"duration":0.017492,"end_time":"2023-09-04T22:37:34.040209","exception":false,"start_time":"2023-09-04T22:37:34.022717","status":"completed"},"tags":[]}},{"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE\nGCS_PATH = KaggleDatasets().get_gcs_path()\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nIMAGE_SIZE = [1024, 1024]\nIMAGE_RESIZE = [380, 380]\nEPOCHS = 50","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.075543Z","iopub.status.busy":"2023-09-04T22:37:34.075259Z","iopub.status.idle":"2023-09-04T22:37:34.080942Z","shell.execute_reply":"2023-09-04T22:37:34.080162Z"},"papermill":{"duration":0.026069,"end_time":"2023-09-04T22:37:34.083102","exception":false,"start_time":"2023-09-04T22:37:34.057033","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🔨 Prepare Dataset","metadata":{"papermill":{"duration":0.017006,"end_time":"2023-09-04T22:37:34.117563","exception":false,"start_time":"2023-09-04T22:37:34.100557","status":"completed"},"tags":[]}},{"cell_type":"code","source":"TRAINING_FILENAMES, VALID_FILENAMES = train_test_split(\n    tf.io.gfile.glob(GCS_PATH + '/tfrecords/train*.tfrec'),\n    test_size=0.1, random_state=5\n)\nTEST_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/tfrecords/test*.tfrec')\nprint('Train TFRecord Files:', len(TRAINING_FILENAMES))\nprint('Validation TFRecord Files:', len(VALID_FILENAMES))\nprint('Test TFRecord Files:', len(TEST_FILENAMES))","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.153948Z","iopub.status.busy":"2023-09-04T22:37:34.153208Z","iopub.status.idle":"2023-09-04T22:37:34.182743Z","shell.execute_reply":"2023-09-04T22:37:34.181799Z"},"papermill":{"duration":0.050332,"end_time":"2023-09-04T22:37:34.185032","exception":false,"start_time":"2023-09-04T22:37:34.134700","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image(image):\n    image = tf.image.decode_jpeg(image, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    return image","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.223312Z","iopub.status.busy":"2023-09-04T22:37:34.223050Z","iopub.status.idle":"2023-09-04T22:37:34.227977Z","shell.execute_reply":"2023-09-04T22:37:34.227082Z"},"papermill":{"duration":0.025717,"end_time":"2023-09-04T22:37:34.229984","exception":false,"start_time":"2023-09-04T22:37:34.204267","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def read_tfrecord(example, labeled):\n    tfrecord_format = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"target\": tf.io.FixedLenFeature([], tf.int64)\n    } if labeled else {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"image_name\": tf.io.FixedLenFeature([], tf.string)\n    }\n    example = tf.io.parse_single_example(example, tfrecord_format)\n    image = decode_image(example['image'])\n    if labeled:\n        label = tf.cast(example['target'], tf.int32)\n        return image, label\n    idnum = example['image_name']\n    return image, idnum","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.265676Z","iopub.status.busy":"2023-09-04T22:37:34.265398Z","iopub.status.idle":"2023-09-04T22:37:34.272864Z","shell.execute_reply":"2023-09-04T22:37:34.272099Z"},"papermill":{"duration":0.027915,"end_time":"2023-09-04T22:37:34.274926","exception":false,"start_time":"2023-09-04T22:37:34.247011","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False # disable order, increase speed\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTOTUNE) # automatically interleaves reads from multiple files\n    dataset = dataset.with_options(ignore_order) # uses data as soon as it streams in, rather than in its original order\n    dataset = dataset.map(partial(read_tfrecord, labeled=labeled), num_parallel_calls=AUTOTUNE)\n    # returns a dataset of (image, label) pairs if labeled=True or (image, id) pairs if labeled=False\n    return dataset","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.310946Z","iopub.status.busy":"2023-09-04T22:37:34.310669Z","iopub.status.idle":"2023-09-04T22:37:34.316596Z","shell.execute_reply":"2023-09-04T22:37:34.315813Z"},"papermill":{"duration":0.026467,"end_time":"2023-09-04T22:37:34.318790","exception":false,"start_time":"2023-09-04T22:37:34.292323","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data augmentation","metadata":{"papermill":{"duration":0.017266,"end_time":"2023-09-04T22:37:34.353170","exception":false,"start_time":"2023-09-04T22:37:34.335904","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def get_mat(rotation, shear, height_zoom, width_zoom, height_shift, width_shift):\n    # returns 3x3 transformmatrix which transforms indicies\n        \n    # CONVERT DEGREES TO RADIANS\n    rotation = math.pi * rotation / 180.\n    shear    = math.pi * shear    / 180.\n\n    def get_3x3_mat(lst):\n        return tf.reshape(tf.concat([lst],axis=0), [3,3])\n    \n    # ROTATION MATRIX\n    c1   = tf.math.cos(rotation)\n    s1   = tf.math.sin(rotation)\n    one  = tf.constant([1],dtype='float32')\n    zero = tf.constant([0],dtype='float32')\n    \n    rotation_matrix = get_3x3_mat([c1,   s1,   zero, \n                                   -s1,  c1,   zero, \n                                   zero, zero, one])    \n    # SHEAR MATRIX\n    c2 = tf.math.cos(shear)\n    s2 = tf.math.sin(shear)    \n    \n    shear_matrix = get_3x3_mat([one,  s2,   zero, \n                                zero, c2,   zero, \n                                zero, zero, one])        \n    # ZOOM MATRIX\n    zoom_matrix = get_3x3_mat([one/height_zoom, zero,           zero, \n                               zero,            one/width_zoom, zero, \n                               zero,            zero,           one])    \n    # SHIFT MATRIX\n    shift_matrix = get_3x3_mat([one,  zero, height_shift, \n                                zero, one,  width_shift, \n                                zero, zero, one])\n    \n    return K.dot(K.dot(rotation_matrix, shear_matrix), \n                 K.dot(zoom_matrix,     shift_matrix))","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.388899Z","iopub.status.busy":"2023-09-04T22:37:34.388581Z","iopub.status.idle":"2023-09-04T22:37:34.399639Z","shell.execute_reply":"2023-09-04T22:37:34.398832Z"},"papermill":{"duration":0.031208,"end_time":"2023-09-04T22:37:34.401683","exception":false,"start_time":"2023-09-04T22:37:34.370475","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def transform(image,label):\n    # input image - is one image of size [dim,dim,3] not a batch of [b,dim,dim,3]\n    # output - image randomly rotated, sheared, zoomed, and shifted\n    DIM = IMAGE_SIZE[0]\n    XDIM = DIM%2 #fix for size 331\n    \n    rot = 180. * tf.random.normal([1],dtype='float32')\n    shr = 2. * tf.random.normal([1],dtype='float32') \n    h_zoom = 1.0 + tf.random.normal([1],dtype='float32')/8.\n    w_zoom = 1.0 + tf.random.normal([1],dtype='float32')/8.\n    h_shift = 8. * tf.random.normal([1],dtype='float32') \n    w_shift = 8. * tf.random.normal([1],dtype='float32') \n  \n    # GET TRANSFORMATION MATRIX\n    m = get_mat(rot,shr,h_zoom,w_zoom,h_shift,w_shift) \n\n    # LIST DESTINATION PIXEL INDICES\n    x = tf.repeat( tf.range(DIM//2,-DIM//2,-1), DIM )\n    y = tf.tile( tf.range(-DIM//2,DIM//2),[DIM] )\n    z = tf.ones([DIM*DIM],dtype='int32')\n    idx = tf.stack( [x,y,z] )\n    \n    # ROTATE DESTINATION PIXELS ONTO ORIGIN PIXELS\n    idx2 = K.dot(m,tf.cast(idx,dtype='float32'))\n    idx2 = K.cast(idx2,dtype='int32')\n    idx2 = K.clip(idx2,-DIM//2+XDIM+1,DIM//2)\n    \n    # FIND ORIGIN PIXEL VALUES           \n    idx3 = tf.stack( [DIM//2-idx2[0,], DIM//2-1+idx2[1,]] )\n    d = tf.gather_nd(image,tf.transpose(idx3))\n        \n    return tf.reshape(d,[DIM,DIM,3]),label","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.438131Z","iopub.status.busy":"2023-09-04T22:37:34.437814Z","iopub.status.idle":"2023-09-04T22:37:34.450378Z","shell.execute_reply":"2023-09-04T22:37:34.449604Z"},"papermill":{"duration":0.033158,"end_time":"2023-09-04T22:37:34.452424","exception":false,"start_time":"2023-09-04T22:37:34.419266","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def augmentation_pipeline(image, label):\n    image,_ = transform(image,label)\n    image = tf.image.random_flip_left_right(image)\n    image = tf.image.random_hue(image, 0.01)\n    image = tf.image.random_saturation(image, 0.7, 1.3)\n    image = tf.image.random_contrast(image, 0.8, 1.2)\n    image = tf.image.random_brightness(image, 0.1)\n    image = tf.image.resize(image, IMAGE_RESIZE)\n    return image, label","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.488282Z","iopub.status.busy":"2023-09-04T22:37:34.487976Z","iopub.status.idle":"2023-09-04T22:37:34.494101Z","shell.execute_reply":"2023-09-04T22:37:34.493339Z"},"papermill":{"duration":0.026153,"end_time":"2023-09-04T22:37:34.496070","exception":false,"start_time":"2023-09-04T22:37:34.469917","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def reshape_pipeline(image, label):\n    image = tf.image.resize(image, IMAGE_RESIZE)\n    return image, label","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.531965Z","iopub.status.busy":"2023-09-04T22:37:34.531693Z","iopub.status.idle":"2023-09-04T22:37:34.535757Z","shell.execute_reply":"2023-09-04T22:37:34.534994Z"},"papermill":{"duration":0.024507,"end_time":"2023-09-04T22:37:34.537846","exception":false,"start_time":"2023-09-04T22:37:34.513339","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Define loading methods","metadata":{"papermill":{"duration":0.017288,"end_time":"2023-09-04T22:37:34.572477","exception":false,"start_time":"2023-09-04T22:37:34.555189","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\n    dataset = dataset.map(augmentation_pipeline, num_parallel_calls=AUTOTUNE)\n    dataset = dataset.repeat()\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTOTUNE)\n    return dataset","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.608245Z","iopub.status.busy":"2023-09-04T22:37:34.607938Z","iopub.status.idle":"2023-09-04T22:37:34.613172Z","shell.execute_reply":"2023-09-04T22:37:34.612426Z"},"papermill":{"duration":0.025374,"end_time":"2023-09-04T22:37:34.615248","exception":false,"start_time":"2023-09-04T22:37:34.589874","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_validation_dataset(ordered=False):\n    dataset = load_dataset(VALID_FILENAMES, labeled=True, ordered=ordered)\n    dataset = dataset.map(reshape_pipeline, num_parallel_calls=AUTOTUNE)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTOTUNE)\n    return dataset","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.651178Z","iopub.status.busy":"2023-09-04T22:37:34.650919Z","iopub.status.idle":"2023-09-04T22:37:34.655966Z","shell.execute_reply":"2023-09-04T22:37:34.655161Z"},"papermill":{"duration":0.025696,"end_time":"2023-09-04T22:37:34.658236","exception":false,"start_time":"2023-09-04T22:37:34.632540","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_test_dataset(ordered=False):\n    dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=ordered)\n    dataset = dataset.map(reshape_pipeline, num_parallel_calls=AUTOTUNE)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTOTUNE)\n    return dataset","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.693971Z","iopub.status.busy":"2023-09-04T22:37:34.693703Z","iopub.status.idle":"2023-09-04T22:37:34.698758Z","shell.execute_reply":"2023-09-04T22:37:34.697900Z"},"papermill":{"duration":0.025494,"end_time":"2023-09-04T22:37:34.700785","exception":false,"start_time":"2023-09-04T22:37:34.675291","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.736369Z","iopub.status.busy":"2023-09-04T22:37:34.736096Z","iopub.status.idle":"2023-09-04T22:37:34.740897Z","shell.execute_reply":"2023-09-04T22:37:34.740116Z"},"papermill":{"duration":0.025002,"end_time":"2023-09-04T22:37:34.742932","exception":false,"start_time":"2023-09-04T22:37:34.717930","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"NUM_TRAINING_IMAGES = count_data_items(TRAINING_FILENAMES)\nNUM_VALIDATION_IMAGES = count_data_items(VALID_FILENAMES)\nNUM_TEST_IMAGES = count_data_items(TEST_FILENAMES)\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\nprint(\n    'Dataset: {} training images, {} validation images, {} unlabeled test images'.format(\n        NUM_TRAINING_IMAGES, NUM_VALIDATION_IMAGES, NUM_TEST_IMAGES\n    )\n)","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.779391Z","iopub.status.busy":"2023-09-04T22:37:34.779115Z","iopub.status.idle":"2023-09-04T22:37:34.784637Z","shell.execute_reply":"2023-09-04T22:37:34.783883Z"},"papermill":{"duration":0.026091,"end_time":"2023-09-04T22:37:34.786807","exception":false,"start_time":"2023-09-04T22:37:34.760716","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')\ntest_csv = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.823077Z","iopub.status.busy":"2023-09-04T22:37:34.822803Z","iopub.status.idle":"2023-09-04T22:37:34.939804Z","shell.execute_reply":"2023-09-04T22:37:34.938818Z"},"papermill":{"duration":0.137852,"end_time":"2023-09-04T22:37:34.942198","exception":false,"start_time":"2023-09-04T22:37:34.804346","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"total_img = train_csv['target'].size\n\nmalignant = np.count_nonzero(train_csv['target'])\nbenign = total_img - malignant\n\nprint('Examples:\\n    Total: {}\\n    Positive: {} ({:.2f}% of total)\\n'.format(\n    total_img, malignant, 100 * malignant / total_img))","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:34.978724Z","iopub.status.busy":"2023-09-04T22:37:34.978383Z","iopub.status.idle":"2023-09-04T22:37:34.984549Z","shell.execute_reply":"2023-09-04T22:37:34.983704Z"},"papermill":{"duration":0.026793,"end_time":"2023-09-04T22:37:34.986729","exception":false,"start_time":"2023-09-04T22:37:34.959936","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = get_training_dataset()\nvalid_dataset = get_validation_dataset()","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:35.023297Z","iopub.status.busy":"2023-09-04T22:37:35.023031Z","iopub.status.idle":"2023-09-04T22:37:36.084340Z","shell.execute_reply":"2023-09-04T22:37:36.083373Z"},"papermill":{"duration":1.082688,"end_time":"2023-09-04T22:37:36.087040","exception":false,"start_time":"2023-09-04T22:37:35.004352","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def show_batch(ds):\n    plt.figure(figsize=(20,4))\n    for idx, data in enumerate(iter(ds)):\n        ax = plt.subplot(2,10,idx+1)\n        img, target = data\n        img = img.numpy()\n        plt.imshow(img)\n        if target.numpy():\n            plt.title(\"MALIGNANT\")\n        else:\n            plt.title(\"BENIGN\")\n        plt.axis(\"off\")","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:36.124177Z","iopub.status.busy":"2023-09-04T22:37:36.123892Z","iopub.status.idle":"2023-09-04T22:37:36.130440Z","shell.execute_reply":"2023-09-04T22:37:36.129613Z"},"papermill":{"duration":0.027509,"end_time":"2023-09-04T22:37:36.132522","exception":false,"start_time":"2023-09-04T22:37:36.105013","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#show_batch(train_dataset.unbatch().take(20))","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:36.168619Z","iopub.status.busy":"2023-09-04T22:37:36.168293Z","iopub.status.idle":"2023-09-04T22:37:36.172033Z","shell.execute_reply":"2023-09-04T22:37:36.171256Z"},"papermill":{"duration":0.023876,"end_time":"2023-09-04T22:37:36.173883","exception":false,"start_time":"2023-09-04T22:37:36.150007","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Examples WITH Melanoma')\nimgs = train_csv.loc[train_csv.target==1].sample(10).image_name.values\nplt.figure(figsize=(20,8))\nfor i,k in enumerate(imgs):\n    plt.subplot(2,10,i+1); plt.axis('off')\n    img = Image.open('../input/siim-isic-melanoma-classification/jpeg/train/%s.jpg'%k)\n    img = img.resize(IMAGE_RESIZE)\n    plt.imshow(img)\nplt.show()\n\nprint('\\n\\nExamples WITHOUT Melanoma')\nplt.figure(figsize=(20,8))\nimgs = train_csv.loc[train_csv.target==0].sample(10).image_name.values\nfor i,k in enumerate(imgs):\n    plt.subplot(2,10,i+1); plt.axis('off')\n    img = Image.open('../input/siim-isic-melanoma-classification/jpeg/train/%s.jpg'%k)\n    img = img.resize(IMAGE_RESIZE)\n    plt.imshow(img)\nplt.show()","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:36.210198Z","iopub.status.busy":"2023-09-04T22:37:36.209932Z","iopub.status.idle":"2023-09-04T22:37:42.059832Z","shell.execute_reply":"2023-09-04T22:37:42.058737Z"},"papermill":{"duration":5.872974,"end_time":"2023-09-04T22:37:42.064496","exception":false,"start_time":"2023-09-04T22:37:36.191522","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🔥 Build the Model","metadata":{"papermill":{"duration":0.025252,"end_time":"2023-09-04T22:37:42.115851","exception":false,"start_time":"2023-09-04T22:37:42.090599","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# def exponential_decay(lr0, s):\n#     def exponential_decay_fn(epoch):\n#         return lr0 * 0.1 **(epoch / s)\n#     return exponential_decay_fn\n\n# exponential_decay_fn = exponential_decay(0.01, 20)\n\n# lr_scheduler = tf.keras.callbacks.LearningRateScheduler(exponential_decay_fn)","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:42.168156Z","iopub.status.busy":"2023-09-04T22:37:42.167310Z","iopub.status.idle":"2023-09-04T22:37:42.171694Z","shell.execute_reply":"2023-09-04T22:37:42.170823Z"},"papermill":{"duration":0.032488,"end_time":"2023-09-04T22:37:42.173756","exception":false,"start_time":"2023-09-04T22:37:42.141268","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LR_START = 0.00001\nLR_MAX = 0.00005 * strategy.num_replicas_in_sync\nLR_MIN = 0.00001\nLR_RAMPUP_EPOCHS = 5\nLR_SUSTAIN_EPOCHS = 0\nLR_EXP_DECAY = .8\n\ndef lrfn(epoch):\n    if epoch < LR_RAMPUP_EPOCHS:\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        lr = LR_MAX\n    else:\n        lr = (LR_MAX - LR_MIN) * LR_EXP_DECAY**(epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS) + LR_MIN\n    return lr\n    \nlr_callback = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=True)\n\nrng = [i for i in range(EPOCHS)]\ny = [lrfn(x) for x in rng]\nplt.plot(rng, y)\nprint(\"Learning rate schedule: {:.3g} to {:.3g} to {:.3g}\".format(y[0], max(y), y[-1]))","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:42.225521Z","iopub.status.busy":"2023-09-04T22:37:42.225216Z","iopub.status.idle":"2023-09-04T22:37:42.492551Z","shell.execute_reply":"2023-09-04T22:37:42.491675Z"},"papermill":{"duration":0.295611,"end_time":"2023-09-04T22:37:42.494635","exception":false,"start_time":"2023-09-04T22:37:42.199024","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_model(output_bias = None, metrics = None):    \n    if output_bias is not None:\n        output_bias = tf.keras.initializers.Constant(output_bias)\n        \n    base_model = tf.keras.applications.DenseNet201(input_shape=(*IMAGE_RESIZE, 3),\n                                                   include_top=False,\n                                                   weights='imagenet',\n                                                   pooling='avg')\n    \n    base_model.trainable = True\n    \n    model = keras.Sequential([\n        base_model,\n        layers.Dense(1, activation='sigmoid',\n                              bias_initializer=output_bias)\n    ])\n    \n    model.compile(optimizer='adam',\n                  loss=BinaryCrossentropy(label_smoothing=0.05),\n                  metrics=metrics)\n    \n    return model","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:42.549367Z","iopub.status.busy":"2023-09-04T22:37:42.548697Z","iopub.status.idle":"2023-09-04T22:37:42.556262Z","shell.execute_reply":"2023-09-04T22:37:42.555363Z"},"papermill":{"duration":0.037054,"end_time":"2023-09-04T22:37:42.558367","exception":false,"start_time":"2023-09-04T22:37:42.521313","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\nVALID_STEPS = NUM_VALIDATION_IMAGES // BATCH_SIZE","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:42.612384Z","iopub.status.busy":"2023-09-04T22:37:42.612103Z","iopub.status.idle":"2023-09-04T22:37:42.616248Z","shell.execute_reply":"2023-09-04T22:37:42.615319Z"},"papermill":{"duration":0.033367,"end_time":"2023-09-04T22:37:42.618479","exception":false,"start_time":"2023-09-04T22:37:42.585112","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"initial_bias = np.log([malignant/benign])\ninitial_bias","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:42.671625Z","iopub.status.busy":"2023-09-04T22:37:42.671166Z","iopub.status.idle":"2023-09-04T22:37:42.677527Z","shell.execute_reply":"2023-09-04T22:37:42.676417Z"},"papermill":{"duration":0.03555,"end_time":"2023-09-04T22:37:42.679870","exception":false,"start_time":"2023-09-04T22:37:42.644320","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"weight_for_0 = (1 / benign)*(total_img)/2.0 \nweight_for_1 = (1 / malignant)*(total_img)/2.0\n\nclass_weight = {0: weight_for_0, 1: weight_for_1}\n\nprint('Weight for class 0: {:.2f}'.format(weight_for_0))\nprint('Weight for class 1: {:.2f}'.format(weight_for_1))","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:42.734148Z","iopub.status.busy":"2023-09-04T22:37:42.733433Z","iopub.status.idle":"2023-09-04T22:37:42.740185Z","shell.execute_reply":"2023-09-04T22:37:42.739158Z"},"papermill":{"duration":0.036396,"end_time":"2023-09-04T22:37:42.742455","exception":false,"start_time":"2023-09-04T22:37:42.706059","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model = make_model(output_bias = initial_bias, metrics=tf.keras.metrics.AUC(name='auc'))","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:37:42.800285Z","iopub.status.busy":"2023-09-04T22:37:42.799406Z","iopub.status.idle":"2023-09-04T22:38:48.732619Z","shell.execute_reply":"2023-09-04T22:38:48.731437Z"},"papermill":{"duration":65.965459,"end_time":"2023-09-04T22:38:48.735664","exception":false,"start_time":"2023-09-04T22:37:42.770205","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint_cb = ModelCheckpoint(\"melanoma_model.h5\",save_best_only=True)\nearly_stopping_cb = EarlyStopping(patience=10,restore_best_weights=True)","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:38:48.792391Z","iopub.status.busy":"2023-09-04T22:38:48.792064Z","iopub.status.idle":"2023-09-04T22:38:48.796969Z","shell.execute_reply":"2023-09-04T22:38:48.796082Z"},"papermill":{"duration":0.035361,"end_time":"2023-09-04T22:38:48.799128","exception":false,"start_time":"2023-09-04T22:38:48.763767","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 🚅 Train","metadata":{"papermill":{"duration":0.026294,"end_time":"2023-09-04T22:38:48.852194","exception":false,"start_time":"2023-09-04T22:38:48.825900","status":"completed"},"tags":[]}},{"cell_type":"code","source":"history = model.fit(\n    train_dataset, epochs=EPOCHS,\n    steps_per_epoch=STEPS_PER_EPOCH,\n    validation_data=valid_dataset,\n    validation_steps=VALID_STEPS,\n    callbacks=[checkpoint_cb, early_stopping_cb, lr_callback],\n    class_weight=class_weight\n)","metadata":{"execution":{"iopub.execute_input":"2023-09-04T22:38:48.906644Z","iopub.status.busy":"2023-09-04T22:38:48.906330Z","iopub.status.idle":"2023-09-04T23:16:00.071402Z","shell.execute_reply":"2023-09-04T23:16:00.069949Z"},"papermill":{"duration":2231.453553,"end_time":"2023-09-04T23:16:00.332303","exception":false,"start_time":"2023-09-04T22:38:48.878750","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📈 Evaluation","metadata":{"papermill":{"duration":0.25967,"end_time":"2023-09-04T23:16:00.850810","exception":false,"start_time":"2023-09-04T23:16:00.591140","status":"completed"},"tags":[]}},{"cell_type":"code","source":"def plot_metrics(history):\n  metrics = ['loss', 'auc']\n  for n, metric in enumerate(metrics):\n    name = metric.replace(\"_\",\" \").capitalize()\n    plt.subplot(2,2,n+1)\n    plt.plot(history.epoch, history.history[metric], color=colors[0], label='Train')\n    plt.plot(history.epoch, history.history['val_'+metric],\n             color=colors[0], linestyle=\"--\", label='Val')\n    plt.xlabel('Epoch')\n    plt.ylabel(name)\n    plt.legend()","metadata":{"execution":{"iopub.execute_input":"2023-09-04T23:16:01.373169Z","iopub.status.busy":"2023-09-04T23:16:01.372729Z","iopub.status.idle":"2023-09-04T23:16:01.381858Z","shell.execute_reply":"2023-09-04T23:16:01.380700Z"},"papermill":{"duration":0.272901,"end_time":"2023-09-04T23:16:01.384034","exception":false,"start_time":"2023-09-04T23:16:01.111133","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_metrics(history)","metadata":{"execution":{"iopub.execute_input":"2023-09-04T23:16:01.942049Z","iopub.status.busy":"2023-09-04T23:16:01.941535Z","iopub.status.idle":"2023-09-04T23:16:08.216170Z","shell.execute_reply":"2023-09-04T23:16:08.214419Z"},"papermill":{"duration":6.567234,"end_time":"2023-09-04T23:16:08.219033","exception":false,"start_time":"2023-09-04T23:16:01.651799","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📝 Prediction","metadata":{"papermill":{"duration":0.279709,"end_time":"2023-09-04T23:16:08.778587","exception":false,"start_time":"2023-09-04T23:16:08.498878","status":"completed"},"tags":[]}},{"cell_type":"code","source":"test_ds = get_test_dataset(ordered=True)\n\nprint('Computing predictions...')\ntest_images_ds = test_ds.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images_ds)","metadata":{"execution":{"iopub.execute_input":"2023-09-04T23:16:09.397248Z","iopub.status.busy":"2023-09-04T23:16:09.396114Z","iopub.status.idle":"2023-09-04T23:17:21.230762Z","shell.execute_reply":"2023-09-04T23:17:21.229054Z"},"papermill":{"duration":72.11305,"end_time":"2023-09-04T23:17:21.233503","exception":false,"start_time":"2023-09-04T23:16:09.120453","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# 📃 Submission file","metadata":{"papermill":{"duration":0.296265,"end_time":"2023-09-04T23:17:21.808275","exception":false,"start_time":"2023-09-04T23:17:21.512010","status":"completed"},"tags":[]}},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/siim-isic-melanoma-classification/sample_submission.csv')\nsub.head()","metadata":{"execution":{"iopub.execute_input":"2023-09-04T23:17:22.394368Z","iopub.status.busy":"2023-09-04T23:17:22.393127Z","iopub.status.idle":"2023-09-04T23:17:22.429744Z","shell.execute_reply":"2023-09-04T23:17:22.428359Z"},"papermill":{"duration":0.318228,"end_time":"2023-09-04T23:17:22.432235","exception":false,"start_time":"2023-09-04T23:17:22.114007","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('Generating submission.csv file...')\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(NUM_TEST_IMAGES))).numpy().astype('U')","metadata":{"execution":{"iopub.execute_input":"2023-09-04T23:17:22.969014Z","iopub.status.busy":"2023-09-04T23:17:22.968048Z","iopub.status.idle":"2023-09-04T23:17:32.001177Z","shell.execute_reply":"2023-09-04T23:17:31.999541Z"},"papermill":{"duration":9.303451,"end_time":"2023-09-04T23:17:32.004113","exception":false,"start_time":"2023-09-04T23:17:22.700662","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_df = pd.DataFrame({'image_name': test_ids, 'target': np.concatenate(probabilities)})\npred_df.head()","metadata":{"execution":{"iopub.execute_input":"2023-09-04T23:17:32.580420Z","iopub.status.busy":"2023-09-04T23:17:32.579946Z","iopub.status.idle":"2023-09-04T23:17:32.610437Z","shell.execute_reply":"2023-09-04T23:17:32.608993Z"},"papermill":{"duration":0.3393,"end_time":"2023-09-04T23:17:32.612711","exception":false,"start_time":"2023-09-04T23:17:32.273411","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"del sub['target']\nsub = sub.merge(pred_df, on='image_name')\nsub.to_csv('submission.csv', index=False)\nsub.head()","metadata":{"execution":{"iopub.execute_input":"2023-09-04T23:17:33.141810Z","iopub.status.busy":"2023-09-04T23:17:33.140663Z","iopub.status.idle":"2023-09-04T23:17:33.196488Z","shell.execute_reply":"2023-09-04T23:17:33.195173Z"},"papermill":{"duration":0.322924,"end_time":"2023-09-04T23:17:33.198800","exception":false,"start_time":"2023-09-04T23:17:32.875876","status":"completed"},"tags":[]},"execution_count":null,"outputs":[]}]}