{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"from functools import partial\nfrom glob import glob\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport matplotlib.pyplot as plt\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.model_selection import StratifiedKFold, train_test_split","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE\n# tf.config.experimental_run_functions_eagerly(True)\n\nGCS_PATH = KaggleDatasets().get_gcs_path()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy() # default distribution strategy in Tensorflow. Works on CPU and single GPU.\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(\"Num GPUs Available: \", len(tf.config.experimental.list_physical_devices('GPU')))","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# EDA","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv = pd.read_csv('../input/siim-isic-melanoma-classification/train.csv')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_csv","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"target_counts = train_csv.target.value_counts()\nprint(f\"0: {target_counts[0]} - {target_counts[0]*100/(target_counts[0]+target_counts[1]):.2f}% of total\")\nprint(f\"1: {target_counts[1]} - {target_counts[1]*100/(target_counts[0]+target_counts[1]):.2f}% of total\")\nprint(f\"Ratio: {target_counts[0] / target_counts[1]:.2f} : 1\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Random Over-sampling","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df_0 = train_csv[train_csv.target == 0]\ntrain_df_1 = train_csv[train_csv.target == 1]\ntrain_df_1_resampled = train_df_1.sample(target_counts[0], replace=True)\nprint(f\"Upsampled counts - \")\nprint(f\"0: {len(train_df_0)}\")\nprint(f\"1: {len(train_df_1_resampled)}\")","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Load data from TFRecords","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"markdown","source":"BATCH_SIZE = 16 * strategy.num_replicas_in_sync\ntrain_files = tf.io.gfile.glob(GCS_PATH + '/tfrecords/train*.tfrec')\ntest_files = tf.io.gfile.glob(GCS_PATH + '/tfrecords/test*.tfrec')","execution_count":null},{"metadata":{"trusted":true},"cell_type":"markdown","source":"%%time\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0  # convert image to floats in [0, 1] range\n    image = tf.reshape(image, [*IMG_SIZE, 3]) # explicit size needed for TPU\n    return image\n\ndef read_tfrecord(example, labeled):\n    tfrecord_format = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"image_name\": tf.io.FixedLenFeature([], tf.string),\n        \"target\": tf.io.FixedLenFeature([], tf.int64)\n    } if labeled else {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"image_name\": tf.io.FixedLenFeature([], tf.string)\n    }\n    example = tf.io.parse_single_example(example, tfrecord_format)\n    image = decode_image(example['image'])\n    if labeled:\n        label = tf.cast(example['target'], tf.int32)\n        return image, label\n    idnum = example['image_name']\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    # Read from TFRecords. For optimal performance, reading from multiple files at once and\n    # disregarding data order. Order does not matter since we will be shuffling the data anyway.\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False # disable order, increase speed\n\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO) # automatically interleaves reads from multiple files\n    dataset = dataset.with_options(ignore_order) # uses data as soon as it streams in, rather than in its original order\n    dataset = dataset.map(partial(read_tfrecord, labeled=labeled), num_parallel_calls=AUTO)\n    # returns a dataset of (image, label) pairs if labeled=True or (image, id) pairs if labeled=False\n    return dataset\n\ndef get_train_vald_dataset(vald_split=0.2, ordered=False):\n    dataset = load_dataset(train_files, labeled=True, ordered=ordered)\n    n = sum(1 for record in dataset)\n    n_vald = int(vald_split * n)\n    n_train = n - n_vald\n    #train_dataset = dataset.map(data_augment, num_parallel_calls=AUTO)\n    train_dataset = dataset.take(n_train)\n    train_dataset = train_dataset.repeat() # the training dataset must repeat for several epochs\n    train_dataset = train_dataset.shuffle(2048)\n    train_dataset = train_dataset.batch(BATCH_SIZE)\n    train_dataset = train_dataset.prefetch(AUTO) # prefetch next batch while training (autotune prefetch buffer size)\n    \n    vald_dataset = dataset.skip(n_train)\n    n1 = sum(1 for rec in vald_dataset)\n    vald_dataset = vald_dataset.batch(BATCH_SIZE)\n    vald_dataset = vald_dataset.cache()\n    vald_dataset = vald_dataset.prefetch(AUTO) # prefetch next batch while training (autotune prefetch buffer size)\n    if n_vald != n1:\n        print(\"Validation Dataset sizes - \", n_vald, n1)\n    return n_train, train_dataset, n1, vald_dataset\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(test_files, labeled=False, ordered=ordered)\n    n = sum(1 for record in dataset)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO) # prefetch next batch while training (autotune prefetch buffer size)\n    return n, dataset\n\nn_train, train_dataset, n_vald, vald_dataset = get_train_vald_dataset()\nn_test, test_dataset = get_test_dataset(True)\nprint(f'Dataset: {n_train} training images, {n_vald} validation images, {n_test} unlabeled test images')","execution_count":null},{"metadata":{},"cell_type":"markdown","source":"# Load data from JPEG","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"IMG_SIZE = [1024, 1024]\nBATCH_SIZE = 32","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"image_dir = \"../input/siim-isic-melanoma-classification/jpeg\"\ntrain_dir = \"train\"\ntest_dir = \"test/\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"balanced_train_df = pd.concat([train_df_0, train_df_1_resampled])\nbalanced_train_df.image_name = balanced_train_df.image_name + \".jpg\"\nbalanced_train_df.target = balanced_train_df.target.astype(str)\nbalanced_train_df","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"k = StratifiedKFold(n_splits=1, shuffle=True, random_state=0)\nfolds = k.split(balanced_train_df, y=balanced_train_df.target)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"train_df, vald_df = train_test_split(balanced_train_df, test_size=0.2, stratify=balanced_train_df.target, shuffle=True, random_state=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_datagen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255, \n                                 rotation_range=360,\n                                 horizontal_flip=True,\n                                 vertical_flip=True)\ntrain_dataset = train_datagen.flow_from_dataframe(\n    dataframe=train_df,\n    directory=\"../input/siim-isic-melanoma-classification/jpeg/train\",\n    x_col=\"image_name\",\n    y_col=\"target\",\n    class_mode=\"binary\",\n    batch_size=BATCH_SIZE,\n    target_size=IMG_SIZE,\n    seed=0)\nvald_datagen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255)\nvald_dataset = train_datagen.flow_from_dataframe(\n    dataframe=vald_df,\n    directory=\"../input/siim-isic-melanoma-classification/jpeg/train\",\n    x_col=\"image_name\",\n    y_col=\"target\",\n    class_mode=\"binary\",\n    batch_size=BATCH_SIZE,\n    target_size=IMG_SIZE,\n    seed=0)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"n_train, n_vald = len(train_dataset), len(vald_dataset)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"# Model","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"def initialize_model(model_name=\"\"):\n    #pretrained_model = tf.keras.applications.MobileNetV2(input_shape=[*IMAGE_SIZE, 3], include_top=False)\n    pretrained_model = tf.keras.applications.Xception(input_shape=[*IMG_SIZE, 3], include_top=False, weights='imagenet')\n    #pretrained_model = tf.keras.applications.VGG16(weights='imagenet', include_top=False ,input_shape=[*IMAGE_SIZE, 3])\n    #pretrained_model = tf.keras.applications.ResNet50(weights='imagenet', include_top=False, input_shape=[*IMAGE_SIZE, 3])\n    #pretrained_model = tf.keras.applications.MobileNet(weights='imagenet', include_top=False, input_shape=[*IMAGE_SIZE, 3])\n    # EfficientNet can be loaded through efficientnet.tfkeras library (https://github.com/qubvel/efficientnet)\n    #pretrained_model = efficientnet.tfkeras.EfficientNetB0(weights='imagenet', include_top=False)\n    \n    pretrained_model.trainable = False\n\n    model = tf.keras.Sequential([\n        pretrained_model,\n        tf.keras.layers.GlobalAveragePooling2D(),\n        #tf.keras.layers.Flatten(),\n        tf.keras.layers.Dense(8, activation='relu'),\n        tf.keras.layers.Dense(1, activation='sigmoid')\n    ])\n\n    model.compile(\n        optimizer='adam',\n        loss = 'binary_crossentropy',\n        metrics=['AUC']\n    )\n\n    return model","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"with strategy.scope():\n    model = initialize_model()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"TRAIN_STEPS = n_train // BATCH_SIZE\nVALD_STEPS = n_vald // BATCH_SIZE\nEPOCHS = 10\nprint(f\"Training steps = {TRAIN_STEPS}, Validation steps = {VALD_STEPS}\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"t = model.fit(train_dataset, epochs=EPOCHS, steps_per_epoch=TRAIN_STEPS,\n                    validation_data=vald_dataset, validation_steps=VALD_STEPS)#, callbacks=[lr_callback])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"plt.plot(t)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.save('../working/model.h5')","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_df = pd.read_csv('../input/siim-isic-melanoma-classification/test.csv')\ntest_df.image_name = test_df.image_name + \".jpg\"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_datagen = tf.keras.preprocessing.image.ImageDataGenerator(rescale=1./255)\ntest_dataset = test_datagen.flow_from_dataframe(\n    dataframe=test_df,\n    directory=\"../input/siim-isic-melanoma-classification/jpeg/test\",\n    x_col=\"image_name\",\n    y_col=None,\n    shuffle=False,\n    target_size=IMG_SIZE,\n    class_mode=None,\n    batch_size=32)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"outs = model.predict(test_dataset)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred = pd.DataFrame({'image_name': test_df['image_name'].str.rstrip(\".jpg\"), 'target': outs.ravel()})\npred","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"pred.to_csv('submissions.csv', header=True, index=False)","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}