{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":193224260,"sourceType":"kernelVersion"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install -q efficientnet==1.1.1","metadata":{"_uuid":"ac7d3b8f-556d-4f70-80c8-f57cd7a1abe3","_cell_guid":"69e493a7-666d-441f-9caf-20dbaaa45b77","execution":{"iopub.status.busy":"2024-08-30T05:44:56.663148Z","iopub.execute_input":"2024-08-30T05:44:56.663751Z","iopub.status.idle":"2024-08-30T05:45:00.996836Z","shell.execute_reply.started":"2024-08-30T05:44:56.663719Z","shell.execute_reply":"2024-08-30T05:45:00.995802Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\n# import supporting_functions as sf\nimport tensorflow.keras.layers as L\nimport efficientnet.tfkeras as efn\nimport numpy as np\nfrom sklearn.model_selection import KFold\nfrom kaggle_datasets import KaggleDatasets\nimport pathlib\n\nGCS_PATH = '/kaggle/input/isic-2024-generate-dataset'\n# TRAINING_FILENAMES = np.sort(np.array(tf.io.gfile.glob(GCS_PATH + '/tfrecords/train*.tfrec')))\n\n\nBATCH_SIZE = 32 * 8 # kind of a hack. We should have access to 8 TPUs\nIMAGE_SIZE = [224, 224]\n\nEPOCHS = 50\nFOLDS = 4\n\ndata_dir = pathlib.Path(GCS_PATH + '/lesions').with_suffix('')\n\nprint(IMAGE_SIZE)\nprint(BATCH_SIZE)\nprint(EPOCHS)\n\ndef efiNet_model():\n    model = efn.EfficientNetB3(input_shape=(*IMAGE_SIZE, 3), weights='imagenet', include_top=False)\n    return model","metadata":{"execution":{"iopub.status.busy":"2024-08-30T05:45:00.998464Z","iopub.execute_input":"2024-08-30T05:45:00.998720Z","iopub.status.idle":"2024-08-30T05:45:16.302784Z","shell.execute_reply.started":"2024-08-30T05:45:00.998694Z","shell.execute_reply":"2024-08-30T05:45:16.302027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"_uuid":"f054404d-049d-42ef-9cc5-43feb8f35cff","_cell_guid":"5f9c740e-3d4e-41ee-bdb8-e8075b2aa456","execution":{"iopub.status.busy":"2024-08-30T05:45:16.303754Z","iopub.execute_input":"2024-08-30T05:45:16.304178Z","iopub.status.idle":"2024-08-30T05:45:24.699721Z","shell.execute_reply.started":"2024-08-30T05:45:16.304150Z","shell.execute_reply":"2024-08-30T05:45:24.698899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data_augmentation = tf.keras.Sequential([\n    L.RandomRotation(0.25),\n    L.RandomFlip(\"horizontal_and_vertical\"),\n    L.RandomTranslation(0.1, 0.1),\n    L.RandomZoom(0.15),\n    L.RandomRotation(0.15),\n    L.RandomBrightness(0.1),\n    L.RandomContrast(0.1),\n])","metadata":{"execution":{"iopub.status.busy":"2024-08-30T05:45:24.701444Z","iopub.execute_input":"2024-08-30T05:45:24.701766Z","iopub.status.idle":"2024-08-30T05:45:24.724110Z","shell.execute_reply.started":"2024-08-30T05:45:24.701741Z","shell.execute_reply":"2024-08-30T05:45:24.723400Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_dataset(batch_size):\n    random = tf.random.Generator.from_seed(114514)\n    \n    dataset = tf.keras.utils.image_dataset_from_directory(\n        data_dir,\n        label_mode='binary',\n        class_names=['benign', 'malignant'],\n        validation_split=0.2,\n        subset='training',\n        seed=114514,\n        image_size=IMAGE_SIZE,\n        batch_size=batch_size)\n\n    val_dataset = tf.keras.utils.image_dataset_from_directory(\n        data_dir,\n        label_mode='binary',\n        class_names=['benign', 'malignant'],\n        validation_split=0.2,\n        subset='validation',\n        seed=114514,\n        image_size=IMAGE_SIZE,\n        batch_size=batch_size)\n\n\n    def scale(image, label):\n        image = tf.cast(image, tf.float32)\n#        image /= 255.0\n        return image, label;\n\n    def augmentation(image, label):\n        # Data Augmentation\n        if random.uniform([]) < 0.5:\n            image = data_augmentation(image, training=True)\n        return image, label;\n    \n    AUTOTUNE = tf.data.AUTOTUNE\n    \n    dataset = dataset.map(scale, num_parallel_calls=AUTOTUNE)\n    val_dataset = val_dataset.map(scale, num_parallel_calls=AUTOTUNE)\n\n    # For optimization\n    dataset = dataset.cache()\n    val_dataset = val_dataset.cache()\n\n    # Processing after cacheing\n    dataset = dataset.shuffle(2048, seed=114514)\n#    dataset = dataset.rebatch(batch_size)\n    dataset = dataset.map(augmentation, num_parallel_calls=AUTOTUNE)\n\n    return dataset, val_dataset;","metadata":{"execution":{"iopub.status.busy":"2024-08-30T05:50:00.628317Z","iopub.execute_input":"2024-08-30T05:50:00.628752Z","iopub.status.idle":"2024-08-30T05:50:00.637845Z","shell.execute_reply.started":"2024-08-30T05:50:00.628717Z","shell.execute_reply":"2024-08-30T05:50:00.636719Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining the optimizer and compiling the model\nwith strategy.scope():\n    model = tf.keras.Sequential([\n        efiNet_model(),\n        L.GlobalAveragePooling2D(),\n        #L.Dense(1024, activation = 'relu'),\n        #L.Dropout(0.3) , \n        L.Dense(512, activation= 'relu'), \n        L.Dropout(0.25), \n        #L.Dense(256, activation='relu'), \n        #L.Dropout(0.2), \n        #L.Dense(128, activation='relu'), \n        #L.Dropout(0.15), \n        L.Dense(1, activation='sigmoid')\n    ])\n\n    scheduler = tf.keras.optimizers.schedules.CosineDecay(\n        initial_learning_rate=1e-4,\n        decay_steps=30000 // BATCH_SIZE * 500,\n        alpha=0.01\n    )\n    model.compile(\n        optimizer=tf.keras.optimizers.Adam(scheduler, weight_decay=1e-6),\n        loss = tf.keras.losses.BinaryCrossentropy(),\n        metrics=[tf.keras.metrics.AUC()]\n    )","metadata":{"_uuid":"c5831631-894c-4d95-a05a-c2d79ef21d2a","_cell_guid":"7b73d475-7dbe-4d5b-b68f-cc81d1c9cd13","execution":{"iopub.status.busy":"2024-08-30T05:45:24.733706Z","iopub.execute_input":"2024-08-30T05:45:24.733952Z","iopub.status.idle":"2024-08-30T05:45:49.655146Z","shell.execute_reply.started":"2024-08-30T05:45:24.733927Z","shell.execute_reply":"2024-08-30T05:45:49.654016Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#f = open(\"./model_summary.txt\", \"w\")\n\n#model.summary(print_fn=lambda x: f.write(x + '\\n'))\n#f.close()\n\n#model.get_layer(index=0).summary()","metadata":{"execution":{"iopub.status.busy":"2024-08-30T05:45:49.656361Z","iopub.execute_input":"2024-08-30T05:45:49.656639Z","iopub.status.idle":"2024-08-30T05:45:49.660504Z","shell.execute_reply.started":"2024-08-30T05:45:49.656612Z","shell.execute_reply":"2024-08-30T05:45:49.659627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# skf = KFold(n_splits=FOLDS,shuffle=True,random_state=42)\n# for fold,(idxT,idxV) in enumerate(skf.split(np.arange(16))):\n#     \n#     # CREATE TRAIN AND VALIDATION SUBSETS\n#     files_train = tf.io.gfile.glob([GCS_PATH + '/tfrecords/train%.2i*.tfrec'%x for x in idxT])\n#     \n#     files_valid = tf.io.gfile.glob([GCS_PATH + '/tfrecords/train%.2i*.tfrec'%x for x in idxV])\n#     \n# \n# # hardcoded cuz I'm a hack\n# NUM_VAL_IMAGES = 33126 * (1/FOLDS)\n# NUM_TRAINING_IMAGES = 33126 - NUM_VAL_IMAGES\n# \n# \n# TRAIN_STEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\n# print('Dataset: {} training images'.format(NUM_TRAINING_IMAGES))\n\ntrain_dataset, val_dataset = get_dataset(BATCH_SIZE)","metadata":{"_kg_hide-output":false,"execution":{"iopub.status.busy":"2024-08-30T05:50:06.489732Z","iopub.execute_input":"2024-08-30T05:50:06.490204Z","iopub.status.idle":"2024-08-30T05:50:18.153333Z","shell.execute_reply.started":"2024-08-30T05:50:06.490165Z","shell.execute_reply":"2024-08-30T05:50:18.151945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Optimization\ntrain_dataset = train_dataset.prefetch(tf.data.AUTOTUNE)\nval_dataset = val_dataset.prefetch(tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2024-08-30T05:50:18.155164Z","iopub.execute_input":"2024-08-30T05:50:18.155466Z","iopub.status.idle":"2024-08-30T05:50:18.161702Z","shell.execute_reply.started":"2024-08-30T05:50:18.155438Z","shell.execute_reply":"2024-08-30T05:50:18.160778Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Fitting the model\nhistory = model.fit(\n    train_dataset,\n    epochs=EPOCHS,\n    validation_data=val_dataset\n)","metadata":{"_uuid":"8aba4f30-c5d8-4585-9af2-5db82d11b8a3","_cell_guid":"4fac61a9-34ff-4492-8180-edaa0bfde2b3","execution":{"iopub.status.busy":"2024-08-30T05:50:18.162766Z","iopub.execute_input":"2024-08-30T05:50:18.163038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"save_locally = tf.saved_model.SaveOptions(experimental_io_device='/job:localhost')\ntf.saved_model.save(model, 'model', options=save_locally)","metadata":{"execution":{"iopub.status.busy":"2024-08-30T05:49:33.669227Z","iopub.status.idle":"2024-08-30T05:49:33.669584Z","shell.execute_reply.started":"2024-08-30T05:49:33.669397Z","shell.execute_reply":"2024-08-30T05:49:33.669412Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#test_predictions = model.predict(sf.get_testing_dataset(), steps=TEST_STEPS, use_multiprocessing=True)\n#print(test_predictions)","metadata":{"execution":{"iopub.status.busy":"2024-08-30T05:49:33.671121Z","iopub.status.idle":"2024-08-30T05:49:33.671442Z","shell.execute_reply.started":"2024-08-30T05:49:33.671287Z","shell.execute_reply":"2024-08-30T05:49:33.671302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#GCS_PATH = KaggleDatasets().get_gcs_path('siim-isic-melanoma-classification')\n#TEST_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/tfrecords/test.tfrec')\n\n#for example in tf.python_io.tf_record_iterator(\"../input/siim-isic-melanoma-classification/tfrecords/train00-2071.tfrec\"):\n#    print(tf.train.Example.FromString(example))","metadata":{"execution":{"iopub.status.busy":"2024-08-30T05:49:33.672201Z","iopub.status.idle":"2024-08-30T05:49:33.672488Z","shell.execute_reply.started":"2024-08-30T05:49:33.672344Z","shell.execute_reply":"2024-08-30T05:49:33.672358Z"},"trusted":true},"execution_count":null,"outputs":[]}]}