{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.15","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport random\nimport os\nimport sys\nimport re\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential, Model\nfrom tensorflow.keras.layers import Dense, Dropout, BatchNormalization\nfrom tensorflow.keras.callbacks import EarlyStopping, ModelCheckpoint, ReduceLROnPlateau\nfrom tensorflow.keras import utils\nfrom tensorflow.random import set_seed\nfrom tensorflow.keras.preprocessing import image\nimport matplotlib.pyplot as plt\nfrom kaggle_datasets import KaggleDatasets\nimport logging\ntry:\n    from efficientnet.tfkeras import EfficientNetB7\nexcept:\n    !export PIP_ROOT_USER_ACTION=ignore\n    !pip install -q efficientnet\n    from efficientnet.tfkeras import EfficientNetB7\n\n# Function to set a consistent random seed\ndef seed_everything(seed):\n    np.random.seed(seed)\n    set_seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    os.environ['TF_DETERMINISTIC_OPS'] = '1'\n\n# Set random seed\nSEED = 40\nseed_everything(SEED)\n\nif __name__=='__main__':    \n    logger = logging.getLogger('-')\n    logger.setLevel(logging.INFO)\n    \n    # stream_handler\n    formatter_1 = logging.Formatter('%(asctime)s %(name)s %(levelname)s: %(message)s')\n    stream_handler = logging.StreamHandler()\n    stream_handler.setFormatter(formatter_1)\n    logger.addHandler(stream_handler)\n        \n    logger.propagate = 0","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-12-07T16:29:17.237352Z","iopub.execute_input":"2024-12-07T16:29:17.237581Z","iopub.status.idle":"2024-12-07T16:29:40.204122Z","shell.execute_reply.started":"2024-12-07T16:29:17.237557Z","shell.execute_reply":"2024-12-07T16:29:40.203269Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Display TensorFlow version\nlogger.info(f\"TensorFlow version:{tf.__version__}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T16:29:40.205781Z","iopub.execute_input":"2024-12-07T16:29:40.206572Z","iopub.status.idle":"2024-12-07T16:29:40.210867Z","shell.execute_reply.started":"2024-12-07T16:29:40.206537Z","shell.execute_reply":"2024-12-07T16:29:40.210265Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Accelerator\n##### Detect hardware and return the appropriate distribution strategy: TPU, GPU, or CPU","metadata":{}},{"cell_type":"code","source":"AUTO = tf.data.experimental.AUTOTUNE # optimizes data pipeline performance\n\ntry:\n    # Detect TPU. On Kaggle, TPU_NAME environment variable is automatically set.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU:', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    # Connect to the TPU cluster and initialize the TPU system\n    tf.config.experimental_connect_to_cluster(tpu);\n    tf.tpu.experimental.initialize_tpu_system(tpu);\n    # Use TPUStrategy for distributed training\n    strategy = tf.distribute.experimental.TPUStrategy(tpu);\nelse:\n    # Default strategy for TensorFlow: works on CPU and single GPU\n    strategy = tf.distribute.get_strategy()\n\nlogger.info(f\"REPLICAS:{strategy.num_replicas_in_sync}\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T16:29:40.211646Z","iopub.execute_input":"2024-12-07T16:29:40.211883Z","iopub.status.idle":"2024-12-07T16:29:48.935402Z","shell.execute_reply.started":"2024-12-07T16:29:40.211859Z","shell.execute_reply":"2024-12-07T16:29:48.934703Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Get Data Paths","metadata":{}},{"cell_type":"code","source":"# Get the path to the dataset\nGCS_DS_PATH = KaggleDatasets().get_gcs_path()\n\n# Set the image size for training\nIMAGE_SIZE = [512, 512]  # NOTE: at this size, a GPU may run out of memory. Use TPU for this size.\n\n# Set the batch size, scaling it based on the number of replicas in the strategy\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\n\n# Dictionary for selecting dataset paths based on image size\nGCS_PATH_SELECT = {\n    192: GCS_DS_PATH + '/tfrecords-jpeg-192x192',\n    224: GCS_DS_PATH + '/tfrecords-jpeg-224x224',\n    331: GCS_DS_PATH + '/tfrecords-jpeg-331x331',\n    512: GCS_DS_PATH + '/tfrecords-jpeg-512x512'\n}\n\n# Select the appropriate dataset path based on the chosen image size\nGCS_PATH = GCS_PATH_SELECT[IMAGE_SIZE[0]]\n\n# Get the filenames for training, validation, and testing datasets\nTRAINING_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/train/*.tfrec')\nVALIDATION_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/test/*.tfrec')\n\nlogger.info(\"Got data paths.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T16:29:48.937281Z","iopub.execute_input":"2024-12-07T16:29:48.937553Z","iopub.status.idle":"2024-12-07T16:29:48.986739Z","shell.execute_reply.started":"2024-12-07T16:29:48.937525Z","shell.execute_reply":"2024-12-07T16:29:48.986030Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Set some parameters","metadata":{}},{"cell_type":"code","source":"# Number of epochs for training\nEPOCHS = 25\n\n# Dataset statistics\nNUM_TRAINING_IMAGES = 12753  # Total number of training images\nNUM_TEST_IMAGES = 7382       # Total number of test images\n\n# Calculate the number of steps per epoch\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\n\nlogger.info(\"Parameters Initialized.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T16:29:48.987648Z","iopub.execute_input":"2024-12-07T16:29:48.987931Z","iopub.status.idle":"2024-12-07T16:29:48.992550Z","shell.execute_reply.started":"2024-12-07T16:29:48.987903Z","shell.execute_reply":"2024-12-07T16:29:48.991747Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def decode_image(image_data):\n    \"\"\"\n    Decodes an image into a tensor and normalizes it.\n    \"\"\"\n    image = tf.image.decode_jpeg(image_data, channels=3)  # Decode JPEG image to a uint8 tensor\n    image = tf.cast(image, tf.float32) / 255.0  # Normalize pixel values to [0, 1]\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])  # Explicit size required for TPU compatibility\n    return image\n\ndef data_augment(image, label):\n    \"\"\"\n    Applies random data augmentation to the image.\n    \"\"\"\n    image = tf.image.random_flip_left_right(image, seed=SEED)  # Random horizontal flip\n    return image, label   \n\ndef read_labeled_tfrecord(example):\n    \"\"\"\n    Reads a labeled TFRecord example and extracts the image and label.\n    \"\"\"\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),  # Byte string for the image\n        \"class\": tf.io.FixedLenFeature([], tf.int64),   # Single integer for the class\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)  # Parse the example\n    image = decode_image(example['image'])  # Decode the image\n    label = tf.cast(example['class'], tf.int32)  # Cast the label to int32\n    return image, label  # Return (image, label) pairs\n\ndef read_unlabeled_tfrecord(example):\n    \"\"\"\n    Reads an unlabeled TFRecord example and extracts the image and ID.\n    \"\"\"\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),  # Byte string for the image\n        \"id\": tf.io.FixedLenFeature([], tf.string),     # Single string for the ID\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)  # Parse the example\n    image = decode_image(example['image'])  # Decode the image\n    idnum = example['id']  # Extract the ID\n    return image, idnum  # Return (image, ID) pairs\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    \"\"\"\n    Reads data from TFRecords. For optimal performance, reads multiple files simultaneously (order not perserved)\n    Order doesn't matter as the data will be shuffled anyway.\n    \"\"\"\n    ignore_order = tf.data.Options()  # Represents options for tf.data.Dataset\n    if not ordered:\n        ignore_order.experimental_deterministic = False  # Disable order enforcement for speed\n    \n    dataset = tf.data.TFRecordDataset(filenames)  # Automatically reads from multiple files\n    dataset = dataset.with_options(ignore_order)  # Uses data as it becomes available\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord)\n    # Returns (image, label) pairs if labeled=True, or (image, ID) pairs if labeled=False\n    return dataset\n\ndef get_training_dataset():\n    \"\"\"\n    Prepares the training dataset by loading data, applying augmentations,\n    shuffling, and batching.\n    \"\"\"\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-512x512/train/*.tfrec'), labeled=True)\n    dataset = dataset.map(data_augment, num_parallel_calls=AUTO)  # Apply data augmentation\n    dataset = dataset.repeat()  # Repeat training dataset for multiple epochs\n    dataset = dataset.shuffle(2048)  # Shuffle with a buffer size of 2048\n    dataset = dataset.batch(BATCH_SIZE)  # Batch the data\n    return dataset\n\ndef get_validation_dataset():\n    \"\"\"\n    Prepares the validation dataset by loading data and batching it.\n    Caching is used to speed up subsequent iterations.\n    \"\"\"\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-512x512/val/*.tfrec'), labeled=True, ordered=False)\n    dataset = dataset.batch(BATCH_SIZE)  # Batch the data\n    dataset = dataset.cache()  # Cache the dataset for faster access\n    return dataset\n\ndef get_test_dataset(ordered=False):\n    \"\"\"\n    Prepares the test dataset by loading data and batching it.\n    \"\"\"\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-512x512/test/*.tfrec'), labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)  # Batch the data\n    return dataset\n\n# Create training and validation datasets\ntraining_dataset = get_training_dataset()\nvalidation_dataset = get_validation_dataset()\n\nlogger.info(\"Created training and validation datasets.\")","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T16:29:48.993677Z","iopub.execute_input":"2024-12-07T16:29:48.993961Z","iopub.status.idle":"2024-12-07T16:29:49.208428Z","shell.execute_reply.started":"2024-12-07T16:29:48.993933Z","shell.execute_reply":"2024-12-07T16:29:49.207743Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Function to manage changes in the learning rate during neural network training\nLR_START = 0.00001  # Initial learning rate\nLR_MAX = 0.00005 * strategy.num_replicas_in_sync  # Maximum learning rate (scaled by replicas)\nLR_MIN = 0.00001  # Minimum learning rate\nLR_RAMPUP_EPOCHS = 5  # Number of epochs to ramp up the learning rate\nLR_SUSTAIN_EPOCHS = 0  # Number of epochs to sustain the maximum learning rate\nLR_EXP_DECAY = 0.8  # Exponential decay factor after ramp-up and sustain phases\n\ndef lrfn(epoch):\n    \"\"\"\n    Learning rate schedule function.\n    Adjusts the learning rate based on the epoch number.\n    \"\"\"\n    if epoch < LR_RAMPUP_EPOCHS:\n        # Ramp up phase: Increase linearly from LR_START to LR_MAX\n        lr = (LR_MAX - LR_START) / LR_RAMPUP_EPOCHS * epoch + LR_START\n    elif epoch < LR_RAMPUP_EPOCHS + LR_SUSTAIN_EPOCHS:\n        # Sustain phase: Keep the learning rate at LR_MAX\n        lr = LR_MAX\n    else:\n        # Exponential decay phase\n        lr = (LR_MAX - LR_MIN) * LR_EXP_DECAY**(epoch - LR_RAMPUP_EPOCHS - LR_SUSTAIN_EPOCHS) + LR_MIN\n    return lr\n\n# Create a learning rate scheduler callback\nlr_callback = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=True)\n\n# Plot the learning rate schedule across epochs\nrng = [i for i in range(EPOCHS)]  # List of epoch numbers\ny = [lrfn(x) for x in rng]  # Corresponding learning rates\nplt.plot(rng, y)\nplt.title(\"Learning Rate Schedule\")\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Learning Rate\")\nplt.grid(True)\nplt.show()\n\n# Display key information about the learning rate schedule\nprint(\"Learning rate schedule: {:.3g} to {:.3g} to {:.3g}\".format(y[0], max(y), y[-1]))\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T16:29:49.209235Z","iopub.execute_input":"2024-12-07T16:29:49.209451Z","iopub.status.idle":"2024-12-07T16:29:49.444932Z","shell.execute_reply.started":"2024-12-07T16:29:49.209427Z","shell.execute_reply":"2024-12-07T16:29:49.444218Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def get_model(use_model):\n    \"\"\"\n    Creates a model using a specified base model.\n    Adds a classification head with 104 output classes and softmax activation.\n    \n    Args:\n        use_model: Pre-trained model to use as the base (e.g., EfficientNetB7).\n    \n    Returns:\n        Compiled Keras model ready for training.\n    \"\"\"\n    # Use the specified base model with ImageNet weights\n    base_model = EfficientNetB7(\n        weights='imagenet', \n        include_top=False,  # Exclude the default classification head\n        pooling='avg',      # Global average pooling\n        input_shape=(*IMAGE_SIZE, 3) \n    )\n    \n    # Add a dense layer for classification\n    x = base_model.output\n    predictions = Dense(104, activation='softmax')(x) \n    \n    # Return the final model\n    return Model(inputs=base_model.input, outputs=predictions)\n\n# # Create and compile the model within the defined strategy scope\n# with strategy.scope():    \n#     model = get_model(EfficientNetB7)  \n\n# # Compile the model with Adam optimizer and appropriate loss/metrics\n# model.compile(\n#     optimizer='adam',\n#     loss='sparse_categorical_crossentropy',  # For integer-encoded class labels\n#     metrics=['sparse_categorical_accuracy']  # Accuracy metric for sparse categorical labels\n# )\n\nwith strategy.scope():\n    model = get_model(EfficientNetB7)  # Create the model\n    # model.compile(\n    #     optimizer='adam',\n    #     loss='sparse_categorical_crossentropy',\n    #     metrics=['sparse_categorical_accuracy']\n    # )\n    optimizer = tf.keras.optimizers.Adam()\n    model.compile(\n        optimizer=optimizer,\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T16:35:25.667165Z","iopub.execute_input":"2024-12-07T16:35:25.667571Z","iopub.status.idle":"2024-12-07T16:35:46.039984Z","shell.execute_reply.started":"2024-12-07T16:35:25.667539Z","shell.execute_reply":"2024-12-07T16:35:46.038736Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with strategy.scope():\n    history = model.fit(\n        get_training_dataset(),  # Training dataset\n        steps_per_epoch=STEPS_PER_EPOCH,  # Number of steps per epoch\n        epochs=EPOCHS,  # Total number of epochs\n        callbacks=[\n            EarlyStopping(monitor='val_loss', patience=10, restore_best_weights=True),  # Stop early if validation loss doesn't improve for 10 epochs\n            lr_callback,  # Custom learning rate scheduler\n            ModelCheckpoint(\n                filepath='my_efficientnet_b7.keras',  # Save the best model to this file\n                monitor='val_loss',  # Monitor validation loss\n                save_best_only=True  # Save only when the validation loss improves\n            )\n        ],\n        validation_data=get_validation_dataset()  # Validation dataset\n    )\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T16:40:56.178980Z","iopub.execute_input":"2024-12-07T16:40:56.180185Z","iopub.status.idle":"2024-12-07T16:44:04.114238Z","shell.execute_reply.started":"2024-12-07T16:40:56.180142Z","shell.execute_reply":"2024-12-07T16:44:04.112790Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot the training and validation accuracy\nplt.plot(history.history['sparse_categorical_accuracy'], \n         label='Accuracy on Training Set')\nplt.plot(history.history['val_sparse_categorical_accuracy'], \n         label='Accuracy on Validation Set')\nplt.xlabel('Training Epoch')  # Label for the x-axis\nplt.ylabel('Accuracy')  # Label for the y-axis\nplt.xscale('log')  # Use logarithmic scale for the x-axis\nplt.legend()  # Add a legend to differentiate between training and validation curves\nplt.title('Training and Validation Accuracy')  # Add a title to the plot\nplt.grid(True)  # Add a grid for better readability\nplt.show()  # Display the plot\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T16:30:30.587048Z","iopub.status.idle":"2024-12-07T16:30:30.587465Z","shell.execute_reply.started":"2024-12-07T16:30:30.587225Z","shell.execute_reply":"2024-12-07T16:30:30.587251Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Plot the training and validation loss\nplt.plot(history.history['loss'], \n         label='Loss on Training Set')\nplt.plot(history.history['val_loss'], \n         label='Loss on Validation Set')\nplt.xlabel('Training Epoch')  # Label for the x-axis\nplt.ylabel('Loss')  # Label for the y-axis\nplt.xscale('log')  # Use logarithmic scale for the x-axis\nplt.legend()  # Add a legend to differentiate between training and validation curves\nplt.title('Training and Validation Loss')  # Add a title to the plot\nplt.grid(True)  # Add a grid for better readability\nplt.show()  # Display the plot","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T16:30:30.588527Z","iopub.status.idle":"2024-12-07T16:30:30.588811Z","shell.execute_reply.started":"2024-12-07T16:30:30.588667Z","shell.execute_reply":"2024-12-07T16:30:30.588681Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with strategy.scope():\n    model = tf.keras.models.load_model('my_efficientnet_b7.keras')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T16:30:30.589978Z","iopub.status.idle":"2024-12-07T16:30:30.590298Z","shell.execute_reply.started":"2024-12-07T16:30:30.590134Z","shell.execute_reply":"2024-12-07T16:30:30.590151Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Since we are splitting the dataset and iterating separately for images and IDs, order matters.\ntest_ds = get_test_dataset(ordered=True)\n\nprint('Computing predictions...')\n# Extract the image data from the test dataset\ntest_images_ds = test_ds.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images_ds)  # Get predicted probabilities for each class\npredictions = np.argmax(probabilities, axis=-1)  # Get the class with the highest probability\nprint(predictions)\n\nprint('Creating submission.csv...')\n# Extract the IDs from the test dataset\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(NUM_TEST_IMAGES))).numpy().astype('U')  # All IDs in a single batch\n\n# Save the predictions to a CSV file\nnp.savetxt(\n    'submission.csv', \n    np.rec.fromarrays([test_ids, predictions]), \n    fmt=['%s', '%d'], \n    delimiter=',', \n    header='id,label', \n    comments=''\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-07T16:27:57.332171Z","iopub.status.idle":"2024-12-07T16:27:57.332476Z","shell.execute_reply.started":"2024-12-07T16:27:57.332303Z","shell.execute_reply":"2024-12-07T16:27:57.332317Z"}},"outputs":[],"execution_count":null}]}