{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":6799,"databundleVersionId":4225553,"sourceType":"competition"}],"dockerImageVersionId":30628,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# Import packages\nimport numpy as np\nimport pandas as pd\nimport tensorflow as tf\nimport os\nimport shutil\nfrom tensorflow.keras import layers\nimport matplotlib.pyplot as plt\nfrom matplotlib import image","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2023-12-25T20:06:55.306282Z","iopub.execute_input":"2023-12-25T20:06:55.306588Z","iopub.status.idle":"2023-12-25T20:07:09.321975Z","shell.execute_reply.started":"2023-12-25T20:06:55.30656Z","shell.execute_reply":"2023-12-25T20:07:09.321266Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import packages\n\n# Data analysis\nimport numpy as np\nimport pandas as pd\n\n# File management\nimport os\nimport shutil\n\n# Image visualisation\nimport matplotlib.pyplot as plt\n\n# Neural network\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom keras import layers\n\n# Hide warnings\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:07:13.34593Z","iopub.execute_input":"2023-12-25T20:07:13.346744Z","iopub.status.idle":"2023-12-25T20:07:13.350906Z","shell.execute_reply.started":"2023-12-25T20:07:13.346707Z","shell.execute_reply":"2023-12-25T20:07:13.350186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# If TPU is available\ntry:\n    # Detect TPU\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() \n    print('Running on TPU ', tpu.master())\n\nexcept:\n    tpu = None\n    \n\n# If TPU is defined\nif tpu:\n    # Initialise TPU\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    # Instantiate a TPU distribution strategy\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\n\nelse:\n    # Or instanstiate available strategy\n    strategy = tf.distribute.get_strategy() \n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:07:16.785996Z","iopub.execute_input":"2023-12-25T20:07:16.786804Z","iopub.status.idle":"2023-12-25T20:07:25.132169Z","shell.execute_reply.started":"2023-12-25T20:07:16.786759Z","shell.execute_reply":"2023-12-25T20:07:25.13138Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define constants\n\n# Set batch size for mini-batch gradient descent\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\n# Set number of epochs to train model\nEPOCH_NUM = 50\n\n# Set kernel size for Conv2D layers\nKERNEL_SIZE = 3\n# Set padding mode for Conv2D layers\nPAD_MODE = \"same\"\n# Set activation function for Conv2D layers\nACTIVATION = \"relu\"\n\n# Set pool size for MaxPool2D layers\nPOOL_SIZE = 2\n# Set strides for MaxPool2D layers\nPOOL_STRIDES = 2","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:07:31.817725Z","iopub.execute_input":"2023-12-25T20:07:31.818044Z","iopub.status.idle":"2023-12-25T20:07:31.822551Z","shell.execute_reply.started":"2023-12-25T20:07:31.818013Z","shell.execute_reply":"2023-12-25T20:07:31.821729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Store paths to base, train set and subset dirs\nbase_dir = \"/kaggle/input/imagenet-object-localization-challenge/\"\ndata_dir = base_dir + \"ILSVRC/Data/CLS-LOC/\"\n\n# Fetch train set\nraw_train_ds = tf.data.Dataset.list_files(data_dir + \"train/n01*/*.JPEG\", shuffle=False)\n\n# Find size of train set\ntrain_size = tf.data.experimental.cardinality(raw_train_ds).numpy()\n\n# Shuffle train set\nraw_train_ds = raw_train_ds.shuffle(train_size, reshuffle_each_iteration=False)\n\nprint(f\"Size of train set: {train_size}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:07:35.077055Z","iopub.execute_input":"2023-12-25T20:07:35.077397Z","iopub.status.idle":"2023-12-25T20:07:37.885939Z","shell.execute_reply.started":"2023-12-25T20:07:35.077366Z","shell.execute_reply":"2023-12-25T20:07:37.885029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Import and extract devel labels\ndevel_df = pd.read_csv(base_dir + \"LOC_val_solution.csv\")\ny_devel = devel_df[\"PredictionString\"].str.split(expand=True)[0]\n\n# Select devel images belonging to subset of classes\nkeep = np.where(y_devel.str.startswith(\"n01\").to_numpy())[0]\ndevel_files = np.array(os.listdir(data_dir + \"val/\"))[keep]\n\n# Fetch devel set\ndevel_files = np.array([data_dir + \"val/\" + file for file in devel_files])\nraw_devel_ds = tf.data.Dataset.list_files(devel_files, shuffle=False)\n\n# Select labels belonging to subset of classes\ny_devel = y_devel[y_devel.str.startswith(\"n01\")].values\n\n# Find size of devel set\ndevel_size = tf.data.experimental.cardinality(raw_devel_ds).numpy()\n\nprint(f\"Size of train set: {devel_size}\")\nprint(\"Examples of labels:\")\ny_devel[:10]","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:07:40.42812Z","iopub.execute_input":"2023-12-25T20:07:40.42874Z","iopub.status.idle":"2023-12-25T20:07:55.421343Z","shell.execute_reply.started":"2023-12-25T20:07:40.428705Z","shell.execute_reply":"2023-12-25T20:07:55.420602Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Find class names from dir names\nclass_names = np.array(sorted([dir for dir in os.listdir(data_dir + \"train/\") if dir.startswith(\"n01\")]))\n\n# Set number of classes\nCLASS_NUM = len(class_names)\n\nfor f in raw_train_ds.take(5):\n    print(f.numpy())","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:08:05.868575Z","iopub.execute_input":"2023-12-25T20:08:05.869261Z","iopub.status.idle":"2023-12-25T20:08:06.028861Z","shell.execute_reply.started":"2023-12-25T20:08:05.869225Z","shell.execute_reply":"2023-12-25T20:08:06.027952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Define label extractor for train samples\ndef extract_train_label(file_path):\n    \n    # Split file path into parts\n    parts = tf.strings.split(file_path, os.path.sep)\n    # Extract class dir name\n    one_hot = parts[-2] == class_names\n    # Find index of maximum\n    label = tf.argmax(one_hot)\n    \n    return label\n\n# Define label extractor for devel samples\ndef extract_devel_label(file_path):\n    \n    # One-hot encode indices\n    idx_hot = devel_files == file_path\n    # Find index\n    idx = tf.argmax(idx_hot)\n    # Extract class name\n    class_name = tf.gather(y_devel, idx)\n\n    # Extract class dir name\n    one_hot = class_name == class_names\n    # Find index of maximum\n    label = tf.argmax(one_hot)\n    \n    return label\n\n# Define sample preprocessing pipeline\ndef process_path(file_path, label_fun=extract_train_label, cut=224, S=256, max_delta=0.2):\n    \n    # Convert the compressed string to a 3D uint8 tensor\n    image = tf.io.read_file(file_path)\n    image = tf.io.decode_jpeg(image, channels=3)\n    \n    image =  tf.keras.applications.vgg16.preprocess_input(image)\n   \n    # Find rescaling factor\n    min_side = tf.math.minimum(tf.shape(image)[0], tf.shape(image)[1])\n    scale = S / min_side\n\n    # Compute new dimensions\n    new_height = tf.cast(tf.shape(image)[0], tf.float64) * scale\n    new_width = tf.cast(tf.shape(image)[1], tf.float64) * scale\n    \n    # Convert to float\n    image = tf.image.convert_image_dtype(image, tf.float32)\n    # Rescale by S\n    image = tf.image.resize(image, size=(new_height, new_width), preserve_aspect_ratio=True)\n    # Crop randomly\n    image = tf.image.random_crop(image, size=[cut, cut, 3])\n    # adjust RGB values by random amount\n    image = tf.image.random_hue(image, max_delta=max_delta)\n    \n    label = label_fun(file_path)\n    \n    return image, label\n\nimage, label = process_path(data_dir + \"val/ILSVRC2012_val_00046886.JPEG\", label_fun=extract_devel_label)\ninput_shape = image.shape\n\nplt.imshow(image)\nprint(label)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:08:09.852048Z","iopub.execute_input":"2023-12-25T20:08:09.852372Z","iopub.status.idle":"2023-12-25T20:08:10.209717Z","shell.execute_reply.started":"2023-12-25T20:08:09.852344Z","shell.execute_reply":"2023-12-25T20:08:10.208935Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image.shape","metadata":{"execution":{"iopub.status.busy":"2023-12-25T18:33:43.070158Z","iopub.execute_input":"2023-12-25T18:33:43.070566Z","iopub.status.idle":"2023-12-25T18:33:43.077009Z","shell.execute_reply.started":"2023-12-25T18:33:43.070529Z","shell.execute_reply":"2023-12-25T18:33:43.076023Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Preprocess images\ntrain_ds = raw_train_ds.map(process_path, num_parallel_calls=tf.data.AUTOTUNE)\ndevel_ds = raw_devel_ds.map(lambda path: process_path(path, label_fun=extract_devel_label, max_delta=0), num_parallel_calls=tf.data.AUTOTUNE)\n\n# Show example\nfor image, label in train_ds.take(1):\n    print(\"Image shape: \", image.numpy().shape)\n    print(\"Label: \", label.numpy())\n","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:08:14.39405Z","iopub.execute_input":"2023-12-25T20:08:14.394416Z","iopub.status.idle":"2023-12-25T20:08:15.05583Z","shell.execute_reply.started":"2023-12-25T20:08:14.394381Z","shell.execute_reply":"2023-12-25T20:08:15.05489Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Improve train set performance\ntrain_ds = train_ds \\\n    .cache() \\\n    .shuffle(buffer_size=1000) \\\n    .batch(BATCH_SIZE) \\\n    .prefetch(buffer_size=tf.data.AUTOTUNE)\n\n# Improve devel set performance\ndevel_ds = devel_ds \\\n    .batch(BATCH_SIZE) \\\n    .cache() \\\n    .prefetch(buffer_size=tf.data.AUTOTUNE)","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:08:16.785973Z","iopub.execute_input":"2023-12-25T20:08:16.786335Z","iopub.status.idle":"2023-12-25T20:08:16.801326Z","shell.execute_reply.started":"2023-12-25T20:08:16.786301Z","shell.execute_reply":"2023-12-25T20:08:16.800363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_batch, label_batch = next(iter(train_ds))\n\nplt.figure(figsize=(10, 10))\nfor i in range(9):\n    ax = plt.subplot(3, 3, i + 1)\n    plt.imshow(image_batch[i])\n    label = label_batch[i]\n    plt.title(class_names[label])\n    plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:08:19.172317Z","iopub.execute_input":"2023-12-25T20:08:19.17263Z","iopub.status.idle":"2023-12-25T20:08:22.238641Z","shell.execute_reply.started":"2023-12-25T20:08:19.172601Z","shell.execute_reply":"2023-12-25T20:08:22.237607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build model\ndef design_model():\n    \n    model = keras.models.Sequential([\n\n        # 1st convolutional block\n        layers.Conv2D(input_shape=input_shape, filters=64, kernel_size=KERNEL_SIZE,  kernel_initializer=initializers.RandomNormal(stddev=0.1), padding=PAD_MODE, activation=ACTIVATION),\n        layers.MaxPooling2D(pool_size=POOL_SIZE, strides=POOL_STRIDES),\n\n        # 2nd convolutional block\n        layers.Conv2D(filters=128, kernel_size=KERNEL_SIZE,  kernel_initializer=initializers.RandomNormal(stddev=0.1), padding=PAD_MODE, activation=ACTIVATION),\n        layers.MaxPooling2D(pool_size=POOL_SIZE, strides=POOL_STRIDES),\n\n        # 3rd convolutional block\n        layers.Conv2D(filters=256, kernel_size=KERNEL_SIZE,  kernel_initializer=initializers.RandomNormal(stddev=0.1), padding=PAD_MODE, activation=ACTIVATION),\n        layers.Conv2D(filters=256, kernel_size=KERNEL_SIZE,  kernel_initializer=initializers.RandomNormal(stddev=0.1), padding=PAD_MODE, activation=ACTIVATION),\n        layers.MaxPool2D(pool_size=POOL_SIZE, strides=POOL_STRIDES),\n\n        # 4th convolutional block\n        layers.Conv2D(filters=512, kernel_size=KERNEL_SIZE,  kernel_initializer=initializers.RandomNormal(stddev=0.1), padding=PAD_MODE, activation=ACTIVATION),\n        layers.Conv2D(filters=512, kernel_size=KERNEL_SIZE, kernel_initializer=initializers.RandomNormal(stddev=0.1), padding=PAD_MODE, activation=ACTIVATION),\n        layers.MaxPool2D(pool_size=POOL_SIZE, strides=POOL_STRIDES),\n\n        # 5th convolutional block\n        layers.Conv2D(filters=512, kernel_size=KERNEL_SIZE, kernel_initializer=initializers.RandomNormal(stddev=0.1), padding=PAD_MODE, activation=ACTIVATION),\n        layers.Conv2D(filters=512, kernel_size=KERNEL_SIZE,  kernel_initializer=initializers.RandomNormal(stddev=0.1), padding=PAD_MODE, activation=ACTIVATION),\n        layers.MaxPool2D(pool_size=POOL_SIZE, strides=POOL_STRIDES),\n\n        # Classifier head\n        layers.Flatten(),\n        layers.Dense(4096, activation=ACTIVATION, kernel_initializer=initializers.RandomNormal(stddev=0.1)),\n        layers.Dropout(rate=0.5),\n        layers.Dense(4096, activation=ACTIVATION, kernel_initializer=initializers.RandomNormal(stddev=0.1)),\n        layers.Dropout(rate=0.5),\n        layers.Dense(CLASS_NUM, kernel_initializer=initializers.RandomNormal(stddev=0.1))\n    ])\n    \n    return model\n\n\n\n#std of ten to the minus 2 , random initailization in 11 layers should be in ALL layers ","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:08:34.088753Z","iopub.execute_input":"2023-12-25T20:08:34.089159Z","iopub.status.idle":"2023-12-25T20:08:34.099862Z","shell.execute_reply.started":"2023-12-25T20:08:34.089113Z","shell.execute_reply":"2023-12-25T20:08:34.098982Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_optimiser():\n    \n    sgd_optimiser = keras.optimizers.experimental.SGD(\n            learning_rate=1e-2,\n            momentum=0.9,\n            nesterov=False,\n            weight_decay=5e-4\n        )\n    \n    return sgd_optimiser","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:08:37.921335Z","iopub.execute_input":"2023-12-25T20:08:37.922386Z","iopub.status.idle":"2023-12-25T20:08:37.926627Z","shell.execute_reply.started":"2023-12-25T20:08:37.922345Z","shell.execute_reply":"2023-12-25T20:08:37.92562Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set up model within scope of tpu strategy\nwith strategy.scope():\n    \n    # Build model\n    model = design_model()\n    \n    # Define optimiser\n    sgd_optimiser = create_optimiser()\n    \n    # Compile model\n    model.compile(\n        optimizer = sgd_optimiser,\n        loss=keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n        metrics=[\"sparse_categorical_accuracy\"]\n    )","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:08:40.784386Z","iopub.execute_input":"2023-12-25T20:08:40.785232Z","iopub.status.idle":"2023-12-25T20:08:49.835514Z","shell.execute_reply.started":"2023-12-25T20:08:40.785194Z","shell.execute_reply":"2023-12-25T20:08:49.834372Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.get_weights() #sanity check, 0s are biases that are initialized to 0 automatically by keras","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:08:53.588216Z","iopub.execute_input":"2023-12-25T20:08:53.588679Z","iopub.status.idle":"2023-12-25T20:08:54.275988Z","shell.execute_reply.started":"2023-12-25T20:08:53.588641Z","shell.execute_reply":"2023-12-25T20:08:54.274979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Update learning rate\nLR_Decay = keras.callbacks.ReduceLROnPlateau(\n    monitor=\"val_loss\",\n    factor=0.1,\n    patience=2,\n    mode=\"min\",\n    min_delta=1e-4,\n    min_lr=1e-5\n)\n\n# Train the model\nhistory = model.fit(\n    train_ds,\n    validation_data=devel_ds,\n    epochs=EPOCH_NUM,\n    batch_size=BATCH_SIZE,\n    verbose=False,\n    callbacks=[LR_Decay]\n)\n\n# Store training history as a dataframe\nhistory_df = pd.DataFrame(history.history)\n\nprint(f\"Train loss: {history_df['loss'].iloc[-1]:.3f}, Devel loss: {history_df['val_loss'].iloc[-1]:.3f}\")\nprint(f\"Train accuracy: {history_df['sparse_categorical_accuracy'].iloc[-1]:.3f}, Devel accuracy: {history_df['val_sparse_categorical_accuracy'].iloc[-1]:.3f}\")","metadata":{"execution":{"iopub.status.busy":"2023-12-25T20:10:40.87009Z","iopub.execute_input":"2023-12-25T20:10:40.870584Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualise loss\nhistory_df.loc[:, [\"loss\", \"val_loss\"]].plot(title=\"Loss\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualise accuracy\nhistory_df.loc[:, [\"sparse_categorical_accuracy\", \"val_sparse_categorical_accuracy\"]].plot(title=\"Accuracy\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"#Transfer Learning\n\nThe next step involves building a deeper ConvNet by:\n\ntaking the pretrained base from the previous section,\nenriching it with new untrained convolutional layers and\nattaching it to a new untrained classification head\nBy transferring the learnt weight from our previous model, we can build more complex and deeper ConvNets without too much computational burden. In particular, here we aim to reproduce the following four architectures from Zimonyan & Zisserman 2015:\n\n13-layer ConvNet\n16-layer ConvNet (with 1-by-1 filters)\n16-layer ConvNet (with 3-by-3 filters)\n19-layer ConvNet\nIn the following cell, we pretend that the previously trained model is stored in a keras file to show how pretrained models can be imported into a notebook.","metadata":{}},{"cell_type":"code","source":"# Define model file\nmodel_file = f\"/kaggle/working/{CLASS_NUM}class_model.keras\"\n\n# Save model into file for replication purposes\nmodel.save(model_file)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Build model\ndef design_model(model_file, layer_positions=[0, 2, 4, 5, 7, 8, 10, 11]): #CHANGE: transfer specific layers as per architecture\n    \n    # Import pretrained base\n    pretrained_base = keras.models.load_model(model_file)\n    \n    # Select relevant layers\n    pretrained_layers = [pretrained_base.get_layer(index=i) for i in layer_positions]\n    \n    model = keras.models.Sequential([\n\n        # 1st convolutional block\n        pretrained_layers[0],\n        layers.Conv2D(filters=64, kernel_size=KERNEL_SIZE, padding=PAD_MODE, activation=ACTIVATION),\n        layers.MaxPooling2D(pool_size=POOL_SIZE, strides=POOL_STRIDES),\n\n        # 2nd convolutional block\n        pretrained_layers[1],\n        layers.Conv2D(filters=128, kernel_size=KERNEL_SIZE, padding=PAD_MODE, activation=ACTIVATION),\n        layers.MaxPooling2D(pool_size=POOL_SIZE, strides=POOL_STRIDES),\n\n        # 3rd convolutional block\n        pretrained_layers[2],\n        pretrained_layers[3],\n        layers.MaxPool2D(pool_size=POOL_SIZE, strides=POOL_STRIDES),\n\n        # 4th convolutional block\n        pretrained_layers[4],\n        pretrained_layers[5],\n        layers.MaxPool2D(pool_size=POOL_SIZE, strides=POOL_STRIDES),\n\n        # 5th convolutional block\n        pretrained_layers[6],\n        pretrained_layers[7],\n        layers.MaxPool2D(pool_size=POOL_SIZE, strides=POOL_STRIDES),\n\n        # Classifier head\n        layers.Flatten(),\n        layers.Dense(4096, activation=ACTIVATION),\n        layers.Dropout(rate=0.5),\n        layers.Dense(4096, activation=ACTIVATION),\n        layers.Dropout(rate=0.5),\n        layers.Dense(CLASS_NUM)\n    ])\n    \n    return model","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Set up model within scope of tpu strategy\nwith strategy.scope():\n    \n    # Build model\n    model = design_model(model_file)\n    \n    # Define optimiser\n    sgd_optimiser = create_optimiser()\n    \n    # Compile model\n    model.compile(\n        optimizer = sgd_optimiser,\n        loss=keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n        metrics=[\"sparse_categorical_accuracy\"]\n    )","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Update learning rate\nLR_Decay = keras.callbacks.ReduceLROnPlateau(\n    monitor=\"val_loss\",\n    factor=0.1,\n    patience=2,\n    mode=\"min\",\n    min_delta=1e-4,\n    min_lr=1e-5\n)\n\n# Train the model\nhistory = model.fit(\n    train_ds,\n    validation_data=devel_ds,\n    epochs=70 - EPOCH_NUM,\n    batch_size=BATCH_SIZE,\n    verbose=False,\n    callbacks=[LR_Decay]\n)\n\n# Store training history as a dataframe\nhistory_df = pd.DataFrame(history.history)\n\nprint(f\"Train loss: {history_df['loss'].iloc[-1]:.3f}, Devel loss: {history_df['val_loss'].iloc[-1]:.3f}\")\nprint(f\"Train accuracy: {history_df['sparse_categorical_accuracy'].iloc[-1]:.3f}, Devel accuracy: {history_df['val_sparse_categorical_accuracy'].iloc[-1]:.3f}\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualise loss\nhistory_df.loc[:, [\"loss\", \"val_loss\"]].plot(title=\"Loss\")","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Visualise accuracy\nhistory_df.loc[:, [\"sparse_categorical_accuracy\", \"val_sparse_categorical_accuracy\"]].plot(title=\"Accuracy\")","metadata":{},"execution_count":null,"outputs":[]}]}