{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# A Simple TF 2.2 notebook\n\nThis is intended as a simple, short introduction to the operations competitors will need to perform with TPUs.","metadata":{}},{"cell_type":"code","source":"import tensorflow as tf\nfrom kaggle_datasets import KaggleDatasets\nimport numpy as np\n\nprint(\"Tensorflow version \" + tf.__version__)","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","execution":{"iopub.status.busy":"2022-10-02T04:34:10.315334Z","iopub.execute_input":"2022-10-02T04:34:10.315698Z","iopub.status.idle":"2022-10-02T04:34:16.842659Z","shell.execute_reply.started":"2022-10-02T04:34:10.315610Z","shell.execute_reply":"2022-10-02T04:34:16.840844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Detect my accelerator","metadata":{}},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()  # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    # based on cluster resolver config (specifying cluster information for distributed execution)\n    # for TensorFlow to communicate with various cluster management systems\n    # connect to the cluster\n    # https://www.tensorflow.org/api_docs/python/tf/config/experimental_connect_to_cluster\n    tf.config.experimental_connect_to_cluster(tpu)\n    # tensorflow.org/api_docs/python/tf/tpu/experimental/initialize_tpu_system\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    strategy = tf.distribute.get_strategy() # default distribution strategy in Tensorflow. Works on CPU and single GPU.\n\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:16.847109Z","iopub.execute_input":"2022-10-02T04:34:16.847384Z","iopub.status.idle":"2022-10-02T04:34:22.990051Z","shell.execute_reply.started":"2022-10-02T04:34:16.847352Z","shell.execute_reply":"2022-10-02T04:34:22.988892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tpu","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:22.991270Z","iopub.execute_input":"2022-10-02T04:34:22.991919Z","iopub.status.idle":"2022-10-02T04:34:23.001335Z","shell.execute_reply.started":"2022-10-02T04:34:22.991880Z","shell.execute_reply":"2022-10-02T04:34:23.000359Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tpu.cluster_spec()","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:23.003576Z","iopub.execute_input":"2022-10-02T04:34:23.003936Z","iopub.status.idle":"2022-10-02T04:34:23.012266Z","shell.execute_reply.started":"2022-10-02T04:34:23.003864Z","shell.execute_reply":"2022-10-02T04:34:23.011134Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(tpu.get_job_name())\nprint(tpu.master())\nprint(tpu.get_master())\nprint(tpu.get_tpu_system_metadata())\nprint(tpu.num_accelerators())\nprint(tpu.task_id)\nprint(tpu.task_type)\n# print(tpu.environment)\n# AttributeError: 'TPUClusterResolver' object has no attribute '_environment'\n# print(tpu.tpu_hardware_feature)\n# AttributeError: 'TPUClusterResolver' object has no attribute 'tpu_hardware_feature'\n","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:23.016588Z","iopub.execute_input":"2022-10-02T04:34:23.016863Z","iopub.status.idle":"2022-10-02T04:34:23.027233Z","shell.execute_reply.started":"2022-10-02T04:34:23.016834Z","shell.execute_reply":"2022-10-02T04:34:23.026199Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tensorflow.org/api_docs/python/tf/distribute/experimental/TPUStrategy\ntf.distribute.get_strategy()","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:23.028447Z","iopub.execute_input":"2022-10-02T04:34:23.028701Z","iopub.status.idle":"2022-10-02T04:34:23.040103Z","shell.execute_reply.started":"2022-10-02T04:34:23.028671Z","shell.execute_reply":"2022-10-02T04:34:23.039091Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Get my data path","metadata":{}},{"cell_type":"code","source":"GCS_DS_PATH = KaggleDatasets().get_gcs_path() # you can list the bucket with \"!gsutil ls $GCS_DS_PATH\"","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:23.041585Z","iopub.execute_input":"2022-10-02T04:34:23.041816Z","iopub.status.idle":"2022-10-02T04:34:23.406691Z","shell.execute_reply.started":"2022-10-02T04:34:23.041782Z","shell.execute_reply":"2022-10-02T04:34:23.405530Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"KaggleDatasets()","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:23.408126Z","iopub.execute_input":"2022-10-02T04:34:23.408392Z","iopub.status.idle":"2022-10-02T04:34:23.414446Z","shell.execute_reply.started":"2022-10-02T04:34:23.408362Z","shell.execute_reply":"2022-10-02T04:34:23.413551Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GCS_DS_PATH","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:23.415629Z","iopub.execute_input":"2022-10-02T04:34:23.415918Z","iopub.status.idle":"2022-10-02T04:34:23.427871Z","shell.execute_reply.started":"2022-10-02T04:34:23.415885Z","shell.execute_reply":"2022-10-02T04:34:23.426936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Set some parameters","metadata":{}},{"cell_type":"code","source":"# IMAGE_SIZE = [192, 192] # at this size, a GPU will run out of memory. Use the TPU\nIMAGE_SIZE = [224, 224]\n# IMAGE_SIZE = [331, 331]\n# IMAGE_SIZE = [512, 512]\n\n# increase from 5 to 50 epochs\n# EPOCHS = 50\nEPOCHS = 10\n# WHY multiply by 16?\n# because we will be using Keras VGG16 later, which uses 16 hidden/convolutional layers- (https://viso.ai/deep-learning/vgg-very-deep-convolutional-networks/)\n# https://keras.io/api/applications/vgg/\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\n\n# training set 798*15+783 = 12753 images\n# validation set 232*16 = 3712 images\nNUM_TRAINING_IMAGES = 12753\n# test set 462*15+452 images\nNUM_TEST_IMAGES = 7382\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:23.429147Z","iopub.execute_input":"2022-10-02T04:34:23.429408Z","iopub.status.idle":"2022-10-02T04:34:23.438475Z","shell.execute_reply.started":"2022-10-02T04:34:23.429377Z","shell.execute_reply":"2022-10-02T04:34:23.437696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"192/16","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:23.439449Z","iopub.execute_input":"2022-10-02T04:34:23.439680Z","iopub.status.idle":"2022-10-02T04:34:23.455562Z","shell.execute_reply.started":"2022-10-02T04:34:23.439655Z","shell.execute_reply":"2022-10-02T04:34:23.454571Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"STEPS_PER_EPOCH","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:23.456768Z","iopub.execute_input":"2022-10-02T04:34:23.457392Z","iopub.status.idle":"2022-10-02T04:34:23.467815Z","shell.execute_reply.started":"2022-10-02T04:34:23.457352Z","shell.execute_reply":"2022-10-02T04:34:23.466984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Load my data\n\nThis data is loaded from Kaggle and automatically sharded to maximize parallelization.","metadata":{}},{"cell_type":"code","source":"# Split TFRecordDataset\n# stackoverflow.com/questions/51125266/how-do-i-split-tensorflow-datasets\n# from sklearn.model_selection import train_test_split","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:23.468968Z","iopub.execute_input":"2022-10-02T04:34:23.469286Z","iopub.status.idle":"2022-10-02T04:34:23.475561Z","shell.execute_reply.started":"2022-10-02T04:34:23.469254Z","shell.execute_reply":"2022-10-02T04:34:23.474732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image(image_data):\n    # decode into 3 colour channels - RGB\n    # channels = 0 (default; use number of colour channels in the image), 1 (grayscale), 3 (RGB)\n    # https://docs.w3cub.com/tensorflow~python/tf/image/decode_jpeg\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0  # convert image to floats in [0, 1] range\n    # variable number of rows, 3 cols\n    image = tf.reshape(image, [*IMAGE_SIZE, 3]) # explicit size needed for TPU\n    return image\n\n\n\ndef read_labeled_tfrecord(example):\n    # https://www.tensorflow.org/api_docs/python/tf/io/FixedLenFeature\n    # tf.io.FixedLenFeature(shape, dtype, default_value=None)\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string means bytestring\n        \"class\": tf.io.FixedLenFeature([], tf.int64),  # shape [] means single element\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label # returns a dataset of (image, label) pairs\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string means bytestring\n        \"id\": tf.io.FixedLenFeature([], tf.string),  # shape [] means single element\n        # class is missing, this competitions's challenge is to predict flower classes for the test dataset\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image, idnum # returns a dataset of image(s)\n\n\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    # Read from TFRecords. For optimal performance, reading from multiple files at once and\n    # disregarding data order. Order does not matter since we will be shuffling the data anyway.\n\n    ignore_order = tf.data.Options()\n    # if not FALSE => if TRUE => then execute\n    if not ordered:\n        ignore_order.experimental_deterministic = False # disable order, increase speed\n\n    # https://www.tensorflow.org/api_docs/python/tf/data/TFRecordDataset\n    dataset = tf.data.TFRecordDataset(filenames) # automatically interleaves reads from multiple files\n    dataset = dataset.with_options(ignore_order) # uses data as soon as it streams in, rather than in its original order\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord)\n    # returns a dataset of (image, label) pairs if labeled=True or (image, id) pairs if labeled=False\n    return dataset\n\n\ndef get_training_dataset():\n    # dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/train/*.tfrec'), labeled=True)\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-224x224/train/*.tfrec'), labeled=True)\n    # dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-331x331/train/*.tfrec'), labeled=True)\n    # dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-512x512/train/*.tfrec'), labeled=True)\n    dataset = dataset.repeat() # the training dataset must repeat for several epochs\n    \n    # https://www.tensorflow.org/api_docs/python/tf/data/TFRecordDataset#shuffle\n    # shuffle(# elements to sample from the dataset)\n    dataset = dataset.shuffle(2048)\n    # https://www.tensorflow.org/api_docs/python/tf/data/TFRecordDataset#batch\n    # batch(batch_size, drop_remainder=False) => split the 2048 elements sets of 'batch_size' each\n    # by default (drop_remainder=False), if the last set has less elements than the 'batch_size', keep the last set of sample elements\n    dataset = dataset.batch(BATCH_SIZE)\n    return dataset\n\ndef get_validation_dataset():\n    # dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/val/*.tfrec'), labeled=True, ordered=False)\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-224x224/val/*.tfrec'), labeled=True, ordered=False)\n    # dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-331x331/val/*.tfrec'), labeled=True, ordered=False)\n    # dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-512x512/val/*.tfrec'), labeled=True, ordered=False)\n    dataset = dataset.batch(BATCH_SIZE)\n    # https://www.tensorflow.org/api_docs/python/tf/data/TFRecordDataset#cache\n    dataset = dataset.cache()\n    return dataset\n\ndef get_test_dataset(ordered=False):\n    # dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-192x192/test/*.tfrec'), labeled=False, ordered=ordered)\n    dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-224x224/test/*.tfrec'), labeled=False, ordered=ordered)\n    # dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-331x331/test/*.tfrec'), labeled=False, ordered=ordered)\n    # dataset = load_dataset(tf.io.gfile.glob(GCS_DS_PATH + '/tfrecords-jpeg-512x512/test/*.tfrec'), labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    return dataset\n\ntraining_dataset = get_training_dataset()\nvalidation_dataset = get_validation_dataset()","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:23.479793Z","iopub.execute_input":"2022-10-02T04:34:23.480079Z","iopub.status.idle":"2022-10-02T04:34:23.844577Z","shell.execute_reply.started":"2022-10-02T04:34:23.480048Z","shell.execute_reply":"2022-10-02T04:34:23.843129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# https://www.tensorflow.org/api_docs/python/tf/data/Options\nprint(tf.data.Options())\nprint(tf.data.Options().experimental_deterministic)\nprint(tf.data.Options().experimental_distribute)\nprint(tf.data.Options().experimental_optimization)\nprint(tf.data.Options().experimental_threading)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:23.846058Z","iopub.execute_input":"2022-10-02T04:34:23.846392Z","iopub.status.idle":"2022-10-02T04:34:23.854273Z","shell.execute_reply.started":"2022-10-02T04:34:23.846340Z","shell.execute_reply":"2022-10-02T04:34:23.853244Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build a model on TPU (or GPU, or CPU...) with Tensorflow 2.1!","metadata":{}},{"cell_type":"markdown","source":"Areas to explore:\n* How dataset/objects look in the section above\n* change include_top=False to True\n* change **pretrained_model.trainable** = False to **True => model accuracy dropped from around .35 to 0.225**\n* number of epochs and the improvement to categorical accuracy and loss => **we can obtain the highest accuracy for the lowest loss around 10 epochs =>  beyond that, model loss for validation set increases while model loss for training set decreases (i.e. we start overfitting)**\n* **total number of unique labels** in full and corresponding (192, 224, 331, 512) labeled datasets => **should be 104** (source: https://www.kaggle.com/competitions/tpu-getting-started/data)\n* **image resolution** which produces the lowest loss relative to accuracy => **224x224**\n* show images (inspiration: https://www.kaggle.com/code/ryanholbrook/create-your-first-submission/notebook#Step-9:-Make-a-submission)\n* Shuffle training set + understand train/validation/test set dimensions + model summary + tune learning rate + flip images + flip image saturation levels\n  * AUTO = tf.data.experimental.AUTOTUNE\n  * https://www.tensorflow.org/api_docs/python/tf/data/experimental/AutotuneAlgorithm\n  * line 225 num_parallel_reads https://github.com/tensorflow/tensorflow/blob/master/tensorflow/python/data/ops/readers.py\n  * num_parallel_calls https://www.tensorflow.org/guide/data_performance\n  * tf.image.random_flip_left_right\n  * tf.image.random_saturation\n  * \"Muhammad Ali\" used 512x512 images, which worked poorly for my model (inspiration: https://www.kaggle.com/code/muhammadali14/flower-classification-on-tpu/notebook?scriptVersionId=103754753)\n* Consider switching from Tensorflow to PyTorch and experimenting with other models\n  * (inspiration: https://www.kaggle.com/code/trnduythanhkhttt/term-project-image-classification-on-tpu-pytorch/notebook?scriptVersionId=105355142)\n* Checkpoint/save 'the best' model weights during training/model fitting + customise learning rate\n  * https://www.tensorflow.org/api_docs/python/tf/keras/callbacks/ModelCheckpoint\n  * tfa.optimizers.CyclicalLearningRate\n  * tf.nn.compute_average_loss\n  * Qn: alpha_weight(step) => as step increases, increase 'alpha_weight af' => so this is penalising/increasing the loss for subsequent steps => should this penalty be increasing for subsequent epochs instead (where the model should be presumably better trained)?\n  * Qn: not sure how 'div', 'fit_on_train_per_epoch', 'strategy.reduce' works\n  * Qn: seems to be getting 'the best' weights/variables which reduce loss every batch; BUT why are we only training only on certain batches???\n  * Qn: overall, not sure how the 'pseudo labelling' technique in the notebook works\n  * (inspiration: https://www.kaggle.com/code/olegbaryshnikov/tensorflow-petals-pseudo-labeling-on-tpu/notebook?scriptVersionId=102947808)","metadata":{}},{"cell_type":"code","source":"with strategy.scope():    \n    # tensorflow.org/api_docs/python/tf/keras/applications/vgg16/VGG16\n    # weights: None (random initialization), 'imagenet' (pre-training on ImageNet), or the path to the weights file to be loaded.\n    pretrained_model = tf.keras.applications.VGG16(weights='imagenet',\n                                                   # whether to include the 3 fully-connected layers at the top of the network\n                                                   # i.e. Do not include the ImageNet classifier at the top\n                                                   # ValueError: When setting `include_top=True` and loading `imagenet` weights, `input_shape` should be (224, 224, 3).\n                                                   include_top=False, \n                                                   input_shape=[*IMAGE_SIZE, 3])\n    # Freeze base model\n    # pretrained_model.trainable = False # transfer learning\n    # Train model on our new dataset\n    pretrained_model.trainable = True\n    \n    model = tf.keras.Sequential([\n        pretrained_model,\n        tf.keras.layers.GlobalAveragePooling2D(),\n        tf.keras.layers.Dense(104, activation='softmax')\n    ])\n\n    \nmodel.compile(\n    optimizer='adam',\n    loss = 'sparse_categorical_crossentropy',\n    metrics=['sparse_categorical_accuracy']\n)\n\nhistorical = model.fit(training_dataset, \n          steps_per_epoch=STEPS_PER_EPOCH, \n          epochs=EPOCHS, \n          validation_data=validation_dataset)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:34:23.855473Z","iopub.execute_input":"2022-10-02T04:34:23.855705Z","iopub.status.idle":"2022-10-02T04:37:36.820911Z","shell.execute_reply.started":"2022-10-02T04:34:23.855678Z","shell.execute_reply":"2022-10-02T04:37:36.820058Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:37:36.822239Z","iopub.execute_input":"2022-10-02T04:37:36.822476Z","iopub.status.idle":"2022-10-02T04:37:36.826557Z","shell.execute_reply.started":"2022-10-02T04:37:36.822450Z","shell.execute_reply":"2022-10-02T04:37:36.825485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Inspiration from https://www.kaggle.com/code/ryanholbrook/create-your-first-submission/notebook\ndef display_training_curves(training, validation, title, subplot):\n    if subplot%10==1: # set up the subplots on the first call\n        plt.subplots(figsize=(10,10), facecolor='#F0F0F0')   # grey - encloses each subplot\n        plt.tight_layout()\n    ax = plt.subplot(subplot)\n    ax.set_facecolor('#F8F8F8')   # light grey - within the x and y axes\n    ax.plot(training)\n    ax.plot(validation)\n    ax.set_title('model '+ title)\n    ax.set_ylabel(title)\n    #ax.set_ylim(0.28,1.05)\n    ax.set_xlabel('epoch')\n    ax.legend(['train', 'valid.'])","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:37:36.828167Z","iopub.execute_input":"2022-10-02T04:37:36.828651Z","iopub.status.idle":"2022-10-02T04:37:36.842681Z","shell.execute_reply.started":"2022-10-02T04:37:36.828609Z","shell.execute_reply":"2022-10-02T04:37:36.841861Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Track the loss for the training and validation set across epochs\ndisplay_training_curves(\n    historical.history['loss'],\n    historical.history['val_loss'],\n    'loss',\n    211,\n)\n\n# Track the accuracy for the training and validation set across epochs\ndisplay_training_curves(\n    historical.history['sparse_categorical_accuracy'],\n    historical.history['val_sparse_categorical_accuracy'],\n    'accuracy',\n    212,\n)","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:37:36.844293Z","iopub.execute_input":"2022-10-02T04:37:36.844611Z","iopub.status.idle":"2022-10-02T04:37:37.435401Z","shell.execute_reply.started":"2022-10-02T04:37:36.844573Z","shell.execute_reply":"2022-10-02T04:37:37.434507Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Compute your predictions on the test set!\n\nThis will create a file that can be submitted to the competition.","metadata":{}},{"cell_type":"code","source":"test_ds = get_test_dataset(ordered=True) # since we are splitting the dataset and iterating separately on images and ids, order matters.\n\nprint('Computing predictions...')\n# from each (image, idnum) pair => keep just the image in test_images_ds\ntest_images_ds = test_ds.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images_ds)\n# https://het.as.utexas.edu/HET/Software/Numpy/reference/generated/numpy.argmax.html#numpy.argmax\n# return the col index # with the highest probability from each row of probability elements\n# that is our predicted class/label\npredictions = np.argmax(probabilities, axis=-1)\nprint(predictions)\n\nprint('Generating submission.csv file...')\n# from each (image, idnum) pair => keep just the idnum in test_ids_ds\n# https://www.tensorflow.org/api_docs/python/tf/data/experimental/unbatch\n# https://www.tensorflow.org/api_docs/python/tf/data/Dataset#unbatch\n# i.e. shape the full set of 'idnum' into consecutive elements\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\n# then we extract 7382 test images\n# https://www.reddit.com/r/Python/comments/gpljcq/pandas_what_does_dfastypeu_mean/\n# astype(‘U’) tells numpy to convert the data to Unicode (essentially a string in python 3)\ntest_ids = next(iter(test_ids_ds.batch(NUM_TEST_IMAGES))).numpy().astype('U') # all in one batch\nnp.savetxt('submission.csv', \n           np.rec.fromarrays([test_ids, predictions]), \n           fmt=['%s', '%d'], \n           delimiter=',', \n           header='id,label', \n           comments='')","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:37:37.436463Z","iopub.execute_input":"2022-10-02T04:37:37.436689Z","iopub.status.idle":"2022-10-02T04:38:13.133043Z","shell.execute_reply.started":"2022-10-02T04:37:37.436664Z","shell.execute_reply":"2022-10-02T04:38:13.132099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!head submission.csv","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:38:13.134447Z","iopub.execute_input":"2022-10-02T04:38:13.134669Z","iopub.status.idle":"2022-10-02T04:38:14.239022Z","shell.execute_reply.started":"2022-10-02T04:38:13.134644Z","shell.execute_reply":"2022-10-02T04:38:14.238059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os, fnmatch\ndef find(pattern, path):\n    result = []\n    for root, dirs, files in os.walk(path):\n        for name in files:\n            if fnmatch.fnmatch(name, pattern):\n                result.append(os.path.join(root, name))\n    return result\n\nfind('*.csv', \"../\")","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:38:14.241027Z","iopub.execute_input":"2022-10-02T04:38:14.241401Z","iopub.status.idle":"2022-10-02T04:38:14.417147Z","shell.execute_reply.started":"2022-10-02T04:38:14.241352Z","shell.execute_reply":"2022-10-02T04:38:14.416212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"find('*.csv', \"./\")","metadata":{"execution":{"iopub.status.busy":"2022-10-02T04:38:14.418802Z","iopub.execute_input":"2022-10-02T04:38:14.419064Z","iopub.status.idle":"2022-10-02T04:38:14.426288Z","shell.execute_reply.started":"2022-10-02T04:38:14.419036Z","shell.execute_reply":"2022-10-02T04:38:14.425385Z"},"trusted":true},"execution_count":null,"outputs":[]}]}