{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"#Deep Learning Project 1: Rafael Baez & Issac Rivas\n## Flower Classification with TPUs","metadata":{"id":"pm4h9Mi24oQY"}},{"cell_type":"code","source":"# from google.colab import drive\n# drive.mount('/content/drive', force_remount=True)\n# %cd /content/drive/MyDrive/Assignments/DL/petals/","metadata":{"id":"58MHJP0-5Fe0","outputId":"f5f32d2e-2bf2-4645-8728-5d89e64f8d5a","execution":{"iopub.status.busy":"2022-11-03T05:11:40.385919Z","iopub.execute_input":"2022-11-03T05:11:40.386370Z","iopub.status.idle":"2022-11-03T05:11:40.403477Z","shell.execute_reply.started":"2022-11-03T05:11:40.386282Z","shell.execute_reply":"2022-11-03T05:11:40.402527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install --user -r requirements.txt","metadata":{"id":"71F-s3ed9XNE","outputId":"93698987-e295-4a1c-912d-335f722fd8f2","execution":{"iopub.status.busy":"2022-11-03T05:11:40.405460Z","iopub.execute_input":"2022-11-03T05:11:40.405862Z","iopub.status.idle":"2022-11-03T05:11:40.413874Z","shell.execute_reply.started":"2022-11-03T05:11:40.405821Z","shell.execute_reply":"2022-11-03T05:11:40.412797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ../input","metadata":{"execution":{"iopub.status.busy":"2022-11-03T05:11:40.415523Z","iopub.execute_input":"2022-11-03T05:11:40.416051Z","iopub.status.idle":"2022-11-03T05:11:41.414126Z","shell.execute_reply.started":"2022-11-03T05:11:40.416015Z","shell.execute_reply":"2022-11-03T05:11:41.412898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nprint()\nfor dirname, _, filenames in os.walk(fr'../input'):\n    for filename in filenames:\n        if '.tfrec' in filename:\n            print(os.path.join(dirname, filename))","metadata":{"id":"XZo41JG2_H3k","outputId":"b8f44aa9-a9da-444f-8751-cbacf3061bf4","execution":{"iopub.status.busy":"2022-11-03T05:11:41.418063Z","iopub.execute_input":"2022-11-03T05:11:41.418385Z","iopub.status.idle":"2022-11-03T05:11:41.505986Z","shell.execute_reply.started":"2022-11-03T05:11:41.418354Z","shell.execute_reply":"2022-11-03T05:11:41.504968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport tensorflow as tf\n\nIMAGE_SIZE = [192, 192]\nEPOCHS = 5\nBATCH_SIZE = 16\n\nNUM_TRAINING_IMAGES = 12753\nNUM_TEST_IMAGES = 7382\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\nDATA_PATH = '../input/tpu-getting-started'\nAUTO = tf.data.experimental.AUTOTUNE","metadata":{"id":"pUEetCMA4oQ7","execution":{"iopub.status.busy":"2022-11-03T05:11:41.507395Z","iopub.execute_input":"2022-11-03T05:11:41.508356Z","iopub.status.idle":"2022-11-03T05:11:45.527535Z","shell.execute_reply.started":"2022-11-03T05:11:41.508314Z","shell.execute_reply":"2022-11-03T05:11:45.526523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32) / 255.0\n    image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"class\": tf.io.FixedLenFeature([], tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example[\"image\"])\n    label = tf.cast(example[\"class\"], tf.int32)\n    return image, label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"id\": tf.io.FixedLenFeature([], tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example[\"image\"])\n    idnum = example[\"id\"]\n    return image, idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n        \n    dataset = tf.data.TFRecordDataset(filenames)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord)\n    return dataset\n\ndef get_training_dataset():\n    dataset_orig = load_dataset(tf.io.gfile.glob(DATA_PATH + '/tfrecords-jpeg-192x192/train/*.tfrec'), labeled=True)\n    dataset = dataset_orig.repeat()\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset, dataset_orig\n\ndef get_validation_dataset():\n    dataset_orig = load_dataset(tf.io.gfile.glob(DATA_PATH + '/tfrecords-jpeg-192x192/val/*.tfrec'), labeled=True, ordered=False)\n    dataset = dataset_orig.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTO)\n    return dataset, dataset_orig\n\ndef get_test_dataset(ordered=False):\n    dataset = load_dataset(tf.io.gfile.glob(DATA_PATH + '/tfrecords-jpeg-192x192/test/*.tfrec'), labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\ntraining_dataset, training_ds  = get_training_dataset()\nvalidation_dataset, testing_ds = get_validation_dataset()","metadata":{"id":"QXn8r7qY4oQ_","execution":{"iopub.status.busy":"2022-11-03T05:11:45.528850Z","iopub.execute_input":"2022-11-03T05:11:45.529512Z","iopub.status.idle":"2022-11-03T05:11:47.932621Z","shell.execute_reply.started":"2022-11-03T05:11:45.529465Z","shell.execute_reply":"2022-11-03T05:11:47.931611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\nfig = plt.figure(figsize = (20,20))\ni = 1\nfor image, label in training_dataset.take(1).unbatch():\n    ax = fig.add_subplot(4,4,i)\n    imgplot = plt.imshow(image)\n    ax.set_title(label.numpy())\n    i += 1","metadata":{"id":"lBV_HGZO4oRO","outputId":"c876dbd5-9cea-4895-b151-a99c1cf397d6","execution":{"iopub.status.busy":"2022-11-03T05:11:47.934139Z","iopub.execute_input":"2022-11-03T05:11:47.934754Z","iopub.status.idle":"2022-11-03T05:11:53.101327Z","shell.execute_reply.started":"2022-11-03T05:11:47.934715Z","shell.execute_reply":"2022-11-03T05:11:53.099912Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import tensorflow_datasets as tfds\n# training_iter = tfds.as_numpy(training_ds)\n# testing_iter = tfds.as_numpy(testing_ds)\n# raw_imagesTr = []\n# raw_labelsTr = []\n# for e in training_iter:\n#     raw_imagesTr.append(e[0])\n#     raw_labelsTr.append(e[1])\n# raw_imagesTe = []\n# raw_labelsTe = []\n# for e in testing_iter:\n#     raw_imagesTe.append(e[0])\n#     raw_labelsTe.append(e[1])\n# X_train = np.array(raw_imagesTr)\n# Y_train = np.array(raw_labelsTr)\n# X_test = np.array(raw_imagesTe)\n# Y_test = np.array(raw_labelsTe)\n# print(X_train.shape)\n# print(Y_train.shape)\n# print(X_test.shape)\n# print(Y_test.shape)\n# del X_train\n# del Y_train\n# del X_test\n# del Y_test","metadata":{"id":"ay9oYGpA4oRS","outputId":"a391df8b-6c4e-49ce-8134-cac3d394d7ff","execution":{"iopub.status.busy":"2022-11-03T05:11:53.102809Z","iopub.execute_input":"2022-11-03T05:11:53.103235Z","iopub.status.idle":"2022-11-03T05:11:53.112013Z","shell.execute_reply.started":"2022-11-03T05:11:53.103183Z","shell.execute_reply":"2022-11-03T05:11:53.111117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# trainingLabels = np.unique(Y_train, return_counts=True)\n# testingLabels = np.unique(Y_test, return_counts=True)\n# # There should be 103 labels, and as shown below, it is not evenly distributed.\n# # Neither by class or between the training and testing datasets\n# print(f'Training most common {np.argmax(trainingLabels[1])} = {(trainingLabels[1][np.argmax(trainingLabels[1])]/NUM_TRAINING_IMAGES*100):.2f}')\n# print(f'Testing most common {np.argmax(testingLabels[1])} = {(testingLabels[1][np.argmax(testingLabels[1])]/NUM_TEST_IMAGES*100):.2f}')\n# print('')\n# print('      Training    |  Testing')\n# print('      Count | %   |  Count | %')\n# for idx in range(104):\n#     print(f'{idx:<3} = {(trainingLabels[1][idx]):<6} {(trainingLabels[1][idx]/NUM_TRAINING_IMAGES*100):.2f} |  {(testingLabels[1][idx]):<6} {(testingLabels[1][idx]/NUM_TEST_IMAGES*100):.2f}')\n# print(f'Total Training = {sum(trainingLabels[1])}')\n# print(f'Total Testing = {sum(testingLabels[1])}')","metadata":{"id":"ehsQx7Ww4oRb","outputId":"d52b8b0f-6e57-4982-c4f6-1553c8b64a2d","execution":{"iopub.status.busy":"2022-11-03T05:11:53.116159Z","iopub.execute_input":"2022-11-03T05:11:53.116907Z","iopub.status.idle":"2022-11-03T05:11:53.128116Z","shell.execute_reply.started":"2022-11-03T05:11:53.116872Z","shell.execute_reply":"2022-11-03T05:11:53.127229Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import Conv2D, MaxPool2D, Dense, Flatten, Dropout\nfrom keras.regularizers import l1\ndef plot_results(all_history, sloss = \"sparse_categorical_crossentropy\", smetric = \"sparse_categorical_accuracy\"):\n  loss, val_loss, MSE, val_MSE = [], [], [], []\n  for history in all_history:\n    loss += history.history[sloss]\n    val_loss += history.history[f'val_{sloss}']\n    MSE += history.history[smetric]\n    val_MSE += history.history[f'val_{smetric}']\n\n  fig, ax = plt.subplots()\n  ax.plot(loss,label = 'train')\n  ax.plot(val_loss,label = 'test')\n  ax.set_title(sloss)\n  ax.legend(loc='upper right')\n\n  fig, ax = plt.subplots()\n  ax.plot(MSE,label = 'train')\n  ax.plot(val_MSE,label = 'test')\n  ax.set_title(smetric)\n  ax.legend(loc='lower right')\n  \ndef build_VGG_model(input_shape = (192,192,3), VGG = 4, start_units = 3, kernel = 8, pool_size = 2):\n    tf.keras.backend.clear_session\n    tf.random.set_seed(0)\n    model = tf.keras.models.Sequential()\n    model.add(Conv2D(filters = 2**(start_units), kernel_size = kernel, padding='same',\n                             input_shape=input_shape))\n    for i in range(VGG):\n        model.add(Conv2D(filters = 2**(i+start_units), kernel_size=kernel, padding='same', activation='relu'))\n        model.add(Conv2D(filters = 2**(i+start_units), kernel_size=kernel, padding='same', activation='relu'))\n        model.add(Dropout(0.1))\n        model.add(MaxPool2D(pool_size=pool_size, strides=2, padding = 'same'))\n    # Dense Layers\n    model.add(Flatten())\n    model.add(Dense(units = 4096, activation = 'relu'))\n    model.add(Dense(units = 4096, activation = 'relu'))\n    model.add(Dense(units = 104, activation = 'softmax'))\n    return model\n#VGGmodel = build_VGG_model(VGG=3, start_units = 4, kernel = 8, pool_size = 4);\nVGGmodel = tf.keras.models.load_model('../input/previous-model/saved_model/my_model/')\nVGGmodel.summary()","metadata":{"id":"8yqhtkoF4uVi","outputId":"060f577b-0568-4c7b-98bc-571ac1ffb8e9","execution":{"iopub.status.busy":"2022-11-03T05:11:53.129751Z","iopub.execute_input":"2022-11-03T05:11:53.130513Z","iopub.status.idle":"2022-11-03T05:12:06.744624Z","shell.execute_reply.started":"2022-11-03T05:11:53.130479Z","shell.execute_reply":"2022-11-03T05:12:06.743356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# opt = tf.keras.optimizers.Nadam(learning_rate = 0.00005)\n# VGGmodel.compile(optimizer = opt, \n#                  loss = 'sparse_categorical_crossentropy', \n#                  metrics='sparse_categorical_accuracy')\n# all_history = []\n# history = VGGmodel.fit(\n#     x = training_dataset,\n#     epochs = 8,\n#     steps_per_epoch = STEPS_PER_EPOCH,\n#     validation_data = validation_dataset\n# )\n# all_history.append(history)\n# plot_results(all_history, sloss = 'loss')\n# VGGmodel.save('./saved_model/my_model')","metadata":{"id":"wFErFgCP4wqY","outputId":"eb111dce-9083-454a-8e34-bc551ce032c5","execution":{"iopub.status.busy":"2022-11-03T05:12:06.746316Z","iopub.execute_input":"2022-11-03T05:12:06.746690Z","iopub.status.idle":"2022-11-03T05:12:06.751547Z","shell.execute_reply.started":"2022-11-03T05:12:06.746654Z","shell.execute_reply":"2022-11-03T05:12:06.750336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset = get_test_dataset(ordered = True)\ntest_images_ds = test_dataset.map(lambda image, idnum: image)\nprobabilities = VGGmodel.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)\n#print(predictions)\n\ntest_ids_ds = test_dataset.map(lambda image, idnum: idnum).unbatch()\ncount = 0\nfor _ in test_ids_ds:\n  count += 1\nprint(count)\ntest_ids = next(iter(test_ids_ds.batch(count))).numpy().astype('U')\n\nnp.savetxt(\n    'submission.csv',\n    np.rec.fromarrays([test_ids, predictions]),\n    fmt=['%s', '%d'],\n    delimiter=',',\n    header='id,label',\n    comments='',\n)\n\n# Look at the first few predictions\n!head submission.csv","metadata":{"id":"VcrjLP1gA6dF","outputId":"bcd01744-dbbe-46ab-c2ac-9e44330c5647","execution":{"iopub.status.busy":"2022-11-03T05:12:06.753359Z","iopub.execute_input":"2022-11-03T05:12:06.754191Z","iopub.status.idle":"2022-11-03T05:12:36.343784Z","shell.execute_reply.started":"2022-11-03T05:12:06.754151Z","shell.execute_reply":"2022-11-03T05:12:36.342475Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# test_dataset = get_test_dataset(ordered = True)\n# predictions = np.ones(7382)*67\n\n# test_ids_ds = test_dataset.map(lambda image, idnum: idnum).unbatch()\n# test_ids = next(iter(test_ids_ds.batch(7382))).numpy().astype('U')\n\n# np.savetxt(\n#     'submission.csv',\n#     np.rec.fromarrays([test_ids, predictions]),\n#     fmt=['%s', '%d'],\n#     delimiter=',',\n#     header='id,label',\n#     comments='',\n# )\n\n# # Look at the first few predictions\n# !head submission.csv","metadata":{"id":"Ygur7ascDg39","outputId":"14d24584-e1c5-4ecd-b1a4-d8c48db9b0a3","execution":{"iopub.status.busy":"2022-11-03T05:12:36.346051Z","iopub.execute_input":"2022-11-03T05:12:36.349147Z","iopub.status.idle":"2022-11-03T05:12:36.354140Z","shell.execute_reply.started":"2022-11-03T05:12:36.349110Z","shell.execute_reply":"2022-11-03T05:12:36.352844Z"},"trusted":true},"execution_count":null,"outputs":[]}]}