{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import math, re, os\nimport tensorflow as tf # deep learning\nimport pandas as pd\nimport numpy as np # linear algebra\nfrom matplotlib import pyplot as plt\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.metrics import f1_score, precision_score, recall_score, confusion_matrix\nprint(\"Tensorflow version \", tf.__version__)\nAUTO = tf.data.experimental.AUTOTUNE","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### TPU Detection","execution_count":null},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nelse:\n    print('No TPU detected, look to your right under the \"Accelerator\" tab and switch to \"TPU v3-8\"')\nprint('REPLICAS: ', strategy.num_replicas_in_sync)","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Create *image_dataset* using train.csv converted into a Tensorflow Dataset with (image,label)","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# Get the Google Cloud mirror path for this Kaggle dataset\nGCS_DS_PATH = KaggleDatasets().get_gcs_path()\n!gsutil ls $GCS_DS_PATH","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"CLASS_NAMES = pd.read_csv(GCS_DS_PATH+'/labels.csv')\ntrain_labels = pd.read_csv('/kaggle/input/imet-2020-fgvc7/train.csv')\nids = train_labels.pop('id')\nattributes = train_labels.pop('attribute_ids')\ntrain_df = tf.data.Dataset.from_tensor_slices((ids,attributes))\n\n# Display tensor values\nfor index, tensor in train_df.enumerate().as_numpy_iterator():\n    if index > 5:\n        break\n    print('Image file path:', tensor[0], '  Labels:', tensor[1])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def decode_jpeg(filename, label):\n    bits = tf.io.read_file(GCS_DS_PATH+'/train/'+filename+'.png')\n    image = tf.image.decode_jpeg(bits)\n    image = tf.image.resize_with_crop_or_pad(image,300,300)\n    if tf.shape(image)[2] == 1:\n        image = tf.image.grayscale_to_rgb(image)\n    return image, label\nimage_ds = train_df.map(decode_jpeg)\n\n# Display images and labels\nfor index, tensor in image_ds.enumerate().as_numpy_iterator():\n    if index > 5:\n        break\n    print('Image Shape:',tensor[0].shape, '   Labels:', tensor[1])\n#     if index == 3:\n#         print('Image Data Example: ',tensor[0])","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Normalize Image Data\n\nAs we can see, our images are not all the same size, so we need to normalize them.","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"# TPU_CORES = strategy.num_replicas_in_sync\n# IMAGE_SIZE = [512,512]\n# EPOCHS = 12\nBATCH_SIZE = 16\ndef prepare_for_training(ds, cache=True, shuffle_buffer_size=1000):\n  # This is a small dataset, only load it once, and keep it in memory.\n  # use `.cache(filename)` to cache preprocessing work for datasets that don't\n  # fit in memory.\n  if cache:\n    if isinstance(cache, str):\n      ds = ds.cache(cache)\n    else:\n      ds = ds.cache()\n\n  ds = ds.shuffle(buffer_size=shuffle_buffer_size)\n\n  # Repeat forever\n  ds = ds.repeat()\n\n  ds = ds.batch(BATCH_SIZE)\n\n  # `prefetch` lets the dataset fetch batches in the background while the model\n  # is training.\n  ds = ds.prefetch(buffer_size=AUTO)\n\n  return ds\ntrain_ds = prepare_for_training(image_ds)\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"# TAKES VERY LONG TIME TO RUN, 10+ minutes\n# image_batch, label_batch = next(iter(train_ds))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def get_Title(labels,index):\n    title = ''\n    tag_array = labels[index].split()\n    for index, tag in enumerate(tag_array):\n        tag_array[index] = CLASS_NAMES.iloc[int(tag)].attribute_name\n    return ' '.join(tag_array)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def show_batch(image_batch, label_batch):\n  plt.figure(figsize=(10,10))\n  for n in range(16):\n      ax = plt.subplot(5,5,n+1)\n      plt.imshow(image_batch[n])\n      plt.title(get_Title(label_batch,n))\n      plt.axis('off')\nshow_batch(image_batch.numpy(), label_batch.numpy())","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"### Create Multilabel Classification Model","execution_count":null},{"metadata":{"trusted":true},"cell_type":"code","source":"MobileNetV2 = tf.keras.applications.MobileNetV2(input_shape=[300,300,3], include_top=False)\nMobileNetV2.trainable = False\n\nmodel = tf.keras.Sequential([\n    MobileNetV2,\n    tf.keras.layers.Conv2D(kernel_size=3, filters=24, padding=\"same\", activation=\"relu\"),\n    tf.keras.layers.Conv2D(kernel_size=3, filters=24, padding=\"same\", activation=\"relu\"),\n    tf.keras.layers.MaxPooling2D(pool_size=2),\n    tf.keras.layers.Conv2D(kernel_size=3, filters=12, padding=\"same\", activation=\"relu\"),\n    tf.keras.layers.MaxPooling2D(pool_size=2),\n    tf.keras.layers.Conv2D(kernel_size=3, filters=6, padding=\"same\", activation=\"relu\"),\n    tf.keras.layers.Flatten(),\n    tf.keras.layers.Dense(5, activation=\"softmax\")\n])\n\nmodel.compile(\n    optimizer=\"adam\",\n    loss=\"categorical_crossentropy\",\n    metrics=[\"accuracy\"]\n)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"model.fit(image_batch,label_batch,batch_size=16,epochs=1)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}