{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.7.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":30301,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    \n    print(dirname)\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-11-10T14:17:59.709403Z","iopub.execute_input":"2022-11-10T14:17:59.709660Z","iopub.status.idle":"2022-11-10T14:17:59.825846Z","shell.execute_reply.started":"2022-11-10T14:17:59.709636Z","shell.execute_reply":"2022-11-10T14:17:59.825154Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nimport matplotlib.pyplot as plt\nfrom kaggle_datasets import KaggleDatasets\nfrom tensorflow import keras\nfrom functools import partial\nprint(\"Tensorflow version \" + tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2022-11-10T14:17:45.514656Z","iopub.execute_input":"2022-11-10T14:17:45.515661Z","iopub.status.idle":"2022-11-10T14:17:53.646719Z","shell.execute_reply.started":"2022-11-10T14:17:45.515523Z","shell.execute_reply":"2022-11-10T14:17:53.645190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Device:', tpu.master())\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept:\n    strategy = tf.distribute.get_strategy()\nprint('Number of replicas:', strategy.num_replicas_in_sync)\n","metadata":{"execution":{"iopub.status.busy":"2022-11-10T14:18:03.039214Z","iopub.execute_input":"2022-11-10T14:18:03.039726Z","iopub.status.idle":"2022-11-10T14:18:09.176824Z","shell.execute_reply.started":"2022-11-10T14:18:03.039689Z","shell.execute_reply":"2022-11-10T14:18:09.176100Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## **Some Important Parameters**","metadata":{}},{"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE\nGCS_PATH = KaggleDatasets().get_gcs_path()\nBATCH_SIZE = 16 * strategy.num_replicas_in_sync\nIMAGE_SIZE = [224, 224]\nEPOCHS = 25\nSTEPS_PER_E = 12753//7382","metadata":{"execution":{"iopub.status.busy":"2022-11-10T14:20:25.072889Z","iopub.execute_input":"2022-11-10T14:20:25.073183Z","iopub.status.idle":"2022-11-10T14:20:25.426000Z","shell.execute_reply.started":"2022-11-10T14:20:25.073156Z","shell.execute_reply":"2022-11-10T14:20:25.424561Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# **Load Your Dataset**","metadata":{}},{"cell_type":"code","source":"# DECODE the image, as they are in .tfrec file extension\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image,tf.float32) / 255.0\n    image = tf.reshape(image,[*IMAGE_SIZE,3])\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        'image': tf.io.FixedLenFeature([],tf.string),\n        'class': tf.io.FixedLenFeature([],tf.int64) # shape [] means single element\n    }\n    example = tf.io.parse_single_example(example,LABELED_TFREC_FORMAT)\n    imag = decode_image(example['image'])\n    label = tf.cast(example['class'],tf.int32)\n    return imag,label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC = {\n        'image': tf.io.FixedLenFeature([],tf.string32),\n        'id': tf.io.FixedLenFeature([],tf.string32)\n    }\n    example = tf.io.parse_single_example(example,UNLABELED_TFREC)\n    img = decode_image(example['image'])\n    ids =  example['id']\n    return img,ids\n\ndef load_dataset(filename, labeled = True, ordered = False):\n    ignore_order = tf.data.Options()\n    \n    if not ordered :\n        ignore_order.experimental_deterministic = False # disabled order, increases speed\n        \n    dataset = tf.data.TFRecordDataset(filename)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord)\n    \n    return dataset\n\ndef get_training_dataset():\n    \n    dataset = load_dataset(tf.io.gfile.glob(GCS_PATH+'*/tfrecords-jpeg-224x224/train/*.tfrec'))\n    dataset = dataset.repeat() # the training dataset must repeat for several epochs\n    \n    # shuffle(# elements to sample from the dataset)\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    return dataset\n\n\ndef get_validation_dataset():\n    \n    dataset = load_dataset(tf.io.gfile.glob(GCS_PATH+'*/tfrecords-jpeg-224x224/validation/*.tfrec'))\n    \n    # shuffle(# elements to sample from the dataset)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n\n    return dataset\n\n\ndef get_testing_dataset(order = False):\n    \n    dataset = load_dataset(tf.io.gfile.glob(GCS_PATH+'*/tfrecords-jpeg-224x224/test/*.tfrec'),labeled = True, ordered = order)\n    \n    # shuffle(# elements to sample from the dataset)\n    dataset = dataset.batch(BATCH_SIZE)\n\n    return dataset\n\ntrain_ds = get_training_dataset()\nvalid_ds = get_validation_dataset()","metadata":{"execution":{"iopub.status.busy":"2022-11-10T14:18:16.931351Z","iopub.execute_input":"2022-11-10T14:18:16.931627Z","iopub.status.idle":"2022-11-10T14:18:17.244287Z","shell.execute_reply.started":"2022-11-10T14:18:16.931600Z","shell.execute_reply":"2022-11-10T14:18:17.242893Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import Dense,Flatten,Dropout,BatchNormalization,Conv2D,MaxPool2D,Input\nfrom tensorflow.keras import Sequential\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau\n# from keras ","metadata":{"execution":{"iopub.status.busy":"2022-11-10T14:18:54.238355Z","iopub.execute_input":"2022-11-10T14:18:54.238842Z","iopub.status.idle":"2022-11-10T14:18:54.245647Z","shell.execute_reply.started":"2022-11-10T14:18:54.238815Z","shell.execute_reply":"2022-11-10T14:18:54.243583Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = Sequential((\n    Input([*IMAGE_SIZE,3]),\n    BatchNormalization(),\n    Conv2D(64,3,activation='relu'),\n    MaxPool2D(),\n    Dropout(0.2),\n    BatchNormalization(),\n    Conv2D(128,3,activation='relu'),\n    MaxPool2D(),\n    Dropout(0.2),\n    BatchNormalization(),\n    Conv2D(192,3,activation='relu'),\n    MaxPool2D(padding='same'),\n    Dropout(0.2),\n    BatchNormalization(),\n    Conv2D(64,3,activation='relu'),\n    MaxPool2D(padding='same'),\n    Dropout(0.2),\n    Flatten(),\n    Dense(256,'selu',kernel_initializer='he_normal',),\n    Dropout(0.2),\n    Dense(104,'softmax')\n\n))\n\nmodel.compile(optimizer='adam',loss= 'sparse_categorical_crossentropy',metrics=['sparse_categorical_accuracy'])","metadata":{"execution":{"iopub.status.busy":"2022-11-10T14:18:56.807777Z","iopub.execute_input":"2022-11-10T14:18:56.808272Z","iopub.status.idle":"2022-11-10T14:18:57.038614Z","shell.execute_reply.started":"2022-11-10T14:18:56.808236Z","shell.execute_reply":"2022-11-10T14:18:57.037535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-11-10T14:18:58.617953Z","iopub.execute_input":"2022-11-10T14:18:58.618248Z","iopub.status.idle":"2022-11-10T14:18:58.637816Z","shell.execute_reply.started":"2022-11-10T14:18:58.618223Z","shell.execute_reply":"2022-11-10T14:18:58.636365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"reducelr = ReduceLROnPlateau('val_sparse_categorical_accuracy',0.2,2,1)\nhistory = model.fit(train_ds,epochs = 10,steps_per_epoch=STEPS_PER_E,validation_data=valid_ds)","metadata":{"execution":{"iopub.status.busy":"2022-11-10T14:20:44.577300Z","iopub.execute_input":"2022-11-10T14:20:44.577578Z","iopub.status.idle":"2022-11-10T14:21:17.292691Z","shell.execute_reply.started":"2022-11-10T14:20:44.577552Z","shell.execute_reply":"2022-11-10T14:21:17.291616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():    \n    \n    pretrained_model = tf.keras.applications.VGG16(weights='imagenet',\n                                                   include_top=False, \n                                                   input_shape=[*IMAGE_SIZE, 3])\n\n    pretrained_model.trainable = True\n    \n    model = tf.keras.Sequential([\n        pretrained_model,\n        tf.keras.layers.GlobalAveragePooling2D(),\n        tf.keras.layers.Dense(104, activation='softmax')\n    ])\n\n    \nmodel.compile(\n    optimizer='adam',\n    loss = 'sparse_categorical_crossentropy',\n    metrics=['sparse_categorical_accuracy']\n)\n\nhistorical = model.fit(train_ds, \n          steps_per_epoch=STEPS_PER_E, \n          epochs=EPOCHS, \n          validation_data=valid_ds,\n          )\n# ar@006786","metadata":{"execution":{"iopub.status.busy":"2022-11-10T14:24:12.390087Z","iopub.execute_input":"2022-11-10T14:24:12.390373Z","iopub.status.idle":"2022-11-10T14:24:58.138373Z","shell.execute_reply.started":"2022-11-10T14:24:12.390346Z","shell.execute_reply":"2022-11-10T14:24:58.136615Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}