{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np \nimport pandas as pd \n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        continue\n        #print(os.path.join(dirname, filename))\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-07-26T18:20:56.761095Z","iopub.execute_input":"2021-07-26T18:20:56.761619Z","iopub.status.idle":"2021-07-26T18:20:56.807513Z","shell.execute_reply.started":"2021-07-26T18:20:56.761574Z","shell.execute_reply":"2021-07-26T18:20:56.806588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nimport seaborn as sns\n\nimport keras\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Conv2D , MaxPool2D , Flatten , Dropout \nfrom keras.preprocessing.image import ImageDataGenerator\nfrom keras.optimizers import Adam\n\nfrom sklearn.metrics import classification_report,confusion_matrix\nfrom kaggle_datasets import KaggleDatasets\nimport tensorflow as tf\nimport pathlib\n\nimport cv2\nimport os\nimport math\nimport re","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:20:56.808739Z","iopub.execute_input":"2021-07-26T18:20:56.809131Z","iopub.status.idle":"2021-07-26T18:20:59.881828Z","shell.execute_reply.started":"2021-07-26T18:20:56.809102Z","shell.execute_reply":"2021-07-26T18:20:59.880772Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# detect and init the TPU\ntry:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept:\n    tpu_strategy = tf.distribute.get_strategy()\nprint(\"Device:\", tpu.master())\ntpu_strategy.num_replicas_in_sync","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:20:59.884129Z","iopub.execute_input":"2021-07-26T18:20:59.884597Z","iopub.status.idle":"2021-07-26T18:21:05.204748Z","shell.execute_reply.started":"2021-07-26T18:20:59.884552Z","shell.execute_reply":"2021-07-26T18:21:05.203503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE\nGCS_DS_PATH = KaggleDatasets().get_gcs_path('tpu-getting-started')\n\nGCS_PATH = GCS_DS_PATH + '/tfrecords-jpeg-512x512'\nAUTO = tf.data.experimental.AUTOTUNE\nIMAGE_SIZE = [512, 512] \nTRAINING_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/train/*.tfrec')\nVALIDATION_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(GCS_PATH + '/test/*.tfrec')\n\nBATCH_SIZE = 16 * tpu_strategy.num_replicas_in_sync","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:21:05.206705Z","iopub.execute_input":"2021-07-26T18:21:05.207008Z","iopub.status.idle":"2021-07-26T18:21:05.777328Z","shell.execute_reply.started":"2021-07-26T18:21:05.206979Z","shell.execute_reply":"2021-07-26T18:21:05.776289Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def decode_image(image_data):\n    image =tf.image.decode_jpeg(image_data, channels=3)\n    image =tf.image.resize(image,[*IMAGE_SIZE])  # resize image to the dimension needed for the pretrained model\n    image =tf.cast(image, tf.float32) /255.0\n    image = tf.reshape(image,[*IMAGE_SIZE, 3])\n    return image\n\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([],tf.string), # tf.string means bytestring\n        \"class\": tf.io.FixedLenFeature([],tf.int64),\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_image(example[\"image\"])\n    label = tf.cast(example[\"class\"], tf.int32)\n    return image,label\n\ndef read_unlabeled_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([],tf.string),\n        \"id\": tf.io.FixedLenFeature([],tf.string),\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_image(example[\"image\"])\n    idnum = example[\"id\"]\n    return image,idnum\n\ndef load_dataset(filenames, labeled=True, ordered=False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    \n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=AUTO)\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord if labeled else read_unlabeled_tfrecord, num_parallel_calls=AUTO)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:21:05.778776Z","iopub.execute_input":"2021-07-26T18:21:05.779241Z","iopub.status.idle":"2021-07-26T18:21:05.791427Z","shell.execute_reply.started":"2021-07-26T18:21:05.779199Z","shell.execute_reply":"2021-07-26T18:21:05.790362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef get_training_dataset():\n    dataset = load_dataset(TRAINING_FILENAMES, labeled=True)\n   # dataset = dataset.map(data_augment, num_parallel_calls=AUTO)\n    dataset = dataset.repeat() # repeats for several epochs\n    dataset = dataset.shuffle(buffer_size=2048)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO) # prefetch next batch while training\n    return dataset\n\ndef get_validation_dataset(ordered=False):\n    dataset = load_dataset(VALIDATION_FILENAMES, labeled=True, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.cache()\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\ndef get_test_dataset(ordered = False):\n    dataset = load_dataset(TEST_FILENAMES, labeled=False, ordered=ordered)\n    dataset = dataset.batch(BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\ndef count_data_items(filenames):\n    # the number of data items in the name of the .tfrec \n    n  = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)\n\nfTrainImages = count_data_items(TRAINING_FILENAMES)\nfValidationImages = count_data_items(VALIDATION_FILENAMES)\nfTestImages = count_data_items(TEST_FILENAMES)\nprint(f\"{fTrainImages} training images, {fValidationImages} validation images, {fTestImages} test images \")\n","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:21:05.793087Z","iopub.execute_input":"2021-07-26T18:21:05.793572Z","iopub.status.idle":"2021-07-26T18:21:05.807215Z","shell.execute_reply.started":"2021-07-26T18:21:05.793531Z","shell.execute_reply":"2021-07-26T18:21:05.806245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 16* tpu_strategy.num_replicas_in_sync\ntrain_ds = get_training_dataset()\nval_ds = get_validation_dataset()\ntest_ds = get_test_dataset()\n\nprint(\"Training: \", train_ds)\nprint(\"Validation: \", val_ds)\nprint(\"Test: \", test_ds)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:21:05.8085Z","iopub.execute_input":"2021-07-26T18:21:05.808781Z","iopub.status.idle":"2021-07-26T18:21:06.28105Z","shell.execute_reply.started":"2021-07-26T18:21:05.808754Z","shell.execute_reply":"2021-07-26T18:21:06.280118Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def batch_to_numpy_images_and_labels(data):\n    images,labels = data\n    numpy_images = images.numpy()\n    numpy_labels = labels.numpy()\n    if numpy_labels.dtype == object:\n        numpy_labels = [ None for _ in enumerate(numpy_images)]\n    return numpy_images, numpy_labels\n\ndef display_one_flower(image, title, subplot, red=False, titlesize=16):\n    plt.subplot(*subplot)\n    plt.axis('off')\n    plt.imshow(image)\n    if len(title)>0:\n        plt.title(title, fontsize=int(titlesize) if not red else int(titlesize/1.2),\n                  color= 'red' if red else 'black',\n                fontdict={'verticalalignment':'center'}, \n                  pad=int(titlesize/1.5))\n    return (subplot[0],subplot[1],subplot[2]+1)\n\ndef display_batch_of_images(databatch,predictions=None):\n    images,labels = batch_to_numpy_images_and_labels(databatch)\n    if labels is None:\n        labels  = [None for _ in enumerate(images)]\n        \n    rows = int(math.sqrt(len(images)))\n    cols = len(images)//rows\n    \n    FIGSIZE = 13.0\n    SPACING = 0.1\n    subplot = (rows,cols,1)\n    if(rows < cols):\n        plt.figure(figsize=(FIGSIZE, FIGSIZE/cols*rows))\n    else:\n        plt.figure(figsize=(FIGSIZE/rows*cols, FIGSIZE))\n    \n    #display \n    for i, (image,label) in enumerate(zip(images[:rows*cols], labels[:rows*cols])):\n        title = '{}'.format(label)\n        correct = True\n        if predictions is not None:\n            title, correct = title_from_label_and_target(predictions[i],label)\n        dynamic_titlesize = FIGSIZE*SPACING/max(rows,cols)*40+3\n        subplot = display_one_flower(image,title,subplot, not correct,\n                                    titlesize = dynamic_titlesize)\n    \n    plt.tight_layout()\n    if label is None and predictions is None:\n        plt.subplots_adjust(wspace=0,hspace=0)\n    else:\n        plt.subplots_adjust(wspace=SPACING, hspace=SPACING)\n    plt.show()\n               ","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:21:06.28401Z","iopub.execute_input":"2021-07-26T18:21:06.284482Z","iopub.status.idle":"2021-07-26T18:21:06.300681Z","shell.execute_reply.started":"2021-07-26T18:21:06.284435Z","shell.execute_reply":"2021-07-26T18:21:06.299233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_iter = iter(train_ds.unbatch().batch(20))","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:21:06.302935Z","iopub.execute_input":"2021-07-26T18:21:06.303373Z","iopub.status.idle":"2021-07-26T18:21:06.329086Z","shell.execute_reply.started":"2021-07-26T18:21:06.303329Z","shell.execute_reply":"2021-07-26T18:21:06.328031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"one_batch = next(ds_iter)\ndisplay_batch_of_images(one_batch)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:21:06.330197Z","iopub.execute_input":"2021-07-26T18:21:06.330774Z","iopub.status.idle":"2021-07-26T18:21:10.209766Z","shell.execute_reply.started":"2021-07-26T18:21:06.330733Z","shell.execute_reply":"2021-07-26T18:21:10.207924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.callbacks import EarlyStopping, ReduceLROnPlateau\nearly_stopping = EarlyStopping(monitor='val_loss', patience=3, restore_best_weights=True)\nlrr = ReduceLROnPlateau(monitor='val_loss',patience=3,verbose=1,factor=0.5, min_lr=0.00001)\n\nSTEPS_PER_EPOCH = fTrainImages // BATCH_SIZE","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:21:10.210936Z","iopub.execute_input":"2021-07-26T18:21:10.211375Z","iopub.status.idle":"2021-07-26T18:21:10.21639Z","shell.execute_reply.started":"2021-07-26T18:21:10.211336Z","shell.execute_reply":"2021-07-26T18:21:10.215522Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with tpu_strategy.scope():\n    img_adjust_layer = tf.keras.layers.Lambda(lambda data: tf.keras.applications.xception.preproces_input(tf.cast(data,tf.float32)), input_shape=[*IMAGE_SIZE,3])\n    xce_pretrained_model = tf.keras.applications.Xception(weights='imagenet',include_top=False)    \n    xce_pretrained_model.trainable = True\n    model = tf.keras.Sequential([\n        xce_pretrained_model,\n        tf.keras.layers.GlobalAveragePooling2D(),\n        tf.keras.layers.Dense(104, activation='softmax')\n    ])","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:21:10.2177Z","iopub.execute_input":"2021-07-26T18:21:10.217995Z","iopub.status.idle":"2021-07-26T18:21:20.167437Z","shell.execute_reply.started":"2021-07-26T18:21:10.217965Z","shell.execute_reply":"2021-07-26T18:21:20.166463Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer='adam', loss=tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n              metrics=['accuracy'])\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:21:20.170921Z","iopub.execute_input":"2021-07-26T18:21:20.171208Z","iopub.status.idle":"2021-07-26T18:21:20.227667Z","shell.execute_reply.started":"2021-07-26T18:21:20.17118Z","shell.execute_reply":"2021-07-26T18:21:20.226458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(train_ds, epochs=5, steps_per_epoch=STEPS_PER_EPOCH, callbacks=[early_stopping, lrr],\n         validation_data=val_ds)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:21:20.228942Z","iopub.execute_input":"2021-07-26T18:21:20.229284Z","iopub.status.idle":"2021-07-26T18:25:07.43007Z","shell.execute_reply.started":"2021-07-26T18:21:20.229251Z","shell.execute_reply":"2021-07-26T18:25:07.429284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with tpu_strategy.scope():\n    model2 = tf.keras.Sequential([\n        tf.keras.layers.Conv2D(32,(3, 3), activation='relu', input_shape=(512,512,3)),\n        tf.keras.layers.MaxPooling2D((2,2)),\n        tf.keras.layers.Conv2D(64,(3, 3), activation='relu'),\n        tf.keras.layers.MaxPooling2D((2,2)),\n        tf.keras.layers.Conv2D(64,(3, 3), activation='relu'),\n        tf.keras.layers.Flatten(),\n        tf.keras.layers.Dense(104, activation='softmax'),\n    ])\n   ","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:25:07.431128Z","iopub.execute_input":"2021-07-26T18:25:07.431437Z","iopub.status.idle":"2021-07-26T18:25:09.669685Z","shell.execute_reply.started":"2021-07-26T18:25:07.431399Z","shell.execute_reply":"2021-07-26T18:25:09.66857Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model2.compile(optimizer='adam',\n               loss=tf.keras.losses.SparseCategoricalCrossentropy(from_logits=True),\n               metrics=['accuracy'])\n\nmodel2.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:25:09.671159Z","iopub.execute_input":"2021-07-26T18:25:09.671611Z","iopub.status.idle":"2021-07-26T18:25:09.710264Z","shell.execute_reply.started":"2021-07-26T18:25:09.67157Z","shell.execute_reply":"2021-07-26T18:25:09.709135Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model2.fit(train_ds, epochs=10,\n           steps_per_epoch=STEPS_PER_EPOCH, validation_data=val_ds, callbacks=[early_stopping, lrr])","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:25:09.711615Z","iopub.execute_input":"2021-07-26T18:25:09.711945Z","iopub.status.idle":"2021-07-26T18:27:50.47458Z","shell.execute_reply.started":"2021-07-26T18:25:09.711917Z","shell.execute_reply":"2021-07-26T18:27:50.473477Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = get_test_dataset(ordered=True)\ntest_images_ds = test_ds.map(lambda image, idnum: image)\ntest_ids_ds = test_ds.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(fTestImages))).numpy().astype('U')\nproba = model.predict(test_images_ds)\n\npredictions = np.argmax(proba, axis=-1)\n\nnp.savetxt('submission.csv',\n          np.rec.fromarrays([test_ids,predictions]),\n           fmt=['%s', '%d'],\n           delimiter=',',\n           header='id,label',\n           comments='',)","metadata":{"execution":{"iopub.status.busy":"2021-07-26T18:27:50.475938Z","iopub.execute_input":"2021-07-26T18:27:50.476229Z","iopub.status.idle":"2021-07-26T18:28:13.119684Z","shell.execute_reply.started":"2021-07-26T18:27:50.476202Z","shell.execute_reply":"2021-07-26T18:28:13.118827Z"},"trusted":true},"execution_count":null,"outputs":[]}]}