{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport tensorflow as tf\nfrom tensorflow import keras\nimport matplotlib.pyplot as plt\nimport IPython.display as display\nfrom io import BytesIO\nfrom tensorflow.keras import layers\nfrom scipy import misc\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n_ = []\n\nfor dirname, __, filenames in os.walk('/kaggle/input/tpu-getting-started'):\n    for filename in filenames:\n        _.append(os.path.join(dirname, filename))","metadata":{"execution":{"iopub.status.busy":"2023-06-03T05:52:30.410036Z","iopub.execute_input":"2023-06-03T05:52:30.410417Z","iopub.status.idle":"2023-06-03T05:52:30.429397Z","shell.execute_reply.started":"2023-06-03T05:52:30.410382Z","shell.execute_reply":"2023-06-03T05:52:30.428353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"_.pop(0)\nval = []\ntrain = []\ntest = []\nfor i in _:\n    key = str(i[57:59])\n    if key == 'va':\n        val.append(i)\n    elif key == 'tr':\n        train.append(i)\n    else:\n        test.append(i)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T05:52:30.432750Z","iopub.execute_input":"2023-06-03T05:52:30.433029Z","iopub.status.idle":"2023-06-03T05:52:30.439498Z","shell.execute_reply.started":"2023-06-03T05:52:30.433004Z","shell.execute_reply":"2023-06-03T05:52:30.438505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dataset = tf.data.TFRecordDataset(train)\nval_dataset = tf.data.TFRecordDataset(val)\ntest_dataset = tf.data.TFRecordDataset(test)\n(train_dataset, val_dataset, test_dataset)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T05:52:30.449775Z","iopub.execute_input":"2023-06-03T05:52:30.450669Z","iopub.status.idle":"2023-06-03T05:52:30.485758Z","shell.execute_reply.started":"2023-06-03T05:52:30.450637Z","shell.execute_reply":"2023-06-03T05:52:30.484775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for raw_record in train_dataset.take(1):\n    example = tf.train.Example()\n    example.ParseFromString(raw_record.numpy())\n    #print(example)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T05:52:30.487548Z","iopub.execute_input":"2023-06-03T05:52:30.488240Z","iopub.status.idle":"2023-06-03T05:52:30.509159Z","shell.execute_reply.started":"2023-06-03T05:52:30.488195Z","shell.execute_reply":"2023-06-03T05:52:30.508313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_SIZE = 224 # 192, 224, 331, 512\ntrain_resize_and_rescale = tf.keras.Sequential([\n    layers.Resizing(IMG_SIZE, IMG_SIZE),\n    layers.Rescaling(1./255),\n    layers.RandomFlip(\"horizontal_and_vertical\"),\n    layers.RandomRotation(0.2),\n    layers.RandomCrop(IMG_SIZE, IMG_SIZE, seed=None)\n])\nval_and_test_resize_and_rescale = tf.keras.Sequential([\n    layers.Resizing(IMG_SIZE, IMG_SIZE),\n    layers.Rescaling(1./255)])\ndef decode_train_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32)# / 255.0  # convert image to floats in [0, 1] range\n    #image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    image = train_resize_and_rescale(image)\n    return image\ndef decode_val_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32)# / 255.0  # convert image to floats in [0, 1] range\n    #image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    image = val_and_test_resize_and_rescale(image)\n    return image\ndef decode_test_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels=3)\n    image = tf.cast(image, tf.float32)# / 255.0  # convert image to floats in [0, 1] range\n    #image = tf.reshape(image, [*IMAGE_SIZE, 3])\n    image = val_and_test_resize_and_rescale(image)\n    return image\ndef read_train_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string means bytestring\n        \"class\": tf.io.FixedLenFeature([], tf.int64),  # shape [] means single element\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_train_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label # returns a dataset of (image, label) pairs\n\ndef read_val_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string means bytestring\n        \"class\": tf.io.FixedLenFeature([], tf.int64),  # shape [] means single element\n    }\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    image = decode_val_image(example['image'])\n    label = tf.cast(example['class'], tf.int32)\n    return image, label # returns a dataset of (image, label) pairs\ndef read_test_tfrecord(example):\n    UNLABELED_TFREC_FORMAT = {\n        \"image\": tf.io.FixedLenFeature([], tf.string), # tf.string means bytestring\n        \"id\": tf.io.FixedLenFeature([], tf.string),  # shape [] means single element\n        # class is missing, this competitions's challenge is to predict flower classes for the test dataset\n    }\n    example = tf.io.parse_single_example(example, UNLABELED_TFREC_FORMAT)\n    image = decode_test_image(example['image'])\n    idnum = example['id']\n    return image, idnum # returns a dataset of image(s)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T05:52:30.514273Z","iopub.execute_input":"2023-06-03T05:52:30.514564Z","iopub.status.idle":"2023-06-03T05:52:30.540827Z","shell.execute_reply.started":"2023-06-03T05:52:30.514520Z","shell.execute_reply":"2023-06-03T05:52:30.539809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"parsed_train_dataset = train_dataset.map(read_train_tfrecord).batch(64).shuffle(64)\nparsed_val_dataset = val_dataset.map(read_val_tfrecord).batch(64).shuffle(32)\nparsed_test_dataset = test_dataset.map(read_test_tfrecord).batch(64)\n(parsed_train_dataset, parsed_val_dataset, parsed_test_dataset)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T05:52:30.542377Z","iopub.execute_input":"2023-06-03T05:52:30.542728Z","iopub.status.idle":"2023-06-03T05:52:31.093191Z","shell.execute_reply.started":"2023-06-03T05:52:30.542697Z","shell.execute_reply":"2023-06-03T05:52:31.092116Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"i = 0\nfor image_features in parsed_train_dataset.take(10):\n    image_raw = image_features[1][0].numpy()#*255\n    #_ = display.display(display.Image(data=image_raw))\n    print(image_raw)\n    ","metadata":{"execution":{"iopub.status.busy":"2023-06-03T05:52:31.095464Z","iopub.execute_input":"2023-06-03T05:52:31.095856Z","iopub.status.idle":"2023-06-03T05:52:58.755673Z","shell.execute_reply.started":"2023-06-03T05:52:31.095823Z","shell.execute_reply":"2023-06-03T05:52:58.754695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"checkpoint_dir = \"./ckpt1\"\nif not os.path.exists(checkpoint_dir):\n    os.makedirs(checkpoint_dir)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T05:52:58.757729Z","iopub.execute_input":"2023-06-03T05:52:58.758772Z","iopub.status.idle":"2023-06-03T05:52:58.763942Z","shell.execute_reply.started":"2023-06-03T05:52:58.758727Z","shell.execute_reply":"2023-06-03T05:52:58.762991Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_model():\n    inputs = keras.Input(shape=(224, 224, 3))\n    x = tf.keras.applications.ResNet50(\n        include_top=False,\n        weights=None,\n        input_tensor=inputs,\n        input_shape=(224, 224, 3),\n        pooling=None,\n    )(inputs)\n    x = layers.Conv2D(4096, 5, activation=\"relu\")(x)\n    x = layers.GlobalAveragePooling2D()(x)\n    x = layers.Flatten()(x)\n    x = layers.Dropout(0.6)(x)\n    outputs = layers.Dense(104, activation='softmax')(x)\n    model = keras.Model(inputs, outputs, name=\"model\")\n    model.summary()\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2023-06-03T05:52:58.766648Z","iopub.execute_input":"2023-06-03T05:52:58.767293Z","iopub.status.idle":"2023-06-03T05:52:58.775659Z","shell.execute_reply.started":"2023-06-03T05:52:58.767261Z","shell.execute_reply":"2023-06-03T05:52:58.774725Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_or_restore_model():\n    # Either restore the latest model, or create a fresh one\n    # if there is no checkpoint available.\n    checkpoints = [checkpoint_dir + \"/\" + name for name in os.listdir(checkpoint_dir)]\n    if checkpoints:\n        latest_checkpoint = max(checkpoints, key=os.path.getctime)\n        print(\"Restoring from\", latest_checkpoint)\n        return keras.models.load_model(latest_checkpoint)\n    else:\n        print(\"Restoring from\", '/kaggle/input/ckpt-flowers/ckpt-loss=0.55')\n        return keras.models.load_model('/kaggle/input/ckpt-flowers/ckpt-loss=0.55')\n    #print(\"Creating a new model\")\n    return get_model()\n\nmodel = make_or_restore_model()","metadata":{"execution":{"iopub.status.busy":"2023-06-03T05:52:58.777100Z","iopub.execute_input":"2023-06-03T05:52:58.778061Z","iopub.status.idle":"2023-06-03T05:53:18.570440Z","shell.execute_reply.started":"2023-06-03T05:52:58.778029Z","shell.execute_reply":"2023-06-03T05:53:18.569425Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MAX_EPOCHS = 30\n\ndef compile_and_fit(model, patience=9):\n    early_stopping = tf.keras.callbacks.EarlyStopping(monitor='val_loss',\n                                                      patience=patience,\n                                                      mode='min')\n\n    \n    # This callback saves a SavedModel every 100 batches.\n    # We include the training loss in the saved model name.\n    save_version = tf.keras.callbacks.ModelCheckpoint(filepath=checkpoint_dir + \"/ckpt-loss={loss:.2f}\", \n                                                      save_freq=6*798)\n    \n    model.compile(optimizer='adam',\n              loss=tf.keras.losses.SparseCategoricalCrossentropy(),\n              metrics=['accuracy'])\n    \n    history = model.fit(parsed_train_dataset, epochs=MAX_EPOCHS,\n                        validation_data=parsed_val_dataset,\n                        callbacks=[early_stopping, save_version])\n    return history","metadata":{"execution":{"iopub.status.busy":"2023-06-03T05:53:18.578031Z","iopub.execute_input":"2023-06-03T05:53:18.578350Z","iopub.status.idle":"2023-06-03T05:53:18.586000Z","shell.execute_reply.started":"2023-06-03T05:53:18.578322Z","shell.execute_reply":"2023-06-03T05:53:18.584592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"keras.utils.plot_model(model, \"model.png\", show_shapes=True)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T05:53:18.588370Z","iopub.execute_input":"2023-06-03T05:53:18.589625Z","iopub.status.idle":"2023-06-03T05:53:18.807554Z","shell.execute_reply.started":"2023-06-03T05:53:18.589592Z","shell.execute_reply":"2023-06-03T05:53:18.806660Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"his = compile_and_fit(model)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T05:53:18.808916Z","iopub.execute_input":"2023-06-03T05:53:18.809371Z","iopub.status.idle":"2023-06-03T05:53:18.834643Z","shell.execute_reply.started":"2023-06-03T05:53:18.809338Z","shell.execute_reply":"2023-06-03T05:53:18.833798Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_images_ds = parsed_test_dataset.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)\nprint(predictions)\ntest_ids_ds = parsed_test_dataset.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(462*64))).numpy().astype('U')\nnp.savetxt('submission.csv', \n           np.rec.fromarrays([test_ids, predictions]), \n           fmt=['%s', '%d'], \n           delimiter=',',\n           header='id,label', \n           comments='')","metadata":{"execution":{"iopub.status.busy":"2023-06-03T06:03:57.312830Z","iopub.execute_input":"2023-06-03T06:03:57.313221Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#print(test_ids)","metadata":{"execution":{"iopub.status.busy":"2023-06-03T06:03:01.823873Z","iopub.execute_input":"2023-06-03T06:03:01.824268Z","iopub.status.idle":"2023-06-03T06:03:01.830302Z","shell.execute_reply.started":"2023-06-03T06:03:01.824235Z","shell.execute_reply":"2023-06-03T06:03:01.829225Z"},"trusted":true},"execution_count":null,"outputs":[]}]}