{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.14","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":30787,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport math, os, re, random\n\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport seaborn as sns\n# Save ot csv\nfrom PIL import Image\nfrom tqdm import tqdm\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2024-11-25T10:46:44.831753Z","iopub.execute_input":"2024-11-25T10:46:44.832582Z","iopub.status.idle":"2024-11-25T10:46:45.630222Z","shell.execute_reply.started":"2024-11-25T10:46:44.832545Z","shell.execute_reply":"2024-11-25T10:46:45.629499Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import tensorflow as tf\nimport tensorflow as tf, tensorflow.keras.backend as K\nfrom kaggle_datasets import KaggleDatasets\nprint('Tensorflow version ' + tf.__version__)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T10:46:46.390363Z","iopub.execute_input":"2024-11-25T10:46:46.390779Z","iopub.status.idle":"2024-11-25T10:46:49.064581Z","shell.execute_reply.started":"2024-11-25T10:46:46.390748Z","shell.execute_reply":"2024-11-25T10:46:49.063648Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import keras\nfrom keras.models import Sequential, load_model\nfrom keras.layers import Dense, Dropout, Flatten\nfrom keras.layers import Conv2D, MaxPooling2D, BatchNormalization\nfrom keras.utils import to_categorical","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T10:46:49.193702Z","iopub.execute_input":"2024-11-25T10:46:49.194177Z","iopub.status.idle":"2024-11-25T10:46:49.200104Z","shell.execute_reply.started":"2024-11-25T10:46:49.194148Z","shell.execute_reply":"2024-11-25T10:46:49.199425Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def count_data_items(filenames):\n    # the number of data items is written in the name of the .tfrec files, i.e. flowers00-230.tfrec = 230 data items\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"\ndataset = '/kaggle/input/tpu-getting-started/tfrecords-jpeg-512x512'\n\nTRAINING_FILENAMES = tf.io.gfile.glob(dataset + '/train/*.tfrec')\nVALIDATION_FILENAMES = tf.io.gfile.glob(dataset + '/val/*.tfrec')\nTEST_FILENAMES = tf.io.gfile.glob(dataset + '/test/*.tfrec')\nBATCH_SIZE = 20\n\nNUM_TRAINING_IMAGES = count_data_items(TRAINING_FILENAMES)\nNUM_VALIDATION_IMAGES = count_data_items(VALIDATION_FILENAMES)\nNUM_TEST_IMAGES = count_data_items(TEST_FILENAMES)\nSTEPS_PER_EPOCH = NUM_TRAINING_IMAGES // BATCH_SIZE\nprint(\"BATCH_SIZE\", BATCH_SIZE)\nprint('Dataset: {} training images, {} validation images, {} unlabeled test images'.format(NUM_TRAINING_IMAGES, NUM_VALIDATION_IMAGES, NUM_TEST_IMAGES))","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Convert TFREC to JPEG & CSV","metadata":{}},{"cell_type":"markdown","source":"**TFREC to JPEG**","metadata":{}},{"cell_type":"code","source":"# Initialize lists to store labels, filenames, and splits\nclass_list = []\nfile_list = []\nsplit_list = []","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"in_path = dataset\nout_path = \"images-jpg-512x512\"","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_tfrecord(example, test=False):\n    # Test tfrecord has only image and id keys\n    if test:\n        tfrec_format = {\n            \"image\": tf.io.FixedLenFeature([], tf.string),\n            \"id\": tf.io.FixedLenFeature([], tf.string),\n        }\n        example = tf.io.parse_single_example(example, tfrec_format)\n        ids_or_label = example['id']\n    # Train/Val tfrecord has image and class keys\n    else:\n        tfrec_format = {\n            \"image\": tf.io.FixedLenFeature([], tf.string),\n            \"class\": tf.io.FixedLenFeature([], tf.int64),\n        }\n        example = tf.io.parse_single_example(example, tfrec_format)\n        ids_or_label = tf.cast(example['class'], tf.int32)\n    # example[\"image\"] is the binary encoding of the jpeg image, the following line will decode it to raw image:\n    image = tf.image.decode_jpeg(example['image'], channels=3)\n    return image, ids_or_label\n\ndef read_test_tfrecord(example):\n    return read_tfrecord(example, test=True)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"k = 0\nfor folder_split in [\"train\", \"val\", \"test\"]:\n    root = os.path.join(in_path, folder_split)\n    \n    # Form the list of .tfrec files for the local folder (train, val, test)\n    filenames = [os.path.join(root, f) for f in os.listdir(root)]\n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads=None)\n    dataset = dataset.map(read_tfrecord if folder_split != \"test\" else read_test_tfrecord)\n    \n    # Count approximately the number of elements in the TFRecord\n    n_tot = sum([1 for _ in dataset.batch(64)]) * 64\n    \n    out_folder = os.path.join(out_path, folder_split)\n    os.makedirs(out_folder, exist_ok=True)\n    print(\"Converting split\", folder_split)\n    \n    for iter, data in enumerate(tqdm(dataset, total=n_tot)):\n        image, ids_or_label = data\n        image = Image.fromarray(image.numpy())\n        \n        if folder_split == \"test\":\n            ids = str(ids_or_label.numpy().decode())\n            image.save(os.path.join(out_folder, f\"{ids}.jpeg\"))\n            # For test, we don't need to track class, just the id\n            class_list.append(None)\n            file_list.append(f\"{ids}.jpeg\")\n            split_list.append(\"test\")\n        else:\n            # For train and validation, track the class\n            image.save(os.path.join(out_folder, f\"{k:06d}.jpeg\"))\n            class_list.append(int(ids_or_label.numpy()))  # Convert label to integer\n            file_list.append(f\"{k:06d}.jpeg\")\n            split_list.append(folder_split)\n            k += 1","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"**Create CSV**","metadata":{}},{"cell_type":"code","source":"df = pd.DataFrame()\ndf[\"class\"] = class_list\ndf[\"file\"] = file_list\ndf[\"split\"] = split_list\n\n# Save CSV for train and validation sets\ndf.to_csv(\"train_val_labels.csv\", index=False)\n\n# Optionally print the first few rows of the dataframe\nprint(df.head())","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df = pd.DataFrame()\ndf[\"class\"] = class_list\ndf[\"file\"] = file_list\ndf[\"split\"] = split_list\ndf.head()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Save ot csv\ndf.to_csv(\"train_val.csv\", index=False)","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def display_images(folder_path, folder_name):\n    # List all image files in the folder\n    image_files = [f for f in os.listdir(folder_path) if f.endswith(('.jpg', '.jpeg', '.png'))]\n    \n    # Get the first 5 images\n    first_images = image_files[:5]\n    \n    # Plot the images\n    plt.figure(figsize=(15, 5))\n    for i, image_file in enumerate(first_images):\n        image_path = os.path.join(folder_path, image_file)\n        img = mpimg.imread(image_path)\n        plt.subplot(1, 5, i + 1)\n        plt.imshow(img)\n        plt.title(f\"{folder_name}: {image_file}\")\n        plt.axis('off')\n    plt.show()","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"jpegDataset = \"/kaggle/working/images-jpg-512x512\"\nsubfolders = ['train', 'val', 'test']\n\nfor subfolder in subfolders:\n    subfolder_path = os.path.join(jpegDataset, subfolder)\n    if os.path.exists(subfolder_path):\n        display_images(subfolder_path, subfolder)\n    else:\n        print(f\"Folder not found: {subfolder_path}\")","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Tilling","metadata":{}},{"cell_type":"code","source":"device_name = tf.test.gpu_device_name()\nif device_name != '/device:GPU:0':\n  raise SystemError('GPU device not found')\nprint('Found GPU at: {}'.format(device_name))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T10:46:56.761402Z","iopub.execute_input":"2024-11-25T10:46:56.761756Z","iopub.status.idle":"2024-11-25T10:46:57.060701Z","shell.execute_reply.started":"2024-11-25T10:46:56.761724Z","shell.execute_reply":"2024-11-25T10:46:57.05966Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def read_jpeg(filename):\n    img = tf.io.read_file(filename)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.convert_image_dtype(img, tf.float32)\n    return img\n\nfullimg = read_jpeg('/kaggle/working/images-jpg-512x512/train/000700.jpeg')\nplt.imshow(fullimg);\nplt.axis('off');","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T10:46:58.752593Z","iopub.execute_input":"2024-11-25T10:46:58.752931Z","iopub.status.idle":"2024-11-25T10:46:59.396337Z","shell.execute_reply.started":"2024-11-25T10:46:58.752899Z","shell.execute_reply":"2024-11-25T10:46:59.39545Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def tile_image(fullimg, CHANNELS=3, TILE_HT=128, TILE_WD=128):\n    images = tf.expand_dims(fullimg, axis=0)\n    tiles = tf.image.extract_patches(\n        images=images,\n        sizes=[1, TILE_HT, TILE_WD, 1],\n        strides=[1, TILE_HT//2, TILE_WD//2, 1],\n        rates=[1, 1, 1, 1],\n        padding='VALID')\n    print(tiles.shape)\n\n    tiles = tf.squeeze(tiles, axis=0)\n    nrows = tiles.shape[0]\n    ncols = tiles.shape[1]\n    tiles = tf.reshape(tiles, [nrows, ncols, TILE_HT, TILE_WD, CHANNELS])\n    print(tiles.shape)\n    return tiles\n\ntiles = tile_image(fullimg)\nnrows = tiles.shape[0]\nncols = tiles.shape[1]\nf, ax = plt.subplots(nrows, ncols, figsize=(40,20))\nfor rowno in range(nrows):\n    for colno in range(ncols):\n        img = tiles[rowno][colno]\n        ax[rowno, colno].imshow( tiles[rowno][colno].numpy() );\n        ax[rowno, colno].axis('off')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T10:47:02.33235Z","iopub.execute_input":"2024-11-25T10:47:02.333212Z","iopub.status.idle":"2024-11-25T10:47:04.785484Z","shell.execute_reply.started":"2024-11-25T10:47:02.333176Z","shell.execute_reply":"2024-11-25T10:47:04.784644Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"flower_location = [[88, 244], [118, 257], [118, 285], [160, 276], [296, 379]]\nflower_label = np.zeros((338, 600))\nfor loc in flower_location:\n    flower_label[loc[0]][loc[1]] = 1.0\nplt.imshow(flower_label, cmap='gray');","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T10:47:33.990753Z","iopub.execute_input":"2024-11-25T10:47:33.991492Z","iopub.status.idle":"2024-11-25T10:47:34.219387Z","shell.execute_reply.started":"2024-11-25T10:47:33.991456Z","shell.execute_reply":"2024-11-25T10:47:34.218522Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"labels = tf.expand_dims(flower_label, axis=-1)\nprint(labels.shape)\nlabels = tile_image(labels, 1)\nlabels = tf.reduce_max(labels, axis=[2, 3, 4])\nprint(labels)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T10:47:37.266343Z","iopub.execute_input":"2024-11-25T10:47:37.26671Z","iopub.status.idle":"2024-11-25T10:47:37.285824Z","shell.execute_reply.started":"2024-11-25T10:47:37.266675Z","shell.execute_reply":"2024-11-25T10:47:37.2851Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"f, ax = plt.subplots(nrows, ncols, figsize=(40,20))\nfor rowno in range(nrows):\n    for colno in range(ncols):\n        img = tiles[rowno][colno]\n        ax[rowno, colno].imshow(img.numpy())\n        ax[rowno, colno].axis('off')\n        \n        # Check the label for this tile\n        if labels[rowno, colno] > 0:\n            ax[rowno, colno-2].set_title('Flower', fontsize=20)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-25T10:47:41.141981Z","iopub.execute_input":"2024-11-25T10:47:41.14278Z","iopub.status.idle":"2024-11-25T10:47:45.488411Z","shell.execute_reply.started":"2024-11-25T10:47:41.142747Z","shell.execute_reply":"2024-11-25T10:47:45.487357Z"}},"outputs":[],"execution_count":null}]}