{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.15","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":21154,"databundleVersionId":1243559,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os,re\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf \nfrom tensorflow import keras\nfrom kaggle_datasets import KaggleDatasets\nfrom keras import Sequential,layers,Input,Model,callbacks\nfrom keras.applications.xception import Xception,preprocess_input","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"try:\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept ValueError:\n    strategy = tf.distribute.get_strategy()\n\nprint(f'Replicas : {strategy.num_replicas_in_sync}')","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:12:33.640202Z","iopub.execute_input":"2024-10-30T15:12:33.640663Z","iopub.status.idle":"2024-10-30T15:12:42.536460Z","shell.execute_reply.started":"2024-10-30T15:12:33.640633Z","shell.execute_reply":"2024-10-30T15:12:42.535488Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining some required variables such as batch_size and gcs_path,...\nAUTOTUNE = tf.data.experimental.AUTOTUNE\nBATCH_SIZE = 8 * strategy.num_replicas_in_sync\nIMAGE_SIZE = (512,512)\nGCS_PATH = KaggleDatasets().get_gcs_path()\n\nCLASSES = ['pink primrose',    'hard-leaved pocket orchid', 'canterbury bells', 'sweet pea',     'wild geranium',     'tiger lily',           'moon orchid',              'bird of paradise', 'monkshood',        'globe thistle',         # 00 - 09\n           'snapdragon',       \"colt's foot\",               'king protea',      'spear thistle', 'yellow iris',       'globe-flower',         'purple coneflower',        'peruvian lily',    'balloon flower',   'giant white arum lily', # 10 - 19\n           'fire lily',        'pincushion flower',         'fritillary',       'red ginger',    'grape hyacinth',    'corn poppy',           'prince of wales feathers', 'stemless gentian', 'artichoke',        'sweet william',         # 20 - 29\n           'carnation',        'garden phlox',              'love in the mist', 'cosmos',        'alpine sea holly',  'ruby-lipped cattleya', 'cape flower',              'great masterwort', 'siam tulip',       'lenten rose',           # 30 - 39\n           'barberton daisy',  'daffodil',                  'sword lily',       'poinsettia',    'bolero deep blue',  'wallflower',           'marigold',                 'buttercup',        'daisy',            'common dandelion',      # 40 - 49\n           'petunia',          'wild pansy',                'primula',          'sunflower',     'lilac hibiscus',    'bishop of llandaff',   'gaura',                    'geranium',         'orange dahlia',    'pink-yellow dahlia',    # 50 - 59\n           'cautleya spicata', 'japanese anemone',          'black-eyed susan', 'silverbush',    'californian poppy', 'osteospermum',         'spring crocus',            'iris',             'windflower',       'tree poppy',            # 60 - 69\n           'gazania',          'azalea',                    'water lily',       'rose',          'thorn apple',       'morning glory',        'passion flower',           'lotus',            'toad lily',        'anthurium',             # 70 - 79\n           'frangipani',       'clematis',                  'hibiscus',         'columbine',     'desert-rose',       'tree mallow',          'magnolia',                 'cyclamen ',        'watercress',       'canna lily',            # 80 - 89\n           'hippeastrum ',     'bee balm',                  'pink quill',       'foxglove',      'bougainvillea',     'camellia',             'mallow',                   'mexican petunia',  'bromelia',         'blanket flower',        # 90 - 99\n           'trumpet creeper',  'blackberry lily',           'common tulip',     'wild rose']","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:34.651918Z","iopub.execute_input":"2024-10-30T15:45:34.652426Z","iopub.status.idle":"2024-10-30T15:45:34.663070Z","shell.execute_reply.started":"2024-10-30T15:45:34.652381Z","shell.execute_reply":"2024-10-30T15:45:34.662077Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading TFRecord filenames\ndef get_tfrec_paths(folder):\n    path = GCS_PATH + '/tfrecords-jpeg-512x512/' + folder\n    return tf.io.gfile.glob(f'{path}/*.tfrec')","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:34.899881Z","iopub.execute_input":"2024-10-30T15:45:34.900242Z","iopub.status.idle":"2024-10-30T15:45:34.905096Z","shell.execute_reply.started":"2024-10-30T15:45:34.900211Z","shell.execute_reply":"2024-10-30T15:45:34.904042Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_file_paths = get_tfrec_paths('train')\nval_file_paths = get_tfrec_paths('val')\ntest_file_paths = get_tfrec_paths('test')\n\nprint(f'''number of tfrecord files:\ntrain {len(train_file_paths)}\nvalidation {len(val_file_paths)}\ntest {len(test_file_paths)}''')","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:35.111842Z","iopub.execute_input":"2024-10-30T15:45:35.112154Z","iopub.status.idle":"2024-10-30T15:45:35.127403Z","shell.execute_reply.started":"2024-10-30T15:45:35.112127Z","shell.execute_reply":"2024-10-30T15:45:35.126308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Defining required functions\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data,channels=3)\n    image = tf.cast(image,tf.float32) / 255.\n    image = tf.reshape(image,IMAGE_SIZE + (3,))\n    return image","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:35.408735Z","iopub.execute_input":"2024-10-30T15:45:35.409088Z","iopub.status.idle":"2024-10-30T15:45:35.414653Z","shell.execute_reply.started":"2024-10-30T15:45:35.409058Z","shell.execute_reply":"2024-10-30T15:45:35.413504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reads an unlabeled tfrecord example \ndef read_unlabeled_tfrec(example):\n    the_format = {\n        'image':tf.io.FixedLenFeature([],tf.string),\n        'id':tf.io.FixedLenFeature([],tf.string)\n    }\n    example = tf.io.parse_single_example(example,the_format)\n    image = decode_image(example['image'])\n    idnum = example['id']\n    return image,idnum\n\n# Reads a labeled tfrecord example\ndef read_labeled_tfrec(example):\n    the_format = {\n        'image':tf.io.FixedLenFeature([],tf.string),\n        'class':tf.io.FixedLenFeature([],tf.int64)\n    }\n    example = tf.io.parse_single_example(example,the_format)\n    image = decode_image(example['image'])\n    label = tf.cast(example['class'],tf.int32)\n    return image,label","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:35.673233Z","iopub.execute_input":"2024-10-30T15:45:35.673513Z","iopub.status.idle":"2024-10-30T15:45:35.679643Z","shell.execute_reply.started":"2024-10-30T15:45:35.673488Z","shell.execute_reply":"2024-10-30T15:45:35.678675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# load_dataset\ndef load_data(file_paths,labeled=True ,ordered = False):\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False\n    dset = tf.data.TFRecordDataset(file_paths,num_parallel_reads=AUTOTUNE)\n    dset = dset.with_options(ignore_order)\n    dset = dset.map(read_labeled_tfrec if labeled else read_unlabeled_tfrec,num_parallel_calls=AUTOTUNE)\n    return dset ","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:35.863901Z","iopub.execute_input":"2024-10-30T15:45:35.864169Z","iopub.status.idle":"2024-10-30T15:45:35.869250Z","shell.execute_reply.started":"2024-10-30T15:45:35.864144Z","shell.execute_reply":"2024-10-30T15:45:35.868302Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def augment(image,label):\n    image = tf.image.flip_left_right(image)\n    return image,label","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:36.074087Z","iopub.execute_input":"2024-10-30T15:45:36.074421Z","iopub.status.idle":"2024-10-30T15:45:36.078716Z","shell.execute_reply.started":"2024-10-30T15:45:36.074392Z","shell.execute_reply":"2024-10-30T15:45:36.077692Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# getting datasets and setting some configurations\ndef get_train():\n    dset = load_data(train_file_paths)\n    dset = dset.map(augment)\n    dset = dset.repeat()\n    dset = dset.shuffle(101)\n    dset = dset.batch(BATCH_SIZE)\n    dset = dset.prefetch(AUTOTUNE)\n    return dset\n\ndef get_val():\n    dset = load_data(val_file_paths)\n    dset = dset.batch(BATCH_SIZE)\n    dset = dset.cache()\n    dset = dset.prefetch(AUTOTUNE)\n    return dset\n\ndef get_test():\n    dset = load_data(test_file_paths,labeled=False,ordered=True)\n    dset = dset.batch(BATCH_SIZE)\n    dset = dset.prefetch(AUTOTUNE)\n    return dset","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:36.313357Z","iopub.execute_input":"2024-10-30T15:45:36.313647Z","iopub.status.idle":"2024-10-30T15:45:36.320520Z","shell.execute_reply.started":"2024-10-30T15:45:36.313621Z","shell.execute_reply":"2024-10-30T15:45:36.319186Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = get_train()\nval = get_val()\ntest = get_test()\n\nprint('train batchs : ')\nfor image,label in train.take(3):\n    print(f'image : {image.shape} ,label : {label.shape}')\n\nprint('test batchs : ')\nfor image,idnum in val.take(3):\n    print(f'image : {image.shape} ,id : {idnum.shape}')","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:36.555023Z","iopub.execute_input":"2024-10-30T15:45:36.555375Z","iopub.status.idle":"2024-10-30T15:45:37.508817Z","shell.execute_reply.started":"2024-10-30T15:45:36.555345Z","shell.execute_reply":"2024-10-30T15:45:37.507679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:37.510541Z","iopub.execute_input":"2024-10-30T15:45:37.510839Z","iopub.status.idle":"2024-10-30T15:45:37.515602Z","shell.execute_reply.started":"2024-10-30T15:45:37.510809Z","shell.execute_reply":"2024-10-30T15:45:37.514708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_SIZE = count_data_items(train_file_paths)\nTEST_SIZE = count_data_items(test_file_paths)\nVAL_SIZE = count_data_items(val_file_paths)\n\nprint(f'train : {TRAIN_SIZE}\\ntest : {TEST_SIZE}\\nvalidation : {VAL_SIZE}')","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:37.516512Z","iopub.execute_input":"2024-10-30T15:45:37.516772Z","iopub.status.idle":"2024-10-30T15:45:37.528132Z","shell.execute_reply.started":"2024-10-30T15:45:37.516746Z","shell.execute_reply":"2024-10-30T15:45:37.527313Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# def build_model():\n#     inputs = Input(IMAGE_SIZE + (3,))\n#     x = layers.Conv2D(32,3,padding='same',activation='relu')(inputs)\n#     x = layers.MaxPool2D(padding='same')(x)\n#     for filters in [64,128,256,512,1024]:\n#         residual = x\n#         x = layers.Conv2D(filters,3,padding='same',activation='relu')(x)\n#         x = layers.Dropout(0.3)(x)\n#         x = layers.MaxPool2D(padding='same')(x)\n#         residual = layers.Conv2D(filters,1,strides=2,padding='same')(residual)\n#         x = layers.add([x,residual])\n#     x = layers.Flatten()(x)\n#     x = layers.Dropout(0.5)(x)\n#     outputs = layers.Dense(len(CLASSES),activation='softmax')(x)\n    \n#     return Model(inputs,outputs)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:37.529616Z","iopub.execute_input":"2024-10-30T15:45:37.529885Z","iopub.status.idle":"2024-10-30T15:45:37.537196Z","shell.execute_reply.started":"2024-10-30T15:45:37.529857Z","shell.execute_reply":"2024-10-30T15:45:37.536429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    base = Xception(weights='imagenet',include_top=False,input_shape=IMAGE_SIZE + (3,))\n    base.trainable = False\n    for layer in base.layers[-4:]:\n        layer.trainable = True\n    model = Sequential([\n        base,\n        layers.GlobalAveragePooling2D(),\n        layers.Dense(512,activation='relu'),\n        layers.Dropout(0.5),\n        layers.Dense(len(CLASSES),activation='softmax')\n    ])\n    model.compile(\n        optimizer = keras.optimizers.RMSprop(learning_rate=0.001),\n        loss = 'sparse_categorical_crossentropy',\n        metrics = ['sparse_categorical_accuracy']\n    )","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:37.848361Z","iopub.execute_input":"2024-10-30T15:45:37.848642Z","iopub.status.idle":"2024-10-30T15:45:42.357873Z","shell.execute_reply.started":"2024-10-30T15:45:37.848601Z","shell.execute_reply":"2024-10-30T15:45:42.356667Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:42.359131Z","iopub.execute_input":"2024-10-30T15:45:42.359785Z","iopub.status.idle":"2024-10-30T15:45:42.378694Z","shell.execute_reply.started":"2024-10-30T15:45:42.359750Z","shell.execute_reply":"2024-10-30T15:45:42.377882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"early_stopping = callbacks.EarlyStopping(\n    monitor='val_loss',\n    patience=4,\n    min_delta=0.01,\n    restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:45:43.139376Z","iopub.execute_input":"2024-10-30T15:45:43.139700Z","iopub.status.idle":"2024-10-30T15:45:43.143954Z","shell.execute_reply.started":"2024-10-30T15:45:43.139670Z","shell.execute_reply":"2024-10-30T15:45:43.143095Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"TRAIN_STEPS = TRAIN_SIZE // BATCH_SIZE\nVAL_STEPS = VAL_SIZE // BATCH_SIZE\nhistory = model.fit(\n    train,\n    epochs=40,\n    steps_per_epoch=TRAIN_STEPS,\n    validation_data=[val],\n    validation_steps = VAL_STEPS,\n    callbacks=[early_stopping]\n)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt \nres = pd.DataFrame(history.history)\nfig,ax = plt.subplots(1,2)\nax[0].plot(res[['loss','val_loss']])\nax[1].plot(res[['sparse_categorical_accuracy','val_sparse_categorical_accuracy']])","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:51:16.934641Z","iopub.execute_input":"2024-10-30T15:51:16.935398Z","iopub.status.idle":"2024-10-30T15:51:17.122670Z","shell.execute_reply.started":"2024-10-30T15:51:16.935362Z","shell.execute_reply":"2024-10-30T15:51:17.121532Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_images_ds = test.map(lambda image, idnum: image)\nprobabilities = model.predict(test_images_ds)\npredictions = np.argmax(probabilities, axis=-1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ids_ds = test.map(lambda image, idnum: idnum).unbatch()\ntest_ids = next(iter(test_ids_ds.batch(TEST_SIZE))).numpy().astype('U')\n\nnp.savetxt(\n    'submission.csv',\n    np.rec.fromarrays([test_ids, predictions]),\n    fmt=['%s', '%d'],\n    delimiter=',',\n    header='id,label',\n    comments='',\n)","metadata":{"execution":{"iopub.status.busy":"2024-10-30T15:52:46.968706Z","iopub.execute_input":"2024-10-30T15:52:46.969141Z","iopub.status.idle":"2024-10-30T15:52:51.929562Z","shell.execute_reply.started":"2024-10-30T15:52:46.969105Z","shell.execute_reply":"2024-10-30T15:52:51.928336Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}