{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.15","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"tpu1vmV38","dataSources":[{"sourceId":13836,"databundleVersionId":1718836,"sourceType":"competition"}],"dockerImageVersionId":30788,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import re\nimport os \nimport json\nimport tensorflow as tf \nfrom tensorflow import keras \nfrom keras import layers,Model,Input,callbacks\nfrom kaggle_datasets import KaggleDatasets\nfrom functools import partial\nfrom sklearn.model_selection import train_test_split","metadata":{"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Notes\n\n- Hi,I'm a student and it's what I learned in last days, so don't strict\n- I'll discuss important parts in short terms\n- **I'll be happy raise my mistakes**","metadata":{}},{"cell_type":"markdown","source":"# Initializing and setting up TPUs ","metadata":{}},{"cell_type":"code","source":"try:\n    # Initializing, connecting and setting up TPUs\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver() \n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n\n    # - TPUs on kaggle have 8 cores,they work as independent devices\n    # Defining the strategy of distribution (per cores)\n    strategy = tf.distribute.experimental.TPUStrategy(tpu)\nexcept ValueError:\n    # If TPUs was not available the compiler will continue on another accelerator.\n    strategy = tf.distribute.get_strategy()\n# iwill print out 8 for \nprint(f'num of replicas : {strategy.num_replicas_in_sync}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:19.024323Z","iopub.execute_input":"2024-11-05T15:20:19.024742Z","iopub.status.idle":"2024-11-05T15:20:27.284057Z","shell.execute_reply.started":"2024-11-05T15:20:19.024715Z","shell.execute_reply":"2024-11-05T15:20:27.283308Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Required variables","metadata":{}},{"cell_type":"code","source":"# We'll use it for improving performance automatically\nAUTO = tf.data.experimental.AUTOTUNE \nPATH = '/kaggle/input/cassava-leaf-disease-classification' # - \nIMAGE_SIZE = (512,512,3) # - \n\n# batch per core * num of cores = total batch \nBATCH_SIZE = 24 * strategy.num_replicas_in_sync\n\nprint(f'GCS path : {PATH}')\nprint(f'Batch size : {BATCH_SIZE}')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:27.284997Z","iopub.execute_input":"2024-11-05T15:20:27.285234Z","iopub.status.idle":"2024-11-05T15:20:27.289687Z","shell.execute_reply.started":"2024-11-05T15:20:27.285211Z","shell.execute_reply":"2024-11-05T15:20:27.288921Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Getting tfrecord file names\n\n# train set\nFILES = tf.io.gfile.glob(os.path.join(PATH,'train_tfrecords/*.tfrec'))\n\n# test set \nTEST_FILE_PATHS = tf.io.gfile.glob(os.path.join(PATH,'test_tfrecords/*.tfrec'))\n\n# Splitting train set\nTRAIN_FILE_PATHS,VALID_FILE_PATHS = train_test_split(FILES,test_size=0.3,random_state=101)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:27.291417Z","iopub.execute_input":"2024-11-05T15:20:27.291692Z","iopub.status.idle":"2024-11-05T15:20:27.321126Z","shell.execute_reply.started":"2024-11-05T15:20:27.291666Z","shell.execute_reply":"2024-11-05T15:20:27.320377Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Required functions","metadata":{}},{"cell_type":"code","source":"# Decoding, casting, normalizing and reshaping the image\ndef decode_image(img_data):\n    image = tf.io.decode_jpeg(img_data,channels=3)\n    image = tf.cast(image,tf.float32)/255.\n    image = tf.reshape(image,IMAGE_SIZE)\n    return image","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:27.321902Z","iopub.execute_input":"2024-11-05T15:20:27.322144Z","iopub.status.idle":"2024-11-05T15:20:27.325961Z","shell.execute_reply.started":"2024-11-05T15:20:27.322120Z","shell.execute_reply":"2024-11-05T15:20:27.325273Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# It gets a single example and sparses it \ndef read_tfrecord(example,labeled):\n    # If example was labeled (train & valid sets)\n    format = {\n        'image':tf.io.FixedLenFeature([],tf.string),\n        'target':tf.io.FixedLenFeature([],tf.int64)\n    } if labeled else {\n        # If example was not labeled (test set)\n        'image':tf.io.FixedLenFeature([],tf.string),\n        'image_name':tf.io.FixedLenFeature([],tf.string)\n    }\n\n    # Parsing a single example according to it's format\n    example = tf.io.parse_single_example(example,format)\n    image = decode_image(example['image']) # Decode the image\n\n    # If example was labeled return : tuple(image,lable)\n    # If example was not labeled return : tuple(image,image_id)\n    if labeled:\n        label = tf.cast(example['target'],tf.int32)\n        return image,label\n    else:\n        image_name = example['image_name']\n        return image,image_name","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:27.326776Z","iopub.execute_input":"2024-11-05T15:20:27.327039Z","iopub.status.idle":"2024-11-05T15:20:27.336344Z","shell.execute_reply.started":"2024-11-05T15:20:27.327016Z","shell.execute_reply":"2024-11-05T15:20:27.335568Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Extracting and Parsing entire data\n\ndef load_data(file_paths,labeled,ordered):\n    \n    # For ignoring order of examples in tfrecord and extract based on efficiency\n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False # Order is not matter\n\n    # Extracting data from tfrecord using file_paths\n    # num_parallel_reads=AUTO : Reading from different tfrecord file at the same time\n    dset = tf.data.TFRecordDataset(file_paths,num_parallel_reads=AUTO)\n    dset.with_options(ignore_order)\n\n    # reformatting examples and decoding image using functions we previously defined\n    dset = dset.map(partial(read_tfrecord,labeled=labeled),num_parallel_calls=AUTO)\n    return dset","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:27.337189Z","iopub.execute_input":"2024-11-05T15:20:27.337436Z","iopub.status.idle":"2024-11-05T15:20:27.348148Z","shell.execute_reply.started":"2024-11-05T15:20:27.337412Z","shell.execute_reply":"2024-11-05T15:20:27.347455Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Getting and preparing data sets","metadata":{}},{"cell_type":"code","source":"# Image augmentation \n# Note : TPUs don't support augment methods those change the image size\ndef augment(image,label):\n    img = tf.image.random_flip_left_right(image,2002)\n    img = tf.image.random_flip_up_down(img,2002)\n    return img,label","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:27.349007Z","iopub.execute_input":"2024-11-05T15:20:27.349305Z","iopub.status.idle":"2024-11-05T15:20:27.357472Z","shell.execute_reply.started":"2024-11-05T15:20:27.349278Z","shell.execute_reply":"2024-11-05T15:20:27.356664Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"# Getting and preparing datasets\ndef get_train():\n    dset = load_data(TRAIN_FILE_PATHS,True,False) # - \n    dset = dset.shuffle(1001) # - \n    dset = dset.map(augment,num_parallel_calls=AUTO) # Augmenting data\n    dset = dset.batch(BATCH_SIZE) # Packing samples in batchs\n    dset = dset.repeat() # - \n    dset = dset.prefetch(AUTO) # It Prefetch some batchs to ensure that there are prepared batchs to process\n    return dset \n\ndef get_valid():\n    dset = load_data(VALID_FILE_PATHS,True,False)\n    dset = dset.shuffle(1001)\n    dset = dset.batch(BATCH_SIZE)\n    dset = dset.cache() # Because we'll use validation set several times during training , It's efficient to cache it\n    dset = dset.prefetch(AUTO)\n    return dset \n\ndef get_test():\n    dset = load_data(TEST_FILE_PATHS,False,True)\n    dset = dset.prefetch(AUTO)\n    return dset ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:27.358320Z","iopub.execute_input":"2024-11-05T15:20:27.358572Z","iopub.status.idle":"2024-11-05T15:20:27.367515Z","shell.execute_reply.started":"2024-11-05T15:20:27.358548Z","shell.execute_reply":"2024-11-05T15:20:27.366824Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train = get_train()\nvalid = get_valid()\ntest = get_test()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:27.370112Z","iopub.execute_input":"2024-11-05T15:20:27.370383Z","iopub.status.idle":"2024-11-05T15:20:27.647563Z","shell.execute_reply.started":"2024-11-05T15:20:27.370356Z","shell.execute_reply":"2024-11-05T15:20:27.646737Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Designing the Model","metadata":{}},{"cell_type":"code","source":"lr_schedule = keras.optimizers.schedules.ExponentialDecay(\n    initial_learning_rate=1e-5,\n    decay_steps=10000,\n    decay_rate=0.9\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:27.648578Z","iopub.execute_input":"2024-11-05T15:20:27.648868Z","iopub.status.idle":"2024-11-05T15:20:27.682519Z","shell.execute_reply.started":"2024-11-05T15:20:27.648838Z","shell.execute_reply":"2024-11-05T15:20:27.681620Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with strategy.scope():\n    base = keras.applications.xception.Xception(weights='imagenet',include_top=False,input_shape=IMAGE_SIZE)\n    base.trainable = True\n    # for layer in base.layers[-6:]:\n    #     layer.trainable=True\n    model = keras.Sequential([\n        base,\n        layers.GlobalAveragePooling2D(),\n        layers.Dense(512,activation='relu'),\n        layers.Dropout(0.5),\n        layers.Dense(5,activation='softmax')\n    ])\n\n    model.compile(\n        optimizer=keras.optimizers.RMSprop(lr_schedule),\n        loss='sparse_categorical_crossentropy',\n        metrics=['sparse_categorical_accuracy']\n    )","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:27.683570Z","iopub.execute_input":"2024-11-05T15:20:27.683914Z","iopub.status.idle":"2024-11-05T15:20:41.025858Z","shell.execute_reply.started":"2024-11-05T15:20:27.683873Z","shell.execute_reply":"2024-11-05T15:20:41.024676Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"model.summary()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:41.026927Z","iopub.execute_input":"2024-11-05T15:20:41.027204Z","iopub.status.idle":"2024-11-05T15:20:41.052281Z","shell.execute_reply.started":"2024-11-05T15:20:41.027179Z","shell.execute_reply":"2024-11-05T15:20:41.051373Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import numpy as np \ndef count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) for filename in filenames]\n    return np.sum(n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:41.053265Z","iopub.execute_input":"2024-11-05T15:20:41.053540Z","iopub.status.idle":"2024-11-05T15:20:41.058007Z","shell.execute_reply.started":"2024-11-05T15:20:41.053514Z","shell.execute_reply":"2024-11-05T15:20:41.057178Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"NUM_TRAIN = count_data_items(TRAIN_FILE_PATHS)\nNUM_VALID = count_data_items(VALID_FILE_PATHS)\nNUM_TEST = count_data_items(TEST_FILE_PATHS)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:41.058973Z","iopub.execute_input":"2024-11-05T15:20:41.059224Z","iopub.status.idle":"2024-11-05T15:20:41.074607Z","shell.execute_reply.started":"2024-11-05T15:20:41.059199Z","shell.execute_reply":"2024-11-05T15:20:41.073714Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"early_stopping = callbacks.EarlyStopping(\n    monitor='val_loss',\n    min_delta = 0.01,\n    patience= 3,\n    restore_best_weights = True\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:41.075623Z","iopub.execute_input":"2024-11-05T15:20:41.075884Z","iopub.status.idle":"2024-11-05T15:20:41.087003Z","shell.execute_reply.started":"2024-11-05T15:20:41.075855Z","shell.execute_reply":"2024-11-05T15:20:41.086152Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Fitting","metadata":{}},{"cell_type":"code","source":"hist = model.fit(\n    train,\n    steps_per_epoch= NUM_TRAIN // BATCH_SIZE,\n    epochs=15,\n    validation_data=[valid],\n    callbacks=[early_stopping]\n)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:20:41.087909Z","iopub.execute_input":"2024-11-05T15:20:41.088172Z","iopub.status.idle":"2024-11-05T15:43:21.448738Z","shell.execute_reply.started":"2024-11-05T15:20:41.088147Z","shell.execute_reply":"2024-11-05T15:43:21.447446Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Evaluation and Visulizing","metadata":{}},{"cell_type":"code","source":"model.evaluate(valid)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:43:21.450666Z","iopub.execute_input":"2024-11-05T15:43:21.451014Z","iopub.status.idle":"2024-11-05T15:43:42.240676Z","shell.execute_reply.started":"2024-11-05T15:43:21.450983Z","shell.execute_reply":"2024-11-05T15:43:42.239745Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"import pandas as pd\nimport matplotlib.pyplot as plt \nres = pd.DataFrame(hist.history)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:43:42.241817Z","iopub.execute_input":"2024-11-05T15:43:42.242123Z","iopub.status.idle":"2024-11-05T15:43:43.172018Z","shell.execute_reply.started":"2024-11-05T15:43:42.242094Z","shell.execute_reply":"2024-11-05T15:43:43.170787Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"fig,ax = plt.subplots(2,1)\nax[0].plot(res[['loss','val_loss']])\nax[1].plot(res[['sparse_categorical_accuracy','val_sparse_categorical_accuracy']])","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:43:43.173118Z","iopub.execute_input":"2024-11-05T15:43:43.173386Z","iopub.status.idle":"2024-11-05T15:43:43.434666Z","shell.execute_reply.started":"2024-11-05T15:43:43.173361Z","shell.execute_reply":"2024-11-05T15:43:43.433816Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with open('/kaggle/input/cassava-leaf-disease-classification/label_num_to_disease_map.json','r') as f :\n    classes = json.loads(f.read())","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:50:57.937347Z","iopub.execute_input":"2024-11-05T15:50:57.937723Z","iopub.status.idle":"2024-11-05T15:50:57.972456Z","shell.execute_reply.started":"2024-11-05T15:50:57.937690Z","shell.execute_reply":"2024-11-05T15:50:57.971578Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def visualize_predictions(batch):\n    val_imgs,val_lbl = batch \n    val_prob = model.predict(val_imgs)\n    val_pred = np.argmax(val_prob,axis=-1)\n    \n    correct_color = 'green'\n    wrong_color = 'red'\n    rows = 5\n    cols= 4\n    fig,ax = plt.subplots(rows,cols,figsize=(10,10))\n    fig.tight_layout() \n\n    print('A : Actual class, P : Predicted class\\n')\n    print(classes)\n    counter = 0\n    for i in range(rows):\n        for j in range(cols):\n            ax[i,j].imshow(val_imgs[counter])\n            ax[i,j].axis('off')\n            if val_pred[counter] == val_lbl[counter]:\n                txt = f'P{val_pred[counter]} - A{val_lbl[counter]}'\n                ax[i,j].set_title(txt,color=correct_color)\n            else:\n                txt = f'P{val_pred[counter]} - A{val_lbl[counter]}'\n                ax[i,j].set_title(txt,color=wrong_color)\n            counter+=1\n        ","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:56:42.692563Z","iopub.execute_input":"2024-11-05T15:56:42.693056Z","iopub.status.idle":"2024-11-05T15:56:42.701418Z","shell.execute_reply.started":"2024-11-05T15:56:42.693016Z","shell.execute_reply":"2024-11-05T15:56:42.700365Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"val_batch_size = 20 \nval_dset = valid.shuffle(101).unbatch().batch(val_batch_size)\nval_batch = iter(val_dset)\nvisualize_predictions(next(val_batch))","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T16:03:03.350612Z","iopub.execute_input":"2024-11-05T16:03:03.351056Z","iopub.status.idle":"2024-11-05T16:03:06.857782Z","shell.execute_reply.started":"2024-11-05T16:03:03.351021Z","shell.execute_reply":"2024-11-05T16:03:06.856488Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"with strategy.scope():\n    model.save('best.keras')","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-11-05T15:43:54.482144Z","iopub.execute_input":"2024-11-05T15:43:54.482403Z","iopub.status.idle":"2024-11-05T15:43:56.215395Z","shell.execute_reply.started":"2024-11-05T15:43:54.482377Z","shell.execute_reply":"2024-11-05T15:43:56.214331Z"}},"outputs":[],"execution_count":null}]}