{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Hello Friends 👋\n\nAssuming this as a **Multiclass classification** task, I'm trying-out end-to-end classification (*SparseCategoricalCrossentropy Loss*) (linking KAGGLE-data cloud bucket). \n\n\nNotebook is for-\n* getting started faster\n* beginners who want to try out TPU training\n\nSpecial thanks to-\n* https://www.kaggle.com/ks2019/happywhale-arcface-baseline-tpu\n* https://www.kaggle.com/docs/tpu\n* https://www.kaggle.com/product-feedback/129828","metadata":{}},{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os, sys, cv2\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.callbacks import ReduceLROnPlateau, ModelCheckpoint, EarlyStopping","metadata":{"execution":{"iopub.status.busy":"2022-02-08T03:28:01.446709Z","iopub.execute_input":"2022-02-08T03:28:01.447541Z","iopub.status.idle":"2022-02-08T03:28:07.974501Z","shell.execute_reply.started":"2022-02-08T03:28:01.447447Z","shell.execute_reply":"2022-02-08T03:28:07.973627Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data.csv excluded id_freq>150\ndf= pd.read_csv('../input/happy-whale-and-dolphin/sample_submission.csv')\nprint(df.shape)\nEncoder=LabelEncoder()\nEncoder.classes_ = np.load('../input/happywhale-classification-using-tpu-training/classes.npy', allow_pickle=True)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-08T03:28:07.976438Z","iopub.execute_input":"2022-02-08T03:28:07.976664Z","iopub.status.idle":"2022-02-08T03:28:08.099164Z","shell.execute_reply.started":"2022-02-08T03:28:07.976638Z","shell.execute_reply":"2022-02-08T03:28:08.098616Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.predictions=''","metadata":{"execution":{"iopub.status.busy":"2022-02-08T03:28:08.100187Z","iopub.execute_input":"2022-02-08T03:28:08.100944Z","iopub.status.idle":"2022-02-08T03:28:08.108195Z","shell.execute_reply.started":"2022-02-08T03:28:08.100911Z","shell.execute_reply":"2022-02-08T03:28:08.107322Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_classes= len(Encoder.classes_)\nimg_size = 480\nseed= 2001\nbatch_size=25\nn_classes","metadata":{"execution":{"iopub.status.busy":"2022-02-08T03:28:08.110085Z","iopub.execute_input":"2022-02-08T03:28:08.110338Z","iopub.status.idle":"2022-02-08T03:28:08.120022Z","shell.execute_reply.started":"2022-02-08T03:28:08.110301Z","shell.execute_reply":"2022-02-08T03:28:08.119191Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## TPU Input Pipeline\nUsefull links\n* https://www.tensorflow.org/guide/tpu\n* https://www.tensorflow.org/guide/data_performance","metadata":{}},{"cell_type":"code","source":"def auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    \n    return strategy","metadata":{"execution":{"iopub.status.busy":"2022-02-08T03:28:08.121046Z","iopub.execute_input":"2022-02-08T03:28:08.121652Z","iopub.status.idle":"2022-02-08T03:28:08.131510Z","shell.execute_reply.started":"2022-02-08T03:28:08.121604Z","shell.execute_reply":"2022-02-08T03:28:08.130680Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def readImg(with_labels=True, target_size=(512, 512)):\n    def readOnly(path):\n        file_bytes = tf.io.read_file(path)\n        img = tf.image.decode_jpeg(file_bytes, channels=3)\n        img= tf.cast(img, tf.float32)/255.0\n        return tf.image.resize(img, target_size)\n    def readWithLabels(path, label):\n        return readOnly(path), label\n    return readWithLabels if with_labels else readOnly\n\ndef build_augmenter(with_labels=True):\n    def augment(img):\n        img = tf.image.random_flip_left_right(img)\n        #img = tf.image.random_flip_up_down(img)\n        img = tf.image.random_saturation(img, 0.8, 1.2)\n        img = tf.image.random_brightness(img, 0.1)\n        img = tf.image.random_contrast(img, 0.8, 1.2)\n        return img\n    def augment_with_labels(img, label):\n        return augment(img), label\n    return augment_with_labels if with_labels else augment\n\ndef Build_dataset(paths, labels= None, batch= batch_size,\n                  decode_fn=None, augment_fn=None,\n                  augment= False, repeat= True, shuffle= seed):\n    AUTO = tf.data.experimental.AUTOTUNE\n    slices = paths if labels is None else (paths, labels)\n    \n    dset = tf.data.Dataset.from_tensor_slices(slices)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.map(augment_fn, num_parallel_calls=AUTO) if augment else dset\n    dset = dset.repeat() if repeat else dset\n    dset = dset.shuffle(shuffle) if shuffle else dset\n    dset = dset.batch(batch).prefetch(AUTO)\n    \n    return dset","metadata":{"execution":{"iopub.status.busy":"2022-02-08T03:28:08.132914Z","iopub.execute_input":"2022-02-08T03:28:08.133187Z","iopub.status.idle":"2022-02-08T03:28:08.147399Z","shell.execute_reply.started":"2022-02-08T03:28:08.133159Z","shell.execute_reply":"2022-02-08T03:28:08.146415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_NAME = \"happy-whale-and-dolphin\"\nstrategy = auto_select_accelerator()\nbatch_size = strategy.num_replicas_in_sync * batch_size\nprint('batch size', batch_size)","metadata":{"execution":{"iopub.status.busy":"2022-02-08T03:28:08.148840Z","iopub.execute_input":"2022-02-08T03:28:08.149219Z","iopub.status.idle":"2022-02-08T03:28:13.835283Z","shell.execute_reply.started":"2022-02-08T03:28:08.149190Z","shell.execute_reply":"2022-02-08T03:28:13.834505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GCS_DS_PATH = KaggleDatasets().get_gcs_path(DATASET_NAME)\ntest_paths = GCS_DS_PATH + \"/test_images/\" + df['image']\nGCS_DS_PATH","metadata":{"execution":{"iopub.status.busy":"2022-02-08T03:28:13.836257Z","iopub.execute_input":"2022-02-08T03:28:13.836849Z","iopub.status.idle":"2022-02-08T03:28:14.360884Z","shell.execute_reply.started":"2022-02-08T03:28:13.836815Z","shell.execute_reply":"2022-02-08T03:28:14.359797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_decoder = readImg(with_labels=False, target_size= (img_size, img_size))\ndtest = Build_dataset(paths= test_paths, decode_fn=test_decoder,\n                      labels= None, augment= False, repeat=False, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2022-02-08T03:28:14.362666Z","iopub.execute_input":"2022-02-08T03:28:14.363632Z","iopub.status.idle":"2022-02-08T03:28:14.470663Z","shell.execute_reply.started":"2022-02-08T03:28:14.363587Z","shell.execute_reply":"2022-02-08T03:28:14.469928Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model Predictions","metadata":{}},{"cell_type":"code","source":"with strategy.scope():\n    model= tf.keras.models.load_model('../input/fork-of-happywhale-classification-using-tpu-tra/efficientnetv2m_v0.h5')\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-02-08T03:28:14.473346Z","iopub.execute_input":"2022-02-08T03:28:14.474131Z","iopub.status.idle":"2022-02-08T03:29:42.133390Z","shell.execute_reply.started":"2022-02-08T03:28:14.474085Z","shell.execute_reply":"2022-02-08T03:29:42.131708Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred= model.predict(dtest, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-02-08T03:29:42.136509Z","iopub.execute_input":"2022-02-08T03:29:42.136980Z","iopub.status.idle":"2022-02-08T03:30:52.975655Z","shell.execute_reply.started":"2022-02-08T03:29:42.136940Z","shell.execute_reply":"2022-02-08T03:30:52.974498Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process(pred, k=4):\n    pred= np.argsort(pred)[:, -k:]\n    p=[]\n    for itm in pred:\n        s=''\n        for i in Encoder.inverse_transform(itm[::-1]):\n            s+= (i+' ')\n        s+= 'new_individual'\n        p.append(s)\n    return p","metadata":{"execution":{"iopub.status.busy":"2022-02-08T03:30:52.976638Z","iopub.status.idle":"2022-02-08T03:30:52.977611Z","shell.execute_reply.started":"2022-02-08T03:30:52.977376Z","shell.execute_reply":"2022-02-08T03:30:52.977405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"p= process(pred, k=4)\ndf.predictions= p","metadata":{"execution":{"iopub.status.busy":"2022-02-08T03:30:52.978536Z","iopub.status.idle":"2022-02-08T03:30:52.979075Z","shell.execute_reply.started":"2022-02-08T03:30:52.978860Z","shell.execute_reply":"2022-02-08T03:30:52.978881Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('sample_submission.csv', index=False)\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-08T03:30:52.980388Z","iopub.status.idle":"2022-02-08T03:30:52.981088Z","shell.execute_reply.started":"2022-02-08T03:30:52.980866Z","shell.execute_reply":"2022-02-08T03:30:52.980887Z"},"trusted":true},"execution_count":null,"outputs":[]}]}