{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd\nimport os, sys, cv2, math\nfrom kaggle_datasets import KaggleDatasets\nfrom sklearn.preprocessing import LabelEncoder\nimport matplotlib.pyplot as plt\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers\n\nTPU=True","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-20T07:20:22.693068Z","iopub.execute_input":"2022-02-20T07:20:22.694079Z","iopub.status.idle":"2022-02-20T07:20:28.584992Z","shell.execute_reply.started":"2022-02-20T07:20:22.693925Z","shell.execute_reply":"2022-02-20T07:20:28.584079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# data.csv excluded id_freq>150\ndf= pd.read_csv('../input/dataframe-startnotebook/data.csv')\nEncoder=LabelEncoder()\ndf['id_label']=Encoder.fit_transform(df.individual_id)\nnp.save('classes.npy', Encoder.classes_)\n# enc.classes_ = np.load('classes.npy', allow_pickle=True)\n# enc.inverse_transform([y1, y2])\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T07:20:28.587043Z","iopub.execute_input":"2022-02-20T07:20:28.587460Z","iopub.status.idle":"2022-02-20T07:20:28.763909Z","shell.execute_reply.started":"2022-02-20T07:20:28.587414Z","shell.execute_reply":"2022-02-20T07:20:28.762816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"evaldf= pd.read_csv('../input/happy-whale-and-dolphin/sample_submission.csv')\nevaldf.predictions= 'new_individual '\nevaldf.head()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T07:20:28.765555Z","iopub.execute_input":"2022-02-20T07:20:28.766110Z","iopub.status.idle":"2022-02-20T07:20:28.840162Z","shell.execute_reply.started":"2022-02-20T07:20:28.766070Z","shell.execute_reply":"2022-02-20T07:20:28.839240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"n_classes= df.id_label.max()+1\nimg_size =456\nseed= 2001\nbatch_size=45\nk= 4\nn_classes","metadata":{"execution":{"iopub.status.busy":"2022-02-20T07:20:28.842875Z","iopub.execute_input":"2022-02-20T07:20:28.843553Z","iopub.status.idle":"2022-02-20T07:20:28.851448Z","shell.execute_reply.started":"2022-02-20T07:20:28.843507Z","shell.execute_reply":"2022-02-20T07:20:28.850430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def auto_select_accelerator():\n    try:\n        tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(tpu)\n        tf.tpu.experimental.initialize_tpu_system(tpu)\n        strategy = tf.distribute.experimental.TPUStrategy(tpu)\n        print(\"Running on TPU:\", tpu.master())\n    except ValueError:\n        strategy = tf.distribute.get_strategy()\n    print(f\"Running on {strategy.num_replicas_in_sync} replicas\")\n    \n    return strategy","metadata":{"execution":{"iopub.status.busy":"2022-02-20T07:20:28.853221Z","iopub.execute_input":"2022-02-20T07:20:28.853528Z","iopub.status.idle":"2022-02-20T07:20:28.864583Z","shell.execute_reply.started":"2022-02-20T07:20:28.853487Z","shell.execute_reply":"2022-02-20T07:20:28.863966Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def readImg(target_size=(512, 512)):\n    def readOnly(path):\n        file_bytes = tf.io.read_file(path)\n        img = tf.image.decode_jpeg(file_bytes, channels=3)\n        img= tf.cast(img, tf.bfloat16)/255.0\n        return tf.image.resize(img, target_size)\n    return readOnly\n\ndef build_dataset(paths, bsize=20, decode_fn=None):\n    if decode_fn is None:\n        decode_fn = readImg()\n    AUTO = tf.data.experimental.AUTOTUNE\n    dset = tf.data.Dataset.from_tensor_slices(paths)\n    dset = dset.map(decode_fn, num_parallel_calls=AUTO)\n    dset = dset.batch(bsize).prefetch(AUTO) # overlaps data preprocessing and model execution while training\n    return dset","metadata":{"execution":{"iopub.status.busy":"2022-02-20T07:20:28.866355Z","iopub.execute_input":"2022-02-20T07:20:28.866981Z","iopub.status.idle":"2022-02-20T07:20:28.876501Z","shell.execute_reply.started":"2022-02-20T07:20:28.866936Z","shell.execute_reply":"2022-02-20T07:20:28.875624Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_NAME = \"happy-whale-and-dolphin\"\nstrategy = auto_select_accelerator()\nbatch_size = strategy.num_replicas_in_sync * batch_size\nprint('batch size', batch_size)","metadata":{"execution":{"iopub.status.busy":"2022-02-20T07:20:28.878422Z","iopub.execute_input":"2022-02-20T07:20:28.878787Z","iopub.status.idle":"2022-02-20T07:20:34.615773Z","shell.execute_reply.started":"2022-02-20T07:20:28.878746Z","shell.execute_reply":"2022-02-20T07:20:34.614983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if TPU:\n    GCS_DS_PATH = KaggleDatasets().get_gcs_path(DATASET_NAME)\n    df['paths'] = df.image.apply(lambda x: GCS_DS_PATH+ '/train_images/' + x)\n    evaldf['paths']= evaldf.image.apply(lambda x: GCS_DS_PATH+ '/test_images/' + x)\n    print(GCS_DS_PATH)\nelse:\n    df['paths'] = df.image.apply(lambda x: '../input/happy-whale-and-dolphin/train_images/' + x)\n    evaldf['paths']= evaldf.image.apply(lambda x: '../input/happy-whale-and-dolphin/test_images' + x)","metadata":{"execution":{"iopub.status.busy":"2022-02-20T07:20:34.616995Z","iopub.execute_input":"2022-02-20T07:20:34.617248Z","iopub.status.idle":"2022-02-20T07:20:35.130748Z","shell.execute_reply.started":"2022-02-20T07:20:34.617210Z","shell.execute_reply":"2022-02-20T07:20:35.129709Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"decoder = readImg(target_size=(img_size, img_size))\n\n# Build the tensorflow datasets\ndtrain = build_dataset(df['paths'].values,\n                       bsize=batch_size, decode_fn=decoder)\n\ndeval = build_dataset(evaldf['paths'].values, \n                      bsize=batch_size, decode_fn=decoder)","metadata":{"execution":{"iopub.status.busy":"2022-02-20T07:20:35.132193Z","iopub.execute_input":"2022-02-20T07:20:35.132456Z","iopub.status.idle":"2022-02-20T07:20:35.287599Z","shell.execute_reply.started":"2022-02-20T07:20:35.132428Z","shell.execute_reply":"2022-02-20T07:20:35.286553Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class eluDistance(tf.keras.layers.Layer):\n    def __init__(self, **kwargs):\n        super().__init__(**kwargs)\n    def call(self, anchor, positive, negative):\n        ap_distance = tf.reduce_sum(tf.square(anchor - positive), -1)\n        an_distance = tf.reduce_sum(tf.square(anchor - negative), -1)\n        return (ap_distance, an_distance)\n\ndef buildModel():\n    anchor_input = layers.Input(name=\"anchor\", shape=(img_size, img_size, 3))\n    positive_input = layers.Input(name=\"positive\", shape=(img_size, img_size, 3))\n    negative_input = layers.Input(name=\"negative\", shape=(img_size, img_size, 3))\n    \n    base= tf.keras.applications.ResNet50V2(input_shape=(img_size, img_size, 3),\n                                           include_top=False, pooling='avg')\n    for layer in base.layers:\n        if isinstance(layer, layers.BatchNormalization):\n            layer.trainable = False\n        else:\n            layer.trainable = True\n    \n    dropout = layers.Dropout(0.25, name='dropout')\n    reduce = layers.Dense(512, activation='linear', name='reduce')\n    \n    distances = eluDistance()(\n        reduce(dropout(base(anchor_input))),\n        reduce(dropout(base(positive_input))),\n        reduce(dropout(base(negative_input))),\n    )\n    \n    return  tf.keras.Model(inputs=[anchor_input, positive_input, negative_input], outputs=distances)\n\nclass SiameseModel(tf.keras.Model):\n    def __init__(self, siamese_network, margin=0.5):\n        super(SiameseModel, self).__init__()\n        self.siamese_network = siamese_network\n        self.margin = margin\n        self.loss_tracker = tf.keras.metrics.Mean(name=\"loss\")\n        \n    def call(self, inputs):\n        return self.siamese_network(inputs)\n    \n    def _compute_loss(self, data):\n        ap_distance, an_distance= self.siamese_network(data)\n        loss = ap_distance - an_distance\n        loss = tf.maximum(loss + self.margin, 0.0)\n        return loss\n    \n    def train_step(self, data):\n        with tf.GradientTape() as tape:\n            loss = self._compute_loss(data)\n        gradients = tape.gradient(loss, self.siamese_network.trainable_weights)\n        self.optimizer.apply_gradients(zip(gradients, self.siamese_network.trainable_weights))\n        self.loss_tracker.update_state(loss)\n        return {\"loss\": self.loss_tracker.result()}\n    \n    def test_step(self, data):\n        loss = self._compute_loss(data)\n        self.loss_tracker.update_state(loss)\n        return {\"loss\": self.loss_tracker.result()}\n    @property\n    def metrics(self):\n        return [self.loss_tracker]","metadata":{"execution":{"iopub.status.busy":"2022-02-20T07:20:35.290863Z","iopub.execute_input":"2022-02-20T07:20:35.291202Z","iopub.status.idle":"2022-02-20T07:20:35.311315Z","shell.execute_reply.started":"2022-02-20T07:20:35.291158Z","shell.execute_reply":"2022-02-20T07:20:35.310098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    model=buildModel()\n    siamese_model = SiameseModel(model)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-02-20T07:20:35.313391Z","iopub.execute_input":"2022-02-20T07:20:35.313972Z","iopub.status.idle":"2022-02-20T07:20:49.925163Z","shell.execute_reply.started":"2022-02-20T07:20:35.313926Z","shell.execute_reply":"2022-02-20T07:20:49.924358Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    encoder = tf.keras.Sequential([\n        siamese_model.siamese_network.get_layer('resnet50v2'),\n        siamese_model.siamese_network.get_layer('dropout'),\n        siamese_model.siamese_network.get_layer('reduce'),\n    ])\n    encoder.load_weights('../input/learn-image-embedding-copy2/encoder.h5')\ndel siamese_model","metadata":{"execution":{"iopub.status.busy":"2022-02-20T07:20:49.926362Z","iopub.execute_input":"2022-02-20T07:20:49.926584Z","iopub.status.idle":"2022-02-20T07:20:55.523079Z","shell.execute_reply.started":"2022-02-20T07:20:49.926559Z","shell.execute_reply":"2022-02-20T07:20:55.522111Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    trainDataX= encoder.predict(dtrain, verbose=1)\n    testDataX= encoder.predict(deval, verbose=1)\ntrainDataY= df.id_label.values","metadata":{"execution":{"iopub.status.busy":"2022-02-20T07:20:55.524504Z","iopub.execute_input":"2022-02-20T07:20:55.524833Z","iopub.status.idle":"2022-02-20T07:25:42.590174Z","shell.execute_reply.started":"2022-02-20T07:20:55.524793Z","shell.execute_reply":"2022-02-20T07:25:42.589101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(trainDataX.shape, testDataX.shape, trainDataY.shape)\nnp.save('trainDataXv2', trainDataX)\nnp.save('testDataXv2', testDataX)\nnp.save('trainDataYv2', trainDataY)","metadata":{"execution":{"iopub.status.busy":"2022-02-20T07:25:42.591016Z","iopub.status.idle":"2022-02-20T07:25:42.591772Z","shell.execute_reply.started":"2022-02-20T07:25:42.591522Z","shell.execute_reply":"2022-02-20T07:25:42.591546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}