{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"## (In Progress)","metadata":{}},{"cell_type":"markdown","source":"\n# Hello everybody! As a basis I used this incredible notebook: https://www.kaggle.com/ks2019/happywhale-arcface-baseline-tpu\n\n# In this notebook, I used EfficientNetB6 as the base model. \n# In the original notebook, we made predictions using only one model trained on one fold. I changed the original code and now we make predictions based on 5 trained models. I trained each model separately, because it's the fastest.\n# The use of 5 models increased the accuracy by about 5%, which is a great result","metadata":{}},{"cell_type":"markdown","source":"\nVersion changes:\n\nVersion 1: Quick Save (I always forget about the settings for TPU:-))\n\nVersion 2: (EFF_NET = 6, KNN = 50, Public Score=0.583)\n\nVersion 3: Quick Save (I always forget about the settings for TPU:-))\n\nVersion 4: (EFF_NET = 6, KNN = 100, Public Score=0.584)\n\nVersion 5: (EFF_NET = 6, KNN = 200, Public Score=0.583)\n\nVersion 6: Quick Save (I always forget about the settings for TPU:-))\n\nVersion 7: (EFF_NET = 5, KNN = 200, Public Score=0.586)\n\nVersion 8: Quick Save :-)\n\nVersion 9: ................\n\nVersion 10: (EFF_NET = 5, KNN = 100, IMAGE_SIZE = 768, Public Score=)","metadata":{}},{"cell_type":"code","source":"import os\nIS_COLAB = not os.path.exists('/kaggle/input')\nprint(IS_COLAB) ","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:09.734304Z","iopub.execute_input":"2022-03-11T16:26:09.734705Z","iopub.status.idle":"2022-03-11T16:26:09.768693Z","shell.execute_reply.started":"2022-03-11T16:26:09.734605Z","shell.execute_reply":"2022-03-11T16:26:09.767848Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is\n    # set: this is always the case on Kaggle.\n    tpu = tf.distribute.cluster_resolver.TPUClusterResolver()\n    print('Running on TPU ', tpu.master())\nexcept ValueError:\n    tpu = None\n\nif tpu:\n    tf.config.experimental_connect_to_cluster(tpu)\n    tf.tpu.experimental.initialize_tpu_system(tpu)\n    strategy = tf.distribute.TPUStrategy(tpu)\nelse:\n    # Default distribution strategy in Tensorflow. Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy()\n\nAUTO = tf.data.experimental.AUTOTUNE\nprint(\"REPLICAS: \", strategy.num_replicas_in_sync)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:09.77106Z","iopub.execute_input":"2022-03-11T16:26:09.771696Z","iopub.status.idle":"2022-03-11T16:26:21.68756Z","shell.execute_reply.started":"2022-03-11T16:26:09.771646Z","shell.execute_reply":"2022-03-11T16:26:21.686809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if IS_COLAB:\n  from google.colab import drive\n  drive.mount('/content/drive')\nelse:\n  from kaggle_datasets import KaggleDatasets","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:21.689231Z","iopub.execute_input":"2022-03-11T16:26:21.689552Z","iopub.status.idle":"2022-03-11T16:26:21.697255Z","shell.execute_reply.started":"2022-03-11T16:26:21.689513Z","shell.execute_reply":"2022-03-11T16:26:21.696353Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q efficientnet\n!pip install tensorflow_addons\nimport re\nimport os\nimport numpy as np\nimport pandas as pd\nimport random\nimport math\nimport tensorflow as tf\nimport efficientnet.tfkeras as efn\nfrom sklearn import metrics\nfrom sklearn.model_selection import KFold, train_test_split\nfrom tensorflow.keras import backend as K\nimport tensorflow_addons as tfa\nfrom tqdm.auto import tqdm\nimport matplotlib.pyplot as plt\nimport pickle\nimport json\nimport tensorflow_hub as tfhub\nfrom datetime import datetime","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:21.699521Z","iopub.execute_input":"2022-03-11T16:26:21.699764Z","iopub.status.idle":"2022-03-11T16:26:41.052899Z","shell.execute_reply.started":"2022-03-11T16:26:21.699736Z","shell.execute_reply":"2022-03-11T16:26:41.052157Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Config","metadata":{}},{"cell_type":"code","source":"save_dir = '.'\nEXPERIMENT = 0\nrun_ts = datetime.now().strftime('%Y%m%d-%H%M%S')\nprint(run_ts)\nif IS_COLAB:\n    save_dir = f'/content/drive/MyDrive/Kaggle/HappyWhale-2022/experiments-{EXPERIMENT}/{run_ts}'\n    !mkdir -p {save_dir}","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:41.054021Z","iopub.execute_input":"2022-03-11T16:26:41.054673Z","iopub.status.idle":"2022-03-11T16:26:41.062984Z","shell.execute_reply.started":"2022-03-11T16:26:41.054625Z","shell.execute_reply":"2022-03-11T16:26:41.062068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class config:\n    \n    \n    SEED = 42\n    FOLD_TO_RUN = 0#In this notebook, we do not train models \n    FOLDS = 5\n    DEBUG = False\n    EVALUATE = True\n    RESUME = False\n    RESUME_EPOCH = None\n    \n    \n    ### Dataset\n    BATCH_SIZE = 32 * strategy.num_replicas_in_sync\n    IMAGE_SIZE = 128\n    N_CLASSES = 15587\n    \n    ### Model\n    model_type = 'effnetv1'  \n    EFF_NET = 5\n    EFF_NETV2 = 's-21k-ft1k'\n    FREEZE_BATCH_NORM = False\n    head = 'arcface' \n    EPOCHS = 20\n    LR = 0.001\n    message='baseline'\n    \n    ### Augmentations\n    CUTOUT = False\n    \n    ### Save-Directory\n    save_dir = save_dir\n    \n    ### Inference\n    KNN = 100\n    \ndef count_data_items(filenames):\n    n = [int(re.compile(r\"-([0-9]*)\\.\").search(filename).group(1)) \n         for filename in filenames]\n    return np.sum(n)\n\n# Function to seed everything\ndef seed_everything(seed):\n    random.seed(seed)\n    np.random.seed(seed)\n    os.environ['PYTHONHASHSEED'] = str(seed)\n    tf.random.set_seed(seed)\n    \ndef is_interactive():\n    return 'runtime'    in get_ipython().config.IPKernelApp.connection_file\nIS_INTERACTIVE = is_interactive()\nprint(IS_INTERACTIVE)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:41.064574Z","iopub.execute_input":"2022-03-11T16:26:41.065045Z","iopub.status.idle":"2022-03-11T16:26:41.07615Z","shell.execute_reply.started":"2022-03-11T16:26:41.065016Z","shell.execute_reply":"2022-03-11T16:26:41.075511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_NAME = None\nif config.model_type == 'effnetv1':\n    MODEL_NAME = f'effnetv1_b{config.EFF_NET}'\nelif config.model_type == 'effnetv2':\n    MODEL_NAME = f'effnetv2_{config.EFF_NETV2}'\n\nconfig.MODEL_NAME = MODEL_NAME\nprint(MODEL_NAME)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:41.077352Z","iopub.execute_input":"2022-03-11T16:26:41.077769Z","iopub.status.idle":"2022-03-11T16:26:41.088898Z","shell.execute_reply.started":"2022-03-11T16:26:41.077736Z","shell.execute_reply":"2022-03-11T16:26:41.088128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with open(config.save_dir+'/config.json', 'w') as fp:\n    json.dump({x:dict(config.__dict__)[x] for x in dict(config.__dict__) if not x.startswith('_')}, fp)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:41.0903Z","iopub.execute_input":"2022-03-11T16:26:41.090781Z","iopub.status.idle":"2022-03-11T16:26:41.09836Z","shell.execute_reply.started":"2022-03-11T16:26:41.090741Z","shell.execute_reply":"2022-03-11T16:26:41.097704Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"GCS_PATH = 'gs://kds-d916c3252bf3bc5b3500b904f05f51ce57c8df85221d11b7711bcda9'  # Get GCS Path from kaggle notebook if GCS Path is expired\nif not IS_COLAB:\n    GCS_PATH = KaggleDatasets().get_gcs_path('happywhale-tfrecords-v1')\n    \ntrain_files = np.sort(np.array(tf.io.gfile.glob(GCS_PATH + '/happywhale-2022-train*.tfrec')))\ntest_files = np.sort(np.array(tf.io.gfile.glob(GCS_PATH + '/happywhale-2022-test*.tfrec')))\nprint(GCS_PATH)\nprint(len(train_files),len(test_files),count_data_items(train_files),count_data_items(test_files))","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:41.099582Z","iopub.execute_input":"2022-03-11T16:26:41.099868Z","iopub.status.idle":"2022-03-11T16:26:41.809811Z","shell.execute_reply.started":"2022-03-11T16:26:41.099841Z","shell.execute_reply":"2022-03-11T16:26:41.809017Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data","metadata":{}},{"cell_type":"code","source":"def arcface_format(posting_id, image, label_group, matches):\n    return posting_id, {'inp1': image, 'inp2': label_group}, label_group, matches\n\ndef arcface_inference_format(posting_id, image, label_group, matches):\n    return image,posting_id\n\ndef arcface_eval_format(posting_id, image, label_group, matches):\n    return image,label_group\n\n# Data augmentation function\ndef data_augment(posting_id, image, label_group, matches):\n\n    ### CUTOUT\n    if tf.random.uniform([])>0.5 and config.CUTOUT:\n      N_CUTOUT = 6\n      for cutouts in range(N_CUTOUT):\n        if tf.random.uniform([])>0.5:\n           DIM = config.IMAGE_SIZE\n           CUTOUT_LENGTH = DIM//8\n           x1 = tf.cast( tf.random.uniform([],0,DIM-CUTOUT_LENGTH),tf.int32)\n           x2 = tf.cast( tf.random.uniform([],0,DIM-CUTOUT_LENGTH),tf.int32)\n           filter_ = tf.concat([tf.zeros((x1,CUTOUT_LENGTH)),tf.ones((CUTOUT_LENGTH,CUTOUT_LENGTH)),tf.zeros((DIM-x1-CUTOUT_LENGTH,CUTOUT_LENGTH))],axis=0)\n           filter_ = tf.concat([tf.zeros((DIM,x2)),filter_,tf.zeros((DIM,DIM-x2-CUTOUT_LENGTH))],axis=1)\n           cutout = tf.reshape(1-filter_,(DIM,DIM,1))\n           image = cutout*image\n\n    image = tf.image.random_flip_left_right(image)\n    # image = tf.image.random_flip_up_down(image)\n    image = tf.image.random_hue(image, 0.01)\n    image = tf.image.random_saturation(image, 0.70, 1.30)\n    image = tf.image.random_contrast(image, 0.80, 1.20)\n    image = tf.image.random_brightness(image, 0.10)\n    return posting_id, image, label_group, matches\n\n# Function to decode our images\ndef decode_image(image_data):\n    image = tf.image.decode_jpeg(image_data, channels = 3)\n    image = tf.image.resize(image, [config.IMAGE_SIZE,config.IMAGE_SIZE])\n    image = tf.cast(image, tf.float32) / 255.0\n    return image\n\n# This function parse our images and also get the target variable\ndef read_labeled_tfrecord(example):\n    LABELED_TFREC_FORMAT = {\n        \"image_name\": tf.io.FixedLenFeature([], tf.string),\n        \"image\": tf.io.FixedLenFeature([], tf.string),\n        \"target\": tf.io.FixedLenFeature([], tf.int64),\n#         \"matches\": tf.io.FixedLenFeature([], tf.string)\n    }\n\n    example = tf.io.parse_single_example(example, LABELED_TFREC_FORMAT)\n    posting_id = example['image_name']\n    image = decode_image(example['image'])\n#     label_group = tf.one_hot(tf.cast(example['label_group'], tf.int32), depth = N_CLASSES)\n    label_group = tf.cast(example['target'], tf.int32)\n#     matches = example['matches']\n    matches = 1\n    return posting_id, image, label_group, matches\n\n# This function loads TF Records and parse them into tensors\ndef load_dataset(filenames, ordered = False):\n    \n    ignore_order = tf.data.Options()\n    if not ordered:\n        ignore_order.experimental_deterministic = False \n        \n    dataset = tf.data.TFRecordDataset(filenames, num_parallel_reads = AUTO)\n#     dataset = dataset.cache()\n    dataset = dataset.with_options(ignore_order)\n    dataset = dataset.map(read_labeled_tfrecord, num_parallel_calls = AUTO) \n    return dataset\n\n# This function is to get our training tensors\ndef get_training_dataset(filenames):\n    dataset = load_dataset(filenames, ordered = False)\n    dataset = dataset.map(data_augment, num_parallel_calls = AUTO)\n    dataset = dataset.map(arcface_format, num_parallel_calls = AUTO)\n    dataset = dataset.map(lambda posting_id, image, label_group, matches: (image, label_group))\n    dataset = dataset.repeat()\n    dataset = dataset.shuffle(2048)\n    dataset = dataset.batch(config.BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\n# This function is to get our training tensors\ndef get_val_dataset(filenames):\n    dataset = load_dataset(filenames, ordered = True)\n    dataset = dataset.map(data_augment, num_parallel_calls = AUTO)\n    dataset = dataset.map(arcface_format, num_parallel_calls = AUTO)\n    dataset = dataset.map(lambda posting_id, image, label_group, matches: (image, label_group))\n    dataset = dataset.batch(config.BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\n# This function is to get our training tensors\ndef get_eval_dataset(filenames, get_targets = True):\n    dataset = load_dataset(filenames, ordered = True)\n    dataset = dataset.map(data_augment, num_parallel_calls = AUTO)\n    dataset = dataset.map(arcface_eval_format, num_parallel_calls = AUTO)\n    if not get_targets:\n        dataset = dataset.map(lambda image, target: image)\n    dataset = dataset.batch(config.BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset\n\n# This function is to get our training tensors\ndef get_test_dataset(filenames, get_names = True):\n    dataset = load_dataset(filenames, ordered = True)\n    dataset = dataset.map(data_augment, num_parallel_calls = AUTO)\n    dataset = dataset.map(arcface_inference_format, num_parallel_calls = AUTO)\n    if not get_names:\n        dataset = dataset.map(lambda image, posting_id: image)\n    dataset = dataset.batch(config.BATCH_SIZE)\n    dataset = dataset.prefetch(AUTO)\n    return dataset","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:41.812924Z","iopub.execute_input":"2022-03-11T16:26:41.813187Z","iopub.status.idle":"2022-03-11T16:26:41.844916Z","shell.execute_reply.started":"2022-03-11T16:26:41.813157Z","shell.execute_reply":"2022-03-11T16:26:41.843808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n'''row = 10; col = 8;\nrow = min(row,config.BATCH_SIZE//col)\nN_TRAIN = count_data_items(train_files)\nprint(N_TRAIN)\nds = get_training_dataset(train_files)\n\nfor (sample,label) in ds:\n    img = sample['inp1']\n    plt.figure(figsize=(25,int(25*row/col)))\n    for j in range(row*col):\n        plt.subplot(row,col,j+1)\n        plt.title(label[j].numpy())\n        plt.axis('off')\n        plt.imshow(img[j,])\n    plt.show()\n    break\nprint(img.shape)'''","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:41.846097Z","iopub.execute_input":"2022-03-11T16:26:41.846321Z","iopub.status.idle":"2022-03-11T16:26:41.864231Z","shell.execute_reply.started":"2022-03-11T16:26:41.846295Z","shell.execute_reply":"2022-03-11T16:26:41.863115Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''row = 10; col = 8;\nrow = min(row,config.BATCH_SIZE//col)\nN_TEST = count_data_items(test_files)\nprint(N_TEST)\nds = get_test_dataset(test_files)\n\nfor (img,label) in ds:\n    plt.figure(figsize=(25,int(25*row/col)))\n    for j in range(row*col):\n        plt.subplot(row,col,j+1)\n        plt.title(label[j].numpy())\n        plt.axis('off')\n        plt.imshow(img[j,])\n    plt.show()\n    break\nprint(img.shape)'''","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:41.865737Z","iopub.execute_input":"2022-03-11T16:26:41.866129Z","iopub.status.idle":"2022-03-11T16:26:41.876042Z","shell.execute_reply.started":"2022-03-11T16:26:41.866088Z","shell.execute_reply":"2022-03-11T16:26:41.874945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Model","metadata":{}},{"cell_type":"code","source":"# Arcmarginproduct class keras layer\nclass ArcMarginProduct(tf.keras.layers.Layer):\n    '''\n    Implements large margin arc distance.\n\n    Reference:\n        https://arxiv.org/pdf/1801.07698.pdf\n        https://github.com/lyakaap/Landmark2019-1st-and-3rd-Place-Solution/\n            blob/master/src/modeling/metric_learning.py\n    '''\n    def __init__(self, n_classes, s=30, m=0.50, easy_margin=False,\n                 ls_eps=0.0, **kwargs):\n\n        super(ArcMarginProduct, self).__init__(**kwargs)\n\n        self.n_classes = n_classes\n        self.s = s\n        self.m = m\n        self.ls_eps = ls_eps\n        self.easy_margin = easy_margin\n        self.cos_m = tf.math.cos(m)\n        self.sin_m = tf.math.sin(m)\n        self.th = tf.math.cos(math.pi - m)\n        self.mm = tf.math.sin(math.pi - m) * m\n\n    def get_config(self):\n\n        config = super().get_config().copy()\n        config.update({\n            'n_classes': self.n_classes,\n            's': self.s,\n            'm': self.m,\n            'ls_eps': self.ls_eps,\n            'easy_margin': self.easy_margin,\n        })\n        return config\n\n    def build(self, input_shape):\n        super(ArcMarginProduct, self).build(input_shape[0])\n\n        self.W = self.add_weight(\n            name='W',\n            shape=(int(input_shape[0][-1]), self.n_classes),\n            initializer='glorot_uniform',\n            dtype='float32',\n            trainable=True,\n            regularizer=None)\n\n    def call(self, inputs):\n        X, y = inputs\n        y = tf.cast(y, dtype=tf.int32)\n        cosine = tf.matmul(\n            tf.math.l2_normalize(X, axis=1),\n            tf.math.l2_normalize(self.W, axis=0)\n        )\n        sine = tf.math.sqrt(1.0 - tf.math.pow(cosine, 2))\n        phi = cosine * self.cos_m - sine * self.sin_m\n        if self.easy_margin:\n            phi = tf.where(cosine > 0, phi, cosine)\n        else:\n            phi = tf.where(cosine > self.th, phi, cosine - self.mm)\n        one_hot = tf.cast(\n            tf.one_hot(y, depth=self.n_classes),\n            dtype=cosine.dtype\n        )\n        if self.ls_eps > 0:\n            one_hot = (1 - self.ls_eps) * one_hot + self.ls_eps / self.n_classes\n\n        output = (one_hot * phi) + ((1.0 - one_hot) * cosine)\n        output *= self.s\n        return output","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:41.877516Z","iopub.execute_input":"2022-03-11T16:26:41.878156Z","iopub.status.idle":"2022-03-11T16:26:41.896959Z","shell.execute_reply.started":"2022-03-11T16:26:41.878111Z","shell.execute_reply":"2022-03-11T16:26:41.895831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"'''EFNS = [efn.EfficientNetB0, efn.EfficientNetB1, efn.EfficientNetB2, efn.EfficientNetB3, \n        efn.EfficientNetB4, efn.EfficientNetB5, efn.EfficientNetB6, efn.EfficientNetB7]\n\ndef freeze_BN(model):\n    # Unfreeze layers while leaving BatchNorm layers frozen\n    for layer in model.layers:\n        if not isinstance(layer, tf.keras.layers.BatchNormalization):\n            layer.trainable = True\n        else:\n            layer.trainable = False\n\n# Function to create our EfficientNetB3 model\ndef get_model():\n\n    if config.head=='arcface':\n        head = ArcMarginProduct\n    else:\n        assert 1==2, \"INVALID HEAD\"\n    \n    with strategy.scope():\n        \n        margin = head(\n            n_classes = config.N_CLASSES, \n            s = 30, \n            m = 0.3, \n            name=f'head/{config.head}', \n            dtype='float32'\n            )\n\n        inp = tf.keras.layers.Input(shape = [config.IMAGE_SIZE, config.IMAGE_SIZE, 3], name = 'inp1')\n        label = tf.keras.layers.Input(shape = (), name = 'inp2')\n        \n        if config.model_type == 'effnetv1':\n            x = EFNS[config.EFF_NET](weights = 'noisy-student', include_top = False)(inp)\n            embed = tf.keras.layers.GlobalAveragePooling2D()(x)\n        elif config.model_type == 'effnetv2':\n            FEATURE_VECTOR = f'{EFFNETV2_ROOT}/tfhub_models/efficientnetv2-{config.EFF_NETV2}/feature_vector'\n            embed = tfhub.KerasLayer(FEATURE_VECTOR, trainable=True)(inp)\n            \n        embed = tf.keras.layers.Dropout(0.2)(embed)\n        embed = tf.keras.layers.Dense(512)(embed)\n        x = margin([embed, label])\n        \n        output = tf.keras.layers.Softmax(dtype='float32')(x)\n        \n        model = tf.keras.models.Model(inputs = [inp, label], outputs = [output])\n        embed_model = tf.keras.models.Model(inputs = inp, outputs = embed)  \n        \n        opt = tf.keras.optimizers.Adam(learning_rate = config.LR)\n        if config.FREEZE_BATCH_NORM:\n            freeze_BN(model)\n\n        model.compile(\n            optimizer = opt,\n            loss = [tf.keras.losses.SparseCategoricalCrossentropy()],\n            metrics = [tf.keras.metrics.SparseCategoricalAccuracy(),tf.keras.metrics.SparseTopKCategoricalAccuracy(k=5)]\n            ) \n        \n        return model,embed_model'''","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:41.898643Z","iopub.execute_input":"2022-03-11T16:26:41.898945Z","iopub.status.idle":"2022-03-11T16:26:41.916922Z","shell.execute_reply.started":"2022-03-11T16:26:41.898907Z","shell.execute_reply":"2022-03-11T16:26:41.915934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_lr_callback(plot=False):\n    lr_start   = 0.000001\n    lr_max     = 0.000005 * config.BATCH_SIZE  \n    lr_min     = 0.000001\n    lr_ramp_ep = 4\n    lr_sus_ep  = 0\n    lr_decay   = 0.9\n   \n    def lrfn(epoch):\n        if config.RESUME:\n            epoch = epoch + config.RESUME_EPOCH\n        if epoch < lr_ramp_ep:\n            lr = (lr_max - lr_start) / lr_ramp_ep * epoch + lr_start\n            \n        elif epoch < lr_ramp_ep + lr_sus_ep:\n            lr = lr_max\n            \n        else:\n            lr = (lr_max - lr_min) * lr_decay**(epoch - lr_ramp_ep - lr_sus_ep) + lr_min\n            \n        return lr\n        \n    if plot:\n        epochs = list(range(config.EPOCHS))\n        learning_rates = [lrfn(x) for x in epochs]\n        plt.scatter(epochs,learning_rates)\n        plt.show()\n\n    lr_callback = tf.keras.callbacks.LearningRateScheduler(lrfn, verbose=False)\n    return lr_callback\n\nget_lr_callback(plot=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:41.918212Z","iopub.execute_input":"2022-03-11T16:26:41.919066Z","iopub.status.idle":"2022-03-11T16:26:42.175169Z","shell.execute_reply.started":"2022-03-11T16:26:41.919023Z","shell.execute_reply":"2022-03-11T16:26:42.174566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class Snapshot(tf.keras.callbacks.Callback):\n    \n    def __init__(self,fold,snapshot_epochs=[]):\n        super(Snapshot, self).__init__()\n        self.snapshot_epochs = snapshot_epochs\n        self.fold = fold\n        \n        \n    def on_epoch_end(self, epoch, logs=None):\n        # logs is a dictionary\n#         print(f\"epoch: {epoch}, train_acc: {logs['acc']}, valid_acc: {logs['val_acc']}\")\n        if epoch in self.snapshot_epochs: # your custom condition         \n            self.model.save_weights(config.save_dir+f\"/EF{config.MODEL_NAME}_epoch{epoch}.h5\")\n        self.model.save_weights(config.save_dir+f\"/{config.MODEL_NAME}_last.h5\")","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:42.176345Z","iopub.execute_input":"2022-03-11T16:26:42.176763Z","iopub.status.idle":"2022-03-11T16:26:42.183297Z","shell.execute_reply.started":"2022-03-11T16:26:42.176733Z","shell.execute_reply":"2022-03-11T16:26:42.182537Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Train","metadata":{}},{"cell_type":"code","source":"TRAINING_FILENAMES = [x for i,x in enumerate(train_files) if i%config.FOLDS!=config.FOLD_TO_RUN]\nVALIDATION_FILENAMES = [x for i,x in enumerate(train_files) if i%config.FOLDS==config.FOLD_TO_RUN]\nprint(len(TRAINING_FILENAMES),len(VALIDATION_FILENAMES),count_data_items(TRAINING_FILENAMES),count_data_items(VALIDATION_FILENAMES))","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:42.18458Z","iopub.execute_input":"2022-03-11T16:26:42.184796Z","iopub.status.idle":"2022-03-11T16:26:42.201713Z","shell.execute_reply.started":"2022-03-11T16:26:42.184772Z","shell.execute_reply":"2022-03-11T16:26:42.200652Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"if config.DEBUG:\n    TRAINING_FILENAMES = [TRAINING_FILENAMES[0]]\n    VALIDATION_FILENAMES = [VALIDATION_FILENAMES[0]]\n    print(len(TRAINING_FILENAMES),len(VALIDATION_FILENAMES),count_data_items(TRAINING_FILENAMES),count_data_items(VALIDATION_FILENAMES))\n    test_files = [test_files[0]]","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:42.20344Z","iopub.execute_input":"2022-03-11T16:26:42.20407Z","iopub.status.idle":"2022-03-11T16:26:42.212185Z","shell.execute_reply.started":"2022-03-11T16:26:42.204025Z","shell.execute_reply":"2022-03-11T16:26:42.21142Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"seed_everything(config.SEED)\nVERBOSE = 1\n'''train_dataset = get_training_dataset(TRAINING_FILENAMES)\nval_dataset = get_val_dataset(VALIDATION_FILENAMES)\nSTEPS_PER_EPOCH = count_data_items(TRAINING_FILENAMES) // config.BATCH_SIZE\ntrain_logger = tf.keras.callbacks.CSVLogger(config.save_dir+'/training-log-fold-%i.h5.csv'%config.FOLD_TO_RUN)\n# SAVE BEST MODEL EACH FOLD        \nsv_loss = tf.keras.callbacks.ModelCheckpoint(\n    config.save_dir+f\"/{config.MODEL_NAME}_loss_{config.FOLD_TO_RUN}.h5\", monitor='val_loss', verbose=0, save_best_only=True,\n    save_weights_only=True, mode='min', save_freq='epoch')\n# BUILD MODEL\nK.clear_session()\nmodel,embed_model = get_model()\nsnap = Snapshot(fold=config.FOLD_TO_RUN,snapshot_epochs=[5,8])\nmodel.summary()\n\nif config.RESUME:   \n    model.load_weights(config.resume_model_wts)'''","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:42.213931Z","iopub.execute_input":"2022-03-11T16:26:42.214334Z","iopub.status.idle":"2022-03-11T16:26:42.226308Z","shell.execute_reply.started":"2022-03-11T16:26:42.214213Z","shell.execute_reply":"2022-03-11T16:26:42.225377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print('#### Image Size %i with EfficientNet B%i and batch_size %i'%\n      (config.IMAGE_SIZE,config.EFF_NET,config.BATCH_SIZE))\n\"\"\"#In this notebook, we do not train models \nhistory = model.fit(train_dataset,\n                validation_data = val_dataset,\n                steps_per_epoch = STEPS_PER_EPOCH,\n                epochs = config.EPOCHS,\n                callbacks = [snap,get_lr_callback(),train_logger,sv_loss], \n                verbose = VERBOSE)\n                \n\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:42.228184Z","iopub.execute_input":"2022-03-11T16:26:42.228777Z","iopub.status.idle":"2022-03-11T16:26:42.244027Z","shell.execute_reply.started":"2022-03-11T16:26:42.228742Z","shell.execute_reply":"2022-03-11T16:26:42.242517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#with open(config.save_dir+f'train_{filename.split(\"/\")[-1]}_{config.FOLD_TO_RUN}.npy', 'rb') as f:\n","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:42.245443Z","iopub.execute_input":"2022-03-11T16:26:42.245719Z","iopub.status.idle":"2022-03-11T16:26:42.252949Z","shell.execute_reply.started":"2022-03-11T16:26:42.24569Z","shell.execute_reply":"2022-03-11T16:26:42.252178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#model.load_weights(config.save_dir+f\"/{config.MODEL_NAME}_loss.h5\")\n#embed_models=[]\n#for i in range(5):\n #   model,embed_model = get_model()\n #   embed_models.append((model.load_weights(f\"../input/happywhale-arcface-eff5-768/effnetv1_b5_loss_{i}.h5\"),embed_model))\n    ","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:42.254217Z","iopub.execute_input":"2022-03-11T16:26:42.254454Z","iopub.status.idle":"2022-03-11T16:26:42.265723Z","shell.execute_reply.started":"2022-03-11T16:26:42.254427Z","shell.execute_reply":"2022-03-11T16:26:42.264936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:42.266702Z","iopub.execute_input":"2022-03-11T16:26:42.266941Z","iopub.status.idle":"2022-03-11T16:26:42.278112Z","shell.execute_reply.started":"2022-03-11T16:26:42.266913Z","shell.execute_reply":"2022-03-11T16:26:42.277508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluation","metadata":{}},{"cell_type":"code","source":"def better_than_median(inputs, axis):\n    \"\"\"Compute the mean of the predictions if there are no outliers,\n    or the median if there are outliers.\n\n    Parameter: inputs = ndarray of shape (n_samples, n_folds)\"\"\"\n    spread = inputs.max(axis=axis) - inputs.min(axis=axis) \n    spread_lim = 0.45\n    print(f\"Inliers:  {(spread < spread_lim).sum():7} -> compute mean\")\n    print(f\"Outliers: {(spread >= spread_lim).sum():7} -> compute median\")\n    print(f\"Total:    {len(inputs):7}\")\n    return np.where(spread < spread_lim,\n                    np.mean(inputs, axis=axis),\n                    np.median(inputs, axis=axis))","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:42.279317Z","iopub.execute_input":"2022-03-11T16:26:42.280142Z","iopub.status.idle":"2022-03-11T16:26:42.291909Z","shell.execute_reply.started":"2022-03-11T16:26:42.280097Z","shell.execute_reply":"2022-03-11T16:26:42.291174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#my functions\ndef get_embeddings_np(filename,data_types='train',kfold_list=[x for x in range(5)],dataset='../input/eff7-new-768'):#x for x in range(6)\n    #ds = get_test_dataset([filename],get_names=False)\n    #embeddings = np.mean(np.stack([embed_models[x][1].predict(ds,verbose=0) for x in range(5)]), axis=0)\n    #print (embeddings.shape)\n    val_train={'train':'val','val':'train','test':'test'}\n    embeddings=None\n    for kfold in kfold_list:\n        path=f'{dataset}/{data_types}_{filename.split(\"/\")[-1]}_{kfold}.npy'\n        if os.path.exists(path):\n            print(path)\n            with open(path, 'rb') as f:\n                if embeddings is None:\n                    embeddings=np.load(f)\n                else:\n                    embeddings=np.concatenate((embeddings,np.load(f)),axis=1)\n        else:\n            path=f'{dataset}/{val_train[data_types]}_{filename.split(\"/\")[-1]}_{kfold}.npy'\n            if os.path.exists(path):\n                print(path)\n                with open(path, 'rb') as f:\n                    if embeddings is None:\n                        embeddings=np.load(f)\n                    else:\n                        embeddings=np.concatenate((embeddings,np.load(f)),axis=1)\n    return embeddings","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:42.293Z","iopub.execute_input":"2022-03-11T16:26:42.293339Z","iopub.status.idle":"2022-03-11T16:26:42.30827Z","shell.execute_reply.started":"2022-03-11T16:26:42.293313Z","shell.execute_reply":"2022-03-11T16:26:42.307294Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_ids(filename):\n    ds = get_test_dataset([filename],get_names=True).map(lambda image, image_name: image_name).unbatch()\n    NUM_IMAGES = count_data_items([filename])\n    ids = next(iter(ds.batch(NUM_IMAGES))).numpy().astype('U')\n    return ids\n\ndef get_targets(filename):\n    ds = get_eval_dataset([filename],get_targets=True).map(lambda image, target: target).unbatch()\n    NUM_IMAGES = count_data_items([filename])\n    ids = next(iter(ds.batch(NUM_IMAGES))).numpy()\n    return ids\n\ndef get_embeddings(filename):\n    ds = get_test_dataset([filename],get_names=False)\n    embeddings = np.mean(np.stack([embed_models[x][1].predict(ds,verbose=0) for x in range(5)]), axis=0)\n    #print (embeddings.shape)\n    return embeddings\n\n'''def get_embeddings_np(filename,data_types='train',kfold_list=[x for x in range(5)],dataset='../input/eff7-new-768'):#x for x in range(6)\n    #ds = get_test_dataset([filename],get_names=False)\n    #embeddings = np.mean(np.stack([embed_models[x][1].predict(ds,verbose=0) for x in range(5)]), axis=0)\n    #print (embeddings.shape)\n    val_train={'train':'val','val':'train','test':'test'}\n    embeddings=[]\n    for kfold in kfold_list:\n        path=f'{dataset}/{data_types}_{filename.split(\"/\")[-1]}_{kfold}.npy'\n        if os.path.exists(path):\n            print(path)\n            with open(path, 'rb') as f:\n                embeddings.append(np.load(f))\n        else:\n            path=f'{dataset}/{val_train[data_types]}_{filename.split(\"/\")[-1]}_{kfold}.npy'\n            if os.path.exists(path):\n                print(path)\n                with open(path, 'rb') as f:\n                    embeddings.append(np.load(f))\n                \n                \n    print (len(embeddings))\n    embeddings = np.mean(np.stack(embeddings), axis=0)\n    #embeddings = np.median(np.stack(embeddings), axis=0)\n    #embeddings = better_than_median(np.stack(embeddings), axis=0)\n    \n    return embeddings'''\n\ndef get_predictions(test_df,threshold=0.2):\n    predictions = {}\n    for i,row in tqdm(test_df.iterrows()):\n        if row.image in predictions:\n            if len(predictions[row.image])==5:\n                continue\n            predictions[row.image].append(row.target)\n        elif row.confidence>threshold:\n            predictions[row.image] = [row.target,'new_individual']\n        else:\n            predictions[row.image] = ['new_individual',row.target]\n\n    for x in tqdm(predictions):\n        if len(predictions[x])<5:\n            remaining = [y for y in sample_list if y not in predictions]\n            predictions[x] = predictions[x]+remaining\n            predictions[x] = predictions[x][:5]\n        \n    return predictions\n\ndef map_per_image(label, predictions):\n    \"\"\"Computes the precision score of one image.\n\n    Parameters\n    ----------\n    label : string\n            The true label of the image\n    predictions : list\n            A list of predicted elements (order does matter, 5 predictions allowed per image)\n\n    Returns\n    -------\n    score : double\n    \"\"\"    \n    try:\n        return 1 / (predictions[:5].index(label) + 1)\n    except ValueError:\n        return 0.0\n    \nf = open ('../input/happywhale-splits/individual_ids.json', \"r\")\ntarget_encodings = json.loads(f.read())\ntarget_encodings = {target_encodings[x]:x for x in target_encodings}\nsample_list = ['new_individual', '5bf17305f073', '7593d2aee842', '7362d7a01d00','956562ff2888']","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:42.309675Z","iopub.execute_input":"2022-03-11T16:26:42.309914Z","iopub.status.idle":"2022-03-11T16:26:42.35574Z","shell.execute_reply.started":"2022-03-11T16:26:42.309887Z","shell.execute_reply":"2022-03-11T16:26:42.354723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_predictions_1(test_df):\n    predictions = {}\n    for i,row in tqdm(test_df.iterrows()):\n        if row.image in predictions:\n            if len(predictions[row.image])==5:\n                continue\n            predictions[row.image].append(row.target)\n        else:\n            predictions[row.image] = [row.target]\n            \n    return predictions\ndef get_predictions_2(test_df,predictions):\n    for i,row in tqdm(test_df.iterrows()):\n        if row.image in predictions:\n            if len(predictions[row.image])==5:\n                continue\n            predictions[row.image].append(row.target)\n        else:\n            predictions[row.image] = ['new_individual',row.target] \n            \n    for x in tqdm(predictions):\n        if len(predictions[x])<5:\n            remaining = [y for y in sample_list if y not in predictions]\n            predictions[x] = predictions[x]+remaining\n            predictions[x] = predictions[x][:5]\n    return predictions\n\ndef get_predictions_3(test_df):\n    predictions={}\n    unic_img=np.unique(test_df.image)\n    test_df['confidence']=test_df['confidence']**8\n    for i in tqdm(range(len(unic_img))):\n        temp1=test_df.loc[test_df['image']==unic_img[i]].sort_values('confidence',ascending=False).reset_index(drop=True).loc[0:4]\n        predictions[unic_img[i]] = [temp1.groupby(['target']).sum().sort_values('confidence',ascending=False).index[0],'new_individual']\n    return predictions\ndef get_predictions_4(test_df,predictions):\n    for i,row in tqdm(test_df.iterrows()):\n        if row.image in predictions:\n            if len(predictions[row.image])==5:\n                continue\n            if row.target in predictions[row.image]:\n                continue\n            predictions[row.image].append(row.target)\n        else:\n            predictions[row.image] = ['new_individual',row.target] \n            \n    for x in tqdm(predictions):\n        if len(predictions[x])<5:\n            remaining = [y for y in sample_list if y not in predictions]\n            predictions[x] = predictions[x]+remaining\n            predictions[x] = predictions[x][:5]\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:42.357273Z","iopub.execute_input":"2022-03-11T16:26:42.357619Z","iopub.status.idle":"2022-03-11T16:26:42.376582Z","shell.execute_reply.started":"2022-03-11T16:26:42.357572Z","shell.execute_reply":"2022-03-11T16:26:42.37559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    train_targets = []\n    train_embeddings = []\n    for filename in tqdm(TRAINING_FILENAMES):#TRAINING_FILENAMES\n        embeddings = get_embeddings_np(filename)\n        targets = get_targets(filename)\n        train_embeddings.append(embeddings)\n        train_targets.append(targets)\n    train_embeddings = np.concatenate(train_embeddings)\n    train_targets = np.concatenate(train_targets)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:26:42.381457Z","iopub.execute_input":"2022-03-11T16:26:42.381739Z","iopub.status.idle":"2022-03-11T16:32:30.473286Z","shell.execute_reply.started":"2022-03-11T16:26:42.381709Z","shell.execute_reply":"2022-03-11T16:32:30.472307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_embeddings.shape","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.474757Z","iopub.execute_input":"2022-03-11T16:32:30.475331Z","iopub.status.idle":"2022-03-11T16:32:30.481713Z","shell.execute_reply.started":"2022-03-11T16:32:30.475282Z","shell.execute_reply":"2022-03-11T16:32:30.480929Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_targets=ii","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.483152Z","iopub.execute_input":"2022-03-11T16:32:30.483374Z","iopub.status.idle":"2022-03-11T16:32:30.895297Z","shell.execute_reply.started":"2022-03-11T16:32:30.48335Z","shell.execute_reply":"2022-03-11T16:32:30.893953Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import NearestNeighbors\n'''neigh = NearestNeighbors(n_neighbors=config.KNN,metric='cosine')'''\nneigh = NearestNeighbors(n_neighbors=config.KNN,metric='cosine')\nneigh.fit(train_embeddings)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:59.371412Z","iopub.execute_input":"2022-03-11T16:32:59.371719Z","iopub.status.idle":"2022-03-11T16:32:59.91476Z","shell.execute_reply.started":"2022-03-11T16:32:59.37169Z","shell.execute_reply":"2022-03-11T16:32:59.913729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():\n    test_ids = []\n    test_nn_distances = []\n    test_nn_idxs = []\n    val_targets = []\n    val_embeddings = []\n    for filename in tqdm(VALIDATION_FILENAMES):#(VALIDATION_FILENAMES):\n        embeddings = get_embeddings_np(filename,'val')\n        targets = get_targets(filename)\n        ids = get_ids(filename)\n        distances,idxs = neigh.kneighbors(embeddings, config.KNN, return_distance=True)\n        test_ids.append(ids)\n        test_nn_idxs.append(idxs)\n        test_nn_distances.append(distances)\n        val_embeddings.append(embeddings)\n        val_targets.append(targets)\n    test_nn_distances = np.concatenate(test_nn_distances)\n    test_nn_idxs = np.concatenate(test_nn_idxs)\n    test_ids = np.concatenate(test_ids)\n    val_embeddings = np.concatenate(val_embeddings)\n    val_targets = np.concatenate(val_targets)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:33:06.705642Z","iopub.execute_input":"2022-03-11T16:33:06.706315Z","iopub.status.idle":"2022-03-11T16:36:17.823786Z","shell.execute_reply.started":"2022-03-11T16:33:06.706277Z","shell.execute_reply":"2022-03-11T16:36:17.82281Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_targets","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.90157Z","iopub.status.idle":"2022-03-11T16:32:30.902531Z","shell.execute_reply.started":"2022-03-11T16:32:30.902212Z","shell.execute_reply":"2022-03-11T16:32:30.902243Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"allowed_targets = set([target_encodings[x] for x in np.unique(train_targets)])\nval_targets_df = pd.DataFrame(np.stack([test_ids,val_targets],axis=1),columns=['image','target'])\nval_targets_df['target'] = val_targets_df['target'].astype(int).map(target_encodings)\nval_targets_df.loc[~val_targets_df.target.isin(allowed_targets),'target'] = 'new_individual'\nval_targets_df.target.value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:36:30.760574Z","iopub.execute_input":"2022-03-11T16:36:30.760985Z","iopub.status.idle":"2022-03-11T16:36:30.889161Z","shell.execute_reply.started":"2022-03-11T16:36:30.760941Z","shell.execute_reply":"2022-03-11T16:36:30.888127Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"val_targets_df","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:36:37.751563Z","iopub.execute_input":"2022-03-11T16:36:37.752181Z","iopub.status.idle":"2022-03-11T16:36:37.770077Z","shell.execute_reply.started":"2022-03-11T16:36:37.75213Z","shell.execute_reply":"2022-03-11T16:36:37.769234Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = []\nfor i in tqdm(range(len(test_ids))):\n    id_ = test_ids[i]\n    targets = train_targets[test_nn_idxs[i]]\n    distances = test_nn_distances[i]\n    subset_preds = pd.DataFrame(np.stack([targets,distances],axis=1),columns=['target','distances'])\n    subset_preds['image'] = id_\n    test_df.append(subset_preds)\ntest_df = pd.concat(test_df).reset_index(drop=True)\ntest_df['confidence'] = 1-test_df['distances']\ntest_df = test_df.groupby(['image','target']).confidence.max().reset_index()\ntest_df = test_df.sort_values('confidence',ascending=False).reset_index(drop=True)\ntest_df['target'] = test_df['target'].map(target_encodings)\ntest_df.to_csv('val_neighbors.csv')\ntest_df.image.value_counts().value_counts()\n","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:37:04.697682Z","iopub.execute_input":"2022-03-11T16:37:04.698065Z","iopub.status.idle":"2022-03-11T16:37:16.894174Z","shell.execute_reply.started":"2022-03-11T16:37:04.698027Z","shell.execute_reply":"2022-03-11T16:37:16.892617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    th=0.6\n    all_preds = get_predictions(test_df,threshold=th)\n    cv = 0\n    for i,row in val_targets_df.iterrows():\n        target = row.target\n        preds = all_preds[row.image]\n        val_targets_df.loc[i,th] = map_per_image(target,preds)\n    cv = val_targets_df[th].mean()\n    print(f\"CV at threshold {th}: {cv}\")","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:40:07.895547Z","iopub.execute_input":"2022-03-11T16:40:07.895897Z","iopub.status.idle":"2022-03-11T16:40:40.307463Z","shell.execute_reply.started":"2022-03-11T16:40:07.895865Z","shell.execute_reply":"2022-03-11T16:40:40.306415Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    #with strategy.scope():\n    test_df = []\n    for i in tqdm(range(len(test_ids))):\n        id_ = test_ids[i]\n        targets = train_targets[test_nn_idxs[i]]\n        distances = test_nn_distances[i]\n        subset_preds = pd.DataFrame(np.stack([targets,distances],axis=1),columns=['target','distances'])\n        subset_preds['image'] = id_\n        test_df.append(subset_preds)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.910085Z","iopub.status.idle":"2022-03-11T16:32:30.910762Z","shell.execute_reply.started":"2022-03-11T16:32:30.910566Z","shell.execute_reply":"2022-03-11T16:32:30.910586Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    test_df = pd.concat(test_df).reset_index(drop=True)\n    test_df['confidence'] = 1-test_df['distances']\n    test_df.drop('distances',inplace=True, axis=1)\n    test_df['target'] = test_df['target'].map(target_encodings)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.911719Z","iopub.status.idle":"2022-03-11T16:32:30.912464Z","shell.execute_reply.started":"2022-03-11T16:32:30.912258Z","shell.execute_reply":"2022-03-11T16:32:30.912279Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.913756Z","iopub.status.idle":"2022-03-11T16:32:30.914136Z","shell.execute_reply.started":"2022-03-11T16:32:30.913972Z","shell.execute_reply":"2022-03-11T16:32:30.91399Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"th=0.6\ntest_df1=test_df.loc[test_df['confidence']>th].reset_index(drop=True)\npredictions=get_predictions_3(test_df1)\ntest_df=test_df.groupby(['image','target']).confidence.max().reset_index()\ntest_df=test_df.sort_values('confidence',ascending=False).reset_index(drop=True)\npredictions=get_predictions_4(test_df,predictions)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.915527Z","iopub.status.idle":"2022-03-11T16:32:30.916053Z","shell.execute_reply.started":"2022-03-11T16:32:30.915866Z","shell.execute_reply":"2022-03-11T16:32:30.915886Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"        for i,row in val_targets_df.iterrows():\n            target = row.target\n            preds = predictions[row.image]\n            val_targets_df.loc[i,th] = map_per_image(target,preds)\n        cv = val_targets_df[th].mean()\n        print(f\"CV at threshold {th}: {cv}\")","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.916979Z","iopub.status.idle":"2022-03-11T16:32:30.917709Z","shell.execute_reply.started":"2022-03-11T16:32:30.917519Z","shell.execute_reply":"2022-03-11T16:32:30.917543Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"th=0.65\ntest_df1=test_df.loc[test_df['confidence']>th].reset_index(drop=True)\ntest_df1=test_df1.groupby(['image','target']).confidence.size().reset_index()\ntest_df1 = test_df1.sort_values('confidence',ascending=False).reset_index(drop=True)\nprediction=get_predictions_1(test_df1)\nfor x in tqdm(prediction):\n    if len(prediction[x])<5:\n        prediction[x] = prediction[x]+['new_individual']\ntest_df2=test_df.loc[test_df['confidence']<=th].reset_index(drop=True)\ntest_df2 = test_df2.groupby(['image','target']).confidence.max().reset_index()\ntest_df2 = test_df2.sort_values('confidence',ascending=False).reset_index(drop=True)\ntest_df2['target'] = test_df2['target'].map(target_encodings)\npredictions=get_predictions_2(test_df2,prediction)\nfor i,row in val_targets_df.iterrows():\n    target = row.target\n    preds = predictions[row.image]\n    val_targets_df.loc[i,th] = map_per_image(target,preds)\ncv = val_targets_df[th].mean()\nprint(f\"CV at threshold {th}: {cv}\")","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.918851Z","iopub.status.idle":"2022-03-11T16:32:30.91916Z","shell.execute_reply.started":"2022-03-11T16:32:30.919003Z","shell.execute_reply":"2022-03-11T16:32:30.919018Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Compute CV\nwith strategy.scope():\n    best_th = 0\n    best_cv = 0\n    for th in [0.1*x for x in range(11)]:\n        all_preds = get_predictions(test_df,threshold=th)\n        cv = 0\n        for i,row in val_targets_df.iterrows():\n            target = row.target\n            preds = all_preds[row.image]\n            val_targets_df.loc[i,th] = map_per_image(target,preds)\n        cv = val_targets_df[th].mean()\n        print(f\"CV at threshold {th}: {cv}\")\n        if cv>best_cv:\n            best_th = th\n            best_cv = cv","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.920164Z","iopub.status.idle":"2022-03-11T16:32:30.92047Z","shell.execute_reply.started":"2022-03-11T16:32:30.920311Z","shell.execute_reply":"2022-03-11T16:32:30.920327Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Best threshold\",best_th)\nprint(\"Best cv\",best_cv)\nval_targets_df.describe()","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.921595Z","iopub.status.idle":"2022-03-11T16:32:30.921893Z","shell.execute_reply.started":"2022-03-11T16:32:30.921737Z","shell.execute_reply":"2022-03-11T16:32:30.921752Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"## Adjustment: Since Public lb has nearly 10% 'new_individual' (Be Careful for private LB)\nval_targets_df['is_new_individual'] = val_targets_df.target=='new_individual'\nprint(val_targets_df.is_new_individual.value_counts().to_dict())\nval_scores = val_targets_df.groupby('is_new_individual').mean().T\nval_scores['adjusted_cv'] = val_scores[True]*0.1+val_scores[False]*0.9\nbest_threshold_adjusted = val_scores['adjusted_cv'].idxmax()\nprint(\"best_threshold\",best_threshold_adjusted)\nval_scores","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.922911Z","iopub.status.idle":"2022-03-11T16:32:30.923226Z","shell.execute_reply.started":"2022-03-11T16:32:30.923064Z","shell.execute_reply":"2022-03-11T16:32:30.92308Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Inference","metadata":{}},{"cell_type":"code","source":"train_embeddings = np.concatenate([train_embeddings,val_embeddings])\ntrain_targets = np.concatenate([train_targets,val_targets])\nprint(train_embeddings.shape,train_targets.shape)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:41:56.954625Z","iopub.execute_input":"2022-03-11T16:41:56.955045Z","iopub.status.idle":"2022-03-11T16:41:57.750015Z","shell.execute_reply.started":"2022-03-11T16:41:56.955004Z","shell.execute_reply":"2022-03-11T16:41:57.749069Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.neighbors import NearestNeighbors\nneigh = NearestNeighbors(n_neighbors=200,metric='cosine')\nneigh.fit(train_embeddings)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T17:14:54.522366Z","iopub.execute_input":"2022-03-11T17:14:54.523077Z","iopub.status.idle":"2022-03-11T17:14:54.98713Z","shell.execute_reply.started":"2022-03-11T17:14:54.523025Z","shell.execute_reply":"2022-03-11T17:14:54.98611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with strategy.scope():    \n    test_ids = []\n    test_nn_distances = []\n    test_nn_idxs = []\n    for filename in tqdm(test_files):\n        embeddings = get_embeddings_np(filename,'test')\n        ids = get_ids(filename)\n        distances,idxs = neigh.kneighbors(embeddings, 200, return_distance=True)\n        test_ids.append(ids)\n        test_nn_idxs.append(idxs)\n        test_nn_distances.append(distances)\n    test_nn_distances = np.concatenate(test_nn_distances)\n    test_nn_idxs = np.concatenate(test_nn_idxs)\n    test_ids = np.concatenate(test_ids)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T17:15:05.476663Z","iopub.execute_input":"2022-03-11T17:15:05.477026Z","iopub.status.idle":"2022-03-11T17:20:48.988726Z","shell.execute_reply.started":"2022-03-11T17:15:05.476993Z","shell.execute_reply":"2022-03-11T17:20:48.987645Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_submission = pd.read_csv('../input/happy-whale-and-dolphin/sample_submission.csv',index_col='image')\nprint(len(test_ids),len(sample_submission))\ntest_df = []\nfor i in tqdm(range(len(test_ids))):\n    id_ = test_ids[i]\n    targets = train_targets[test_nn_idxs[i]]\n    distances = test_nn_distances[i]\n    subset_preds = pd.DataFrame(np.stack([targets,distances],axis=1),columns=['target','distances'])\n    subset_preds['image'] = id_\n    test_df.append(subset_preds)\ntest_df = pd.concat(test_df).reset_index(drop=True)\ntest_df['confidence'] = 1-test_df['distances']\ntest_df = test_df.groupby(['image','target']).confidence.max().reset_index()\ntest_df = test_df.sort_values('confidence',ascending=False).reset_index(drop=True)\ntest_df['target'] = test_df['target'].map(target_encodings)\ntest_df.to_csv('test_neighbors.csv')\ntest_df.image.value_counts().value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-03-11T17:21:36.178146Z","iopub.execute_input":"2022-03-11T17:21:36.178528Z","iopub.status.idle":"2022-03-11T17:22:22.180908Z","shell.execute_reply.started":"2022-03-11T17:21:36.178477Z","shell.execute_reply":"2022-03-11T17:22:22.179977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_list = ['938b7e931166', '5bf17305f073', '7593d2aee842', '7362d7a01d00','956562ff2888']","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:52:53.298328Z","iopub.execute_input":"2022-03-11T16:52:53.298988Z","iopub.status.idle":"2022-03-11T16:52:53.303044Z","shell.execute_reply.started":"2022-03-11T16:52:53.298949Z","shell.execute_reply":"2022-03-11T16:52:53.302212Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = {}\nfor i,row in tqdm(test_df.iterrows()):\n    if row.image in predictions:\n        if len(predictions[row.image])==5:\n            continue\n        predictions[row.image].append(row.target)\n    elif row.confidence>0.515:#best_threshold_adjusted:#best_threshold_adjusted\n        predictions[row.image] = [row.target,'new_individual']\n    else:\n        predictions[row.image] = ['new_individual',row.target]\n        \nfor x in tqdm(predictions):\n    if len(predictions[x])<5:\n        remaining = [y for y in sample_list if y not in predictions]\n        predictions[x] = predictions[x]+remaining\n        predictions[x] = predictions[x][:5]\n    predictions[x] = ' '.join(predictions[x])\n    \npredictions = pd.Series(predictions).reset_index()\npredictions.columns = ['image','predictions']\npredictions.to_csv('submission200.csv',index=False)\npredictions.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-11T17:22:41.731708Z","iopub.execute_input":"2022-03-11T17:22:41.732259Z","iopub.status.idle":"2022-03-11T17:25:15.414535Z","shell.execute_reply.started":"2022-03-11T17:22:41.732225Z","shell.execute_reply":"2022-03-11T17:25:15.413593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_predictions_3(test_df,predictions):\n    for i,row in tqdm(test_df.iterrows()):\n        if row.image in predictions:\n            if len(predictions[row.image])==5:\n                continue\n            predictions[row.image].append(row.target)\n        else:\n            predictions[row.image] = ['new_individual',row.target] \n            \n    for x in tqdm(predictions):\n        if len(predictions[x])<5:\n            remaining = [y for y in sample_list if y not in predictions]\n            predictions[x] = predictions[x]+remaining\n            predictions[x] = predictions[x][:5]\n        predictions[x] = ' '.join(predictions[x])\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.929724Z","iopub.status.idle":"2022-03-11T16:32:30.930294Z","shell.execute_reply.started":"2022-03-11T16:32:30.930106Z","shell.execute_reply":"2022-03-11T16:32:30.930125Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    sample_submission = pd.read_csv('../input/happy-whale-and-dolphin/sample_submission.csv',index_col='image')\n    print(len(test_ids),len(sample_submission))\n    test_df = []\n    for i in tqdm(range(len(test_ids))):\n        id_ = test_ids[i]\n        targets = train_targets[test_nn_idxs[i]]\n        distances = test_nn_distances[i]\n        subset_preds = pd.DataFrame(np.stack([targets,distances],axis=1),columns=['target','distances'])\n        subset_preds['image'] = id_\n        test_df.append(subset_preds)\n    test_df = pd.concat(test_df).reset_index(drop=True)\n    test_df['confidence'] = 1-test_df['distances']\n    test_df.drop('distances',inplace=True, axis=1)\n    test_df['target'] = test_df['target'].map(target_encodings)\n","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.931709Z","iopub.status.idle":"2022-03-11T16:32:30.93219Z","shell.execute_reply.started":"2022-03-11T16:32:30.931935Z","shell.execute_reply":"2022-03-11T16:32:30.931959Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"th=0.55\ntest_df1=test_df.loc[test_df['confidence']>th].reset_index(drop=True)\npredictions=get_predictions_3(test_df1)\ntest_df=test_df.groupby(['image','target']).confidence.max().reset_index()\ntest_df=test_df.sort_values('confidence',ascending=False).reset_index(drop=True)\npredictions=get_predictions_4(test_df,predictions)","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.933839Z","iopub.status.idle":"2022-03-11T16:32:30.93456Z","shell.execute_reply.started":"2022-03-11T16:32:30.934258Z","shell.execute_reply":"2022-03-11T16:32:30.934285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for x in tqdm(predictions):\n    predictions[x] = ' '.join(predictions[x])\n    \npredictions = pd.Series(predictions).reset_index()\npredictions.columns = ['image','predictions']\npredictions.to_csv('submission_gld3_55.csv',index=False)\npredictions.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.936105Z","iopub.status.idle":"2022-03-11T16:32:30.936601Z","shell.execute_reply.started":"2022-03-11T16:32:30.936329Z","shell.execute_reply":"2022-03-11T16:32:30.936356Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_predictions_5(test_df,predictions):\n    for i,row in tqdm(test_df.iterrows()):\n        if row.image in predictions:\n            if len(predictions[row.image])==5:\n                continue\n            if row.target in predictions[row.image]:\n                continue\n            predictions[row.image].append(row.target)\n        else:\n            predictions[row.image] = ['new_individual',row.target] \n            \n    for x in tqdm(predictions):\n        if len(predictions[x])<5:\n            remaining = [y for y in sample_list if y not in predictions]\n            predictions[x] = predictions[x]+remaining\n            predictions[x] = predictions[x][:5]\n    return predictions","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.937863Z","iopub.status.idle":"2022-03-11T16:32:30.938286Z","shell.execute_reply.started":"2022-03-11T16:32:30.938081Z","shell.execute_reply":"2022-03-11T16:32:30.938104Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"th=0.5\ntest_df1=test_df.loc[test_df['confidence']>th].reset_index(drop=True)\ntest_df1=test_df1.groupby(['image','target']).confidence.size().reset_index()\ntest_df1 = test_df1.sort_values('confidence',ascending=False).reset_index(drop=True)\nprediction=get_predictions_1(test_df1)\nfor x in tqdm(prediction):\n    if len(prediction[x])<5:\n        prediction[x] = prediction[x]+['new_individual']\ntest_df2=test_df.loc[test_df['confidence']<=th].reset_index(drop=True)\ntest_df2 = test_df2.groupby(['image','target']).confidence.max().reset_index()\ntest_df2 = test_df2.sort_values('confidence',ascending=False).reset_index(drop=True)\npredictions=get_predictions_3(test_df2,prediction)\npredictions = pd.Series(predictions).reset_index()\npredictions.columns = ['image','predictions']\npredictions.to_csv('submission1.csv',index=False)\npredictions.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.93937Z","iopub.status.idle":"2022-03-11T16:32:30.939736Z","shell.execute_reply.started":"2022-03-11T16:32:30.939549Z","shell.execute_reply":"2022-03-11T16:32:30.939566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"    test_df = test_df.groupby(['image','target']).confidence.max().reset_index()\n    test_df = test_df.sort_values('confidence',ascending=False).reset_index(drop=True)\n    test_df['target'] = test_df['target'].map(target_encodings)\n    test_df.to_csv('test_neighbors.csv')\n    test_df.image.value_counts().value_counts()","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.940856Z","iopub.status.idle":"2022-03-11T16:32:30.941209Z","shell.execute_reply.started":"2022-03-11T16:32:30.941015Z","shell.execute_reply":"2022-03-11T16:32:30.941037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_list = ['new_individual', '5bf17305f073', '7593d2aee842', '7362d7a01d00','956562ff2888']","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.942915Z","iopub.status.idle":"2022-03-11T16:32:30.94325Z","shell.execute_reply.started":"2022-03-11T16:32:30.943075Z","shell.execute_reply":"2022-03-11T16:32:30.943097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = {}\nfor i,row in tqdm(test_df.iterrows()):\n    if row.image in predictions:\n        if len(predictions[row.image])==5:\n            continue\n        predictions[row.image].append(row.target)\n    elif row.confidence>0.7:#best_threshold_adjusted\n        predictions[row.image] = [row.target]\n    else:\n        predictions[row.image] = ['new_individual',row.target]\n        \nfor x in tqdm(predictions):\n    if len(predictions[x])<5:\n        remaining = [y for y in sample_list if y not in predictions]\n        predictions[x] = predictions[x]+remaining\n        predictions[x] = predictions[x][:5]\n    predictions[x] = ' '.join(predictions[x])\n    \npredictions = pd.Series(predictions).reset_index()\npredictions.columns = ['image','predictions']\npredictions.to_csv('submission300.csv',index=False)\npredictions.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-11T16:32:30.94452Z","iopub.status.idle":"2022-03-11T16:32:30.945006Z","shell.execute_reply.started":"2022-03-11T16:32:30.94474Z","shell.execute_reply":"2022-03-11T16:32:30.944765Z"},"trusted":true},"execution_count":null,"outputs":[]}]}