{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip uninstall -y transformers\n!pip install git+https://github.com/alaradirik/transformers.git@owlvit-tf\n\nfrom zipfile import ZipFile\nimport transformers\nimport tensorflow as tf\nimport tensorflow_addons as tfa\nfrom PIL import Image\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport numpy as np # linear algebra\nimport requests\nimport os","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-08-13T23:32:31.218517Z","iopub.execute_input":"2022-08-13T23:32:31.219755Z","iopub.status.idle":"2022-08-13T23:33:20.014953Z","shell.execute_reply.started":"2022-08-13T23:32:31.219709Z","shell.execute_reply":"2022-08-13T23:33:20.013958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# [x for x in dir(transformers) if x.lower().startswith(\"tf\")]\nfrom transformers import TFOwlViTVisionModel, OwlViTProcessor","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:33:20.017040Z","iopub.execute_input":"2022-08-13T23:33:20.017569Z","iopub.status.idle":"2022-08-13T23:33:22.121215Z","shell.execute_reply.started":"2022-08-13T23:33:20.017539Z","shell.execute_reply":"2022-08-13T23:33:22.120246Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model_style = \"owlvit-large-patch14\" # [\"owlvit-base-patch16|\"owlvit-base-patch32\"|\"owlvit-large-patch14\"]\n\nprocessor = OwlViTProcessor.from_pretrained(f\"google/{model_style}\")\nvision_model = TFOwlViTVisionModel.from_pretrained(f\"google/{model_style}\", from_pt=True)\n\nurl = \"http://images.cocodataset.org/val2017/000000039769.jpg\"\nimage = Image.open(requests.get(url, stream=True).raw)\nimage","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:33:22.122712Z","iopub.execute_input":"2022-08-13T23:33:22.123078Z","iopub.status.idle":"2022-08-13T23:35:14.539354Z","shell.execute_reply.started":"2022-08-13T23:33:22.123038Z","shell.execute_reply":"2022-08-13T23:35:14.538406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def hf_preprocess_symbolic_input(x, data_format=\"channels_last\", mode=\"torch\"):\n  \"\"\"Preprocesses a tensor encoding a batch of images.\n  Args:\n    x: Input tensor, 3D or 4D.\n    data_format: Data format of the image tensor.\n    mode: One of \"caffe\", \"tf\" or \"torch\".\n      - caffe: will convert the images from RGB to BGR,\n          then will zero-center each color channel with\n          respect to the ImageNet dataset,\n          without scaling.\n      - tf: will scale pixels between -1 and 1,\n          sample-wise.\n      - torch: will scale pixels between 0 and 1 and then\n          will normalize each channel with respect to the\n          ImageNet dataset.\n  Returns:\n      Preprocessed tensor.\n  \"\"\"\n  if mode == 'tf':\n    x /= 127.5\n    x -= 1.\n    return x\n  elif mode == 'torch':\n    x /= 255.\n    mean = [0.48145466, 0.4578275, 0.40821073]\n    std = [0.26862954, 0.26130258, 0.27577711]\n  else:\n    if data_format == 'channels_first':\n      # 'RGB'->'BGR'\n      if tf.keras.backend.ndim(x) == 3:\n        x = x[::-1, ...]\n      else:\n        x = x[:, ::-1, ...]\n    else:\n      # 'RGB'->'BGR'\n      x = x[..., ::-1]\n    mean = [103.939, 116.779, 123.68]\n    std = None\n\n  mean_tensor = tf.keras.backend.constant(-np.array(mean))\n\n  # Zero-center by mean pixel\n  if tf.keras.backend.dtype(x) != tf.keras.backend.dtype(mean_tensor):\n    x = tf.keras.backend.bias_add(\n        x, tf.keras.backend.cast(mean_tensor, tf.keras.backend.dtype(x)), data_format=data_format)\n  else:\n    x = tf.keras.backend.bias_add(x, mean_tensor, data_format)\n  if std is not None:\n    std_tensor = tf.keras.backend.constant(np.array(std), dtype=tf.keras.backend.dtype(x))\n    if data_format == 'channels_first':\n      std_tensor = tf.keras.backend.reshape(std_tensor, (-1, 1, 1))\n    x /= std_tensor\n  return x","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:39:05.758531Z","iopub.execute_input":"2022-08-13T23:39:05.758889Z","iopub.status.idle":"2022-08-13T23:39:05.770571Z","shell.execute_reply.started":"2022-08-13T23:39:05.758858Z","shell.execute_reply":"2022-08-13T23:39:05.769564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def build_guie_model(_model, _h=768, _w=768, do_h_flip=False):\n    \n    # Inputs\n    _inputs = tf.keras.layers.Input(shape=(None,None,3), batch_size=1, dtype=tf.uint8, name=\"inputs\")\n    x = _inputs    \n    \n    # Resize, normalize pixel values & move channels-last to channels-first\n    x = tf.keras.layers.Resizing(_h, _w, crop_to_aspect_ratio=True, interpolation=\"bicubic\")(tf.cast(x, tf.float32))\n    x = hf_preprocess_symbolic_input(x)\n\n    # 1. Horizontal Flip\n    if do_h_flip:\n        x2 = tf.keras.layers.RandomFlip(mode=\"horizontal\")(x, training=True)\n        x2 = tf.keras.layers.Permute((3,1,2))(x2) #bhwc --> bchw\n        x2 = _model({'pixel_values':x2}).pooler_output\n    \n    x = tf.keras.layers.Permute((3,1,2))(x) #bhwc --> bchw\n    \n    # Inference\n    x = _model({'pixel_values':x}).pooler_output\n    \n    if do_h_flip: x = (x+x2)/2\n    \n    x = tf.keras.layers.Lambda(lambda _x: tf.expand_dims(_x, axis=1))(x)\n    x = tf.keras.layers.GlobalAveragePooling2D()(x)\n    \n    # Adaptive Pooling\n    # Embedding and embedding norm\n    _output_1 = tfa.layers.AdaptiveAveragePooling1D(output_size=64, name=\"embedding\")(x)\n    _output_2 = tf.keras.layers.Lambda(lambda x: tf.nn.l2_normalize(x), name=\"embedding_norm\")(_output_1)    \n    \n    # Return model\n    return tf.keras.Model(inputs=_inputs, outputs=[_output_1, _output_2])\n\nguie_model = build_guie_model(vision_model, \n                              _h=840 if \"large\" in model_style else 768, \n                              _w=840 if \"large\" in model_style else 768)\nguie_model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:45:01.401665Z","iopub.execute_input":"2022-08-13T23:45:01.402256Z","iopub.status.idle":"2022-08-13T23:45:04.302222Z","shell.execute_reply.started":"2022-08-13T23:45:01.402208Z","shell.execute_reply":"2022-08-13T23:45:04.301198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def make_submission_zip(model_dir, output_dir=\".\"):\n    with ZipFile(os.path.join(output_dir, 'submission.zip'),'w') as zip:           \n        zip.write(os.path.join(model_dir, 'saved_model.pb'), arcname='saved_model.pb') \n        zip.write(os.path.join(model_dir, 'variables', 'variables.data-00000-of-00001'), arcname='variables/variables.data-00000-of-00001') \n        zip.write(os.path.join(model_dir, 'variables', 'variables.index'), arcname='variables/variables.index') \n\n# Save fresh model to directory\n!rm -rf ./models\nos.makedirs(\"./models\", exist_ok=True)\nguie_model.save(\"./models\")\n\n# Show unzipped contents\nprint(os.listdir(\"./models\"))\n\nmake_submission_zip(\"./models\", output_dir=\".\")","metadata":{"execution":{"iopub.status.busy":"2022-08-13T23:01:26.547474Z","iopub.execute_input":"2022-08-13T23:01:26.547920Z","iopub.status.idle":"2022-08-13T23:01:56.599935Z","shell.execute_reply.started":"2022-08-13T23:01:26.547877Z","shell.execute_reply":"2022-08-13T23:01:56.598807Z"},"trusted":true},"execution_count":null,"outputs":[]}]}