{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import tensorflow as tf","metadata":{"execution":{"iopub.status.busy":"2021-07-13T07:55:30.852242Z","iopub.execute_input":"2021-07-13T07:55:30.852662Z","iopub.status.idle":"2021-07-13T07:55:37.824620Z","shell.execute_reply.started":"2021-07-13T07:55:30.852577Z","shell.execute_reply":"2021-07-13T07:55:37.823437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ../input","metadata":{"execution":{"iopub.status.busy":"2021-07-13T07:55:37.826266Z","iopub.execute_input":"2021-07-13T07:55:37.826681Z","iopub.status.idle":"2021-07-13T07:55:38.576005Z","shell.execute_reply.started":"2021-07-13T07:55:37.826644Z","shell.execute_reply":"2021-07-13T07:55:38.574835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Detect hardware, return appropriate distribution strategy\ntry:\n    # TPU detection. No parameters necessary if TPU_NAME environment variable is set. On Kaggle this is always the case.\n    TPU = tf.distribute.cluster_resolver.TPUClusterResolver()  \nexcept ValueError:\n    TPU = None\n\nif TPU:\n    print(f\"\\n... RUNNING ON TPU - {TPU.master()}...\")\n    tf.config.experimental_connect_to_cluster(TPU)\n    tf.tpu.experimental.initialize_tpu_system(TPU)\n    strategy = tf.distribute.experimental.TPUStrategy(TPU)\nelse:\n    print(f\"\\n... RUNNING ON CPU/GPU ...\")\n    # Yield the default distribution strategy in Tensorflow\n    #   --> Works on CPU and single GPU.\n    strategy = tf.distribute.get_strategy() \n\n# What Is a Replica?\n#    --> A single Cloud TPU device consists of FOUR chips, each of which has TWO TPU cores. \n#    --> Therefore, for efficient utilization of Cloud TPU, a program should make use of each of the EIGHT (4x2) cores. \n#    --> Each replica is essentially a copy of the training graph that is run on each core and \n#        trains a mini-batch containing 1/8th of the overall batch size\nN_REPLICAS = strategy.num_replicas_in_sync\n    \nprint(f\"... # OF REPLICAS: {N_REPLICAS} ...\\n\")\n\nprint(f\"\\n... ACCELERATOR SETUP COMPLTED ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2021-07-13T07:55:38.578811Z","iopub.execute_input":"2021-07-13T07:55:38.579284Z","iopub.status.idle":"2021-07-13T07:55:44.427077Z","shell.execute_reply.started":"2021-07-13T07:55:38.579232Z","shell.execute_reply":"2021-07-13T07:55:44.425747Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"\\n... XLA OPTIMIZATIONS STARTING ...\\n\")\n\nprint(f\"\\n... CONFIGURE JIT (JUST IN TIME) COMPILATION ...\\n\")\n# enable XLA optmizations (10% speedup when using @tf.function calls)\ntf.config.optimizer.set_jit(True)\n\nprint(f\"\\n... XLA OPTIMIZATIONS COMPLETED ...\\n\")","metadata":{"execution":{"iopub.status.busy":"2021-07-13T07:55:44.428693Z","iopub.execute_input":"2021-07-13T07:55:44.429015Z","iopub.status.idle":"2021-07-13T07:55:44.437432Z","shell.execute_reply.started":"2021-07-13T07:55:44.428985Z","shell.execute_reply":"2021-07-13T07:55:44.435964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_secrets import UserSecretsClient\nuser_secrets = UserSecretsClient()\nuser_credential = user_secrets.get_gcloud_credential()\n\n# Step 2: Set the credentials\nuser_secrets.set_tensorflow_credential(user_credential)","metadata":{"execution":{"iopub.status.busy":"2021-07-13T07:55:44.439216Z","iopub.execute_input":"2021-07-13T07:55:44.439562Z","iopub.status.idle":"2021-07-13T07:55:44.763322Z","shell.execute_reply.started":"2021-07-13T07:55:44.439530Z","shell.execute_reply":"2021-07-13T07:55:44.761917Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from kaggle_datasets import KaggleDatasets\nDATA_DIR = KaggleDatasets().get_gcs_path(\"siim-cocolike-tfrecords\")\nDATA_DIR","metadata":{"execution":{"iopub.status.busy":"2021-07-13T07:55:44.765253Z","iopub.execute_input":"2021-07-13T07:55:44.765727Z","iopub.status.idle":"2021-07-13T07:56:36.686152Z","shell.execute_reply.started":"2021-07-13T07:55:44.765676Z","shell.execute_reply":"2021-07-13T07:56:36.684892Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"MODEL_DIR = KaggleDatasets().get_gcs_path(\"retinanet-weights\")\nMODEL_DIR","metadata":{"execution":{"iopub.status.busy":"2021-07-13T07:56:36.687474Z","iopub.execute_input":"2021-07-13T07:56:36.687785Z","iopub.status.idle":"2021-07-13T07:57:05.867691Z","shell.execute_reply.started":"2021-07-13T07:56:36.687742Z","shell.execute_reply":"2021-07-13T07:57:05.866593Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone https://github.com/tensorflow/tpu/","metadata":{"execution":{"iopub.status.busy":"2021-07-13T07:57:05.870068Z","iopub.execute_input":"2021-07-13T07:57:05.870428Z","iopub.status.idle":"2021-07-13T07:57:09.895800Z","shell.execute_reply.started":"2021-07-13T07:57:05.870394Z","shell.execute_reply":"2021-07-13T07:57:09.894808Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile my_retinanet.yaml\ntype: 'retinanet'\narchitecture:\n    num_classes: 2\ntrain:\n    learning_rate:\n        learning_rate: 0.005\n        lr_warmup_init: 0.0005\n    train_file_pattern: gs://kds-ebf9447eebe1e7ebd75c826c574b61e41e4b1e7fad139ffad4b2cb72/fold_0/train*.tfrecord\n    checkpoint:\n        path: gs://kds-409fb23ff0cce62f1f0f77db5d582618ece99d40ba3fd8e5226f947f/detection_retinanet_spinenet-96-best/model.ckpt\n        prefix: spinenet96/\neval:\n    eval_file_pattern: gs://kds-ebf9447eebe1e7ebd75c826c574b61e41e4b1e7fad139ffad4b2cb72/fold_0/eval*.tfrecord\n    num_steps_per_eval: 250\n    use_json_file: False","metadata":{"execution":{"iopub.status.busy":"2021-07-13T07:58:18.018299Z","iopub.execute_input":"2021-07-13T07:58:18.018880Z","iopub.status.idle":"2021-07-13T07:58:18.025116Z","shell.execute_reply.started":"2021-07-13T07:58:18.018821Z","shell.execute_reply":"2021-07-13T07:58:18.024061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install ./tpu/models/","metadata":{"execution":{"iopub.status.busy":"2021-07-13T07:57:09.907893Z","iopub.execute_input":"2021-07-13T07:57:09.908327Z","iopub.status.idle":"2021-07-13T07:57:21.162177Z","shell.execute_reply.started":"2021-07-13T07:57:09.908268Z","shell.execute_reply":"2021-07-13T07:57:21.161109Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install --user 'git+https://github.com/cocodataset/cocoapi#egg=pycocotools&subdirectory=PythonAPI'","metadata":{"execution":{"iopub.status.busy":"2021-07-13T07:57:21.164166Z","iopub.execute_input":"2021-07-13T07:57:21.164647Z","iopub.status.idle":"2021-07-13T07:57:38.838145Z","shell.execute_reply.started":"2021-07-13T07:57:21.164585Z","shell.execute_reply":"2021-07-13T07:57:38.836943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python /kaggle/working/tpu/models/official/detection/main.py \\\n  --model=\"retinanet\" \\\n  --model_dir=\"gs://effdet_siim_output/retinanet/\" \\\n  --use_tpu=True \\\n  --tpu=\"grpc://10.0.0.2:8470\" \\\n  --num_cores=8 \\\n  --mode=train \\\n  --config_file=\"my_retinanet.yaml\" \\\n  --params_override=\"\"","metadata":{"execution":{"iopub.status.busy":"2021-07-13T07:58:56.307001Z","iopub.execute_input":"2021-07-13T07:58:56.307427Z","iopub.status.idle":"2021-07-13T08:06:50.577513Z","shell.execute_reply.started":"2021-07-13T07:58:56.307389Z","shell.execute_reply":"2021-07-13T08:06:50.576081Z"},"trusted":true},"execution_count":null,"outputs":[]}]}