{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Note \n    1 .https://www.kaggle.com/code/ghrangel/download-data-from-drive-to-kaggle-gdown/notebook?kernelSessionId=82151810\n    2 .https://www.kaggle.com/code/rsinda/downloading-data-from-google-drive/notebook","metadata":{}},{"cell_type":"code","source":"%cd ..","metadata":{"execution":{"iopub.status.busy":"2022-07-25T09:59:56.147602Z","iopub.execute_input":"2022-07-25T09:59:56.147904Z","iopub.status.idle":"2022-07-25T09:59:56.153634Z","shell.execute_reply.started":"2022-07-25T09:59:56.147873Z","shell.execute_reply":"2022-07-25T09:59:56.152899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf ./Code4AI","metadata":{"execution":{"iopub.status.busy":"2022-07-25T10:00:11.476245Z","iopub.execute_input":"2022-07-25T10:00:11.477036Z","iopub.status.idle":"2022-07-25T10:00:12.250861Z","shell.execute_reply.started":"2022-07-25T10:00:11.476991Z","shell.execute_reply":"2022-07-25T10:00:12.249812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls .","metadata":{"execution":{"iopub.status.busy":"2022-07-25T10:00:25.723243Z","iopub.execute_input":"2022-07-25T10:00:25.723570Z","iopub.status.idle":"2022-07-25T10:00:26.494664Z","shell.execute_reply.started":"2022-07-25T10:00:25.723533Z","shell.execute_reply":"2022-07-25T10:00:26.493566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls -lf /root/","metadata":{"execution":{"iopub.status.busy":"2022-07-25T10:00:28.698850Z","iopub.execute_input":"2022-07-25T10:00:28.699322Z","iopub.status.idle":"2022-07-25T10:00:29.474035Z","shell.execute_reply.started":"2022-07-25T10:00:28.699284Z","shell.execute_reply":"2022-07-25T10:00:29.473043Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!conda install -y gdown","metadata":{"execution":{"iopub.status.busy":"2022-07-25T09:49:34.239758Z","iopub.execute_input":"2022-07-25T09:49:34.240140Z","iopub.status.idle":"2022-07-25T09:50:39.733296Z","shell.execute_reply.started":"2022-07-25T09:49:34.240073Z","shell.execute_reply":"2022-07-25T09:50:39.732261Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Setting up SSH ==> Clone project \n\nimport gdown\n\n# url = 'https://drive.google.com/file/d/1tA4a2I8e7IyhOU-RMOn6cqJYX6qfj55C/view?usp=sharing'\noutput = 'SSH_key.zip'\ngdown.download('https://drive.google.com/uc?export=download&id=1tA4a2I8e7IyhOU-RMOn6cqJYX6qfj55C', output, quiet=False)","metadata":{"execution":{"iopub.status.busy":"2022-07-25T09:50:39.737410Z","iopub.execute_input":"2022-07-25T09:50:39.737695Z","iopub.status.idle":"2022-07-25T09:50:41.983137Z","shell.execute_reply.started":"2022-07-25T09:50:39.737662Z","shell.execute_reply":"2022-07-25T09:50:41.982443Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!unzip SSH_key.zip","metadata":{"execution":{"iopub.status.busy":"2022-07-25T09:50:41.984469Z","iopub.execute_input":"2022-07-25T09:50:41.984924Z","iopub.status.idle":"2022-07-25T09:50:42.761619Z","shell.execute_reply.started":"2022-07-25T09:50:41.984894Z","shell.execute_reply":"2022-07-25T09:50:42.760589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir /root/.ssh/\n!cp -f ./ssh_folder/* ~/.ssh/\n\n!ls -lf /root/.ssh","metadata":{"execution":{"iopub.status.busy":"2022-07-25T09:50:42.765737Z","iopub.execute_input":"2022-07-25T09:50:42.766028Z","iopub.status.idle":"2022-07-25T09:50:45.043203Z","shell.execute_reply.started":"2022-07-25T09:50:42.765993Z","shell.execute_reply":"2022-07-25T09:50:45.042309Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!chmod 600 /root/.ssh/id_rsa\n!ssh-keyscan -t rsa github.com >> /root/.ssh/known_hosts","metadata":{"execution":{"iopub.status.busy":"2022-07-25T09:50:45.046707Z","iopub.execute_input":"2022-07-25T09:50:45.047042Z","iopub.status.idle":"2022-07-25T09:50:47.168021Z","shell.execute_reply.started":"2022-07-25T09:50:45.047006Z","shell.execute_reply":"2022-07-25T09:50:47.166836Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf ssh_folder SSH_key.zip ","metadata":{"execution":{"iopub.status.busy":"2022-07-25T09:50:47.170093Z","iopub.execute_input":"2022-07-25T09:50:47.170404Z","iopub.status.idle":"2022-07-25T09:50:47.927271Z","shell.execute_reply.started":"2022-07-25T09:50:47.170369Z","shell.execute_reply":"2022-07-25T09:50:47.926052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone git@github.com:KaggleTeamProjects/Code4AI.git\n%cd ./Code4AI\n!mkdir ./checkpoint","metadata":{"execution":{"iopub.status.busy":"2022-07-25T10:00:37.157920Z","iopub.execute_input":"2022-07-25T10:00:37.158262Z","iopub.status.idle":"2022-07-25T10:00:40.329248Z","shell.execute_reply.started":"2022-07-25T10:00:37.158223Z","shell.execute_reply":"2022-07-25T10:00:40.328041Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %load ./src/training/trainer.py\nimport transformers\nimport numpy as np\nimport tensorflow as tf\nimport os\nfrom typing import List\nfrom sklearn.model_selection import KFold\nfrom src.training.utlis import count_samples\nfrom src.tensorflow.TFRecord.tfrecord import TFRecordData\nfrom src.training.utlis import WarmupLinearDecay\nfrom src.ranking_model.codebert_model import get_model, get_model_v2\n\n\nclass Trainer(object):\n\n    def __init__(self, tpu_status, strategy, epochs, batchsize,\n                 dataset_dir, checkpoint_dir,\n                 lr, warmup_rate=0.05):\n        self.n_splits = 5  # fold\n        self.tpu_status = tpu_status\n        self.dataset_dir = dataset_dir\n        self.checkpoint_dir = checkpoint_dir\n        self.batchsize = batchsize\n        self.epochs = epochs\n        self.strategy = strategy\n        self.WARMUP_RATE = warmup_rate\n        self.VERBOSE = 1 if os.environ[\"KAGGLE_KERNEL_RUN_TYPE\"] == \"Interactive\" else 2\n\n        # hyper parameter\n        self.lr = lr\n\n        # setup callbacks\n        self.callbacks = []\n        earlystop = tf.keras.callbacks.EarlyStopping(monitor='val_loss', patience=1)\n        self.callbacks.append(earlystop)\n\n    def _setup_optimizer(self, steps_per_epoch):\n        total_steps = steps_per_epoch * self.epochs\n        warmup_steps = int(self.WARMUP_RATE * total_steps)\n\n        optimizer = transformers.AdamWeightDecay(\n            learning_rate=WarmupLinearDecay(\n                base_learning_rate=self.lr,\n                warmup_steps=warmup_steps,\n                total_steps=total_steps,\n            ),\n            weight_decay_rate=0.01,\n            exclude_from_weight_decay=[\n                \"bias\",\n                \"LayerNorm.bias\",\n                \"LayerNorm.weight\",\n            ],\n        )\n        return optimizer\n\n    def _train(self, train_dataset, val_dataset,\n               steps_per_epoch, validation_steps, list_callbacks, fold):\n        with self.strategy.scope():\n            model = get_model_v2()\n            optimizer = self._setup_optimizer(steps_per_epoch)\n            model.compile(loss=\"mae\", optimizer=optimizer)\n\n        model.fit(\n            train_dataset,\n            steps_per_epoch=steps_per_epoch,\n            validation_data=val_dataset,\n            validation_steps=validation_steps,\n            callbacks=list_callbacks,\n            epochs=self.epochs,\n            verbose=self.VERBOSE\n        )\n\n        folder_fold = self.checkpoint_dir / f\"{fold}\"\n        if not folder_fold.exists():\n            os.makedirs(folder_fold)\n        model.save_weights(folder_fold / f\"model_{fold}.h5\")\n\n    def training_fold_k1(self):\n        for fold in range(1):\n            print(f\">> Dataset: {self.dataset_dir}\")\n            if self.tpu_status is not None:\n                tf.tpu.experimental.initialize_tpu_system(self.tpu_status)\n\n            train_filenames = tf.io.gfile.glob(os.path.join(self.dataset_dir, \"tfrec\", str(fold), \"train\", \"*.tfrec\"))\n            steps_per_epoch = count_samples(train_filenames) // self.batchsize\n            train_dataset = TFRecordData.load(filenames=train_filenames,\n                                              batch_size=self.batchsize,\n                                              strategy=self.strategy)\n\n            val_filenames = tf.io.gfile.glob(os.path.join(self.dataset_dir, \"tfrec\", str(fold), \"validate\", \"*.tfrec\"))\n            validate_steps = count_samples(val_filenames) // self.batchsize\n            validate_dataset = TFRecordData.load(filenames=val_filenames,\n                                                 batch_size=self.batchsize,\n                                                 strategy=self.strategy,\n                                                 ordered=True, repeated=False, cached=True)\n\n            self._train(train_dataset, validate_dataset, steps_per_epoch, validate_steps, self.callbacks, fold=fold)\n\n    def training(self):\n        for i, (train_index, val_index) in enumerate(KFold(n_splits=self.n_splits).split(range(self.n_splits))):\n            print(f\">> Dataset: {self.dataset_dir}\")\n            if self.tpu_status is not None:\n                tf.tpu.experimental.initialize_tpu_system(self.tpu_status)\n\n            # callbacks\n            checkpoint_filepath = self.checkpoint_dir / f'{i}'\n            if not checkpoint_filepath.exists():\n                os.makedirs(checkpoint_filepath)\n            model_checkpoint_callback = tf.keras.callbacks.ModelCheckpoint(\n                filepath=str(checkpoint_filepath),\n                save_weights_only=True,\n                monitor='val_loss',\n                mode='max',\n                save_best_only=True)\n\n            list_callbacks = self.callbacks + [model_checkpoint_callback]\n            train_filenames = np.ravel(\n                [tf.io.gfile.glob(os.path.join(self.dataset_dir, \"tfrec\", str(idx), \"*.tfrec\")) for idx in train_index])\n            steps_per_epoch = count_samples(train_filenames) // self.batchsize\n            train_dataset = TFRecordData.load(filenames=train_filenames, batch_size=self.batchsize,\n                                              strategy=self.strategy)\n            val_filenames = np.ravel(\n                [tf.io.gfile.glob(os.path.join(self.dataset_dir, \"tfrec\", str(idx), \"*.tfrec\")) for idx in val_index])\n            validate_steps = count_samples(val_filenames) // self.batchsize\n            val_dataset = TFRecordData.load(filenames=val_filenames, batch_size=self.batchsize, strategy=self.strategy,\n                                            ordered=True, repeated=False, cached=True)\n            self._train(train_dataset, val_dataset, steps_per_epoch, validate_steps, list_callbacks, fold=i)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T10:01:09.571675Z","iopub.execute_input":"2022-07-25T10:01:09.572182Z","iopub.status.idle":"2022-07-25T10:01:10.974292Z","shell.execute_reply.started":"2022-07-25T10:01:09.572099Z","shell.execute_reply":"2022-07-25T10:01:10.973429Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# %load ./src/main_training_tpu_debug.py\nimport os\nimport sys\n\n__dir__ = os.path.dirname(os.path.abspath(''))\nsys.path.append(__dir__)\nsys.path.insert(0, os.path.abspath(os.path.join(__dir__, '..')))\n\nimport tensorflow as tf\nfrom settings import config_training as cfg\nfrom kaggle_datasets import KaggleDatasets\n\n\ndef show_information():\n    try:\n        TPU = tf.distribute.cluster_resolver.TPUClusterResolver()\n        tf.config.experimental_connect_to_cluster(TPU)\n        tf.tpu.experimental.initialize_tpu_system(TPU)\n        STRATEGY = tf.distribute.experimental.TPUStrategy(TPU)\n        BATCH_SIZE = 64 * STRATEGY.num_replicas_in_sync\n    except Exception:\n        TPU = None\n        STRATEGY = tf.distribute.get_strategy()\n        BATCH_SIZE = 1\n\n    print(f\"TPU: {TPU} - STRATEGY: {STRATEGY} - BATCH_SIZE: {BATCH_SIZE}.\")\n    print(\"TensorFlow \", tf.__version__)\n\n    if TPU is not None:\n        print(\"Using TPU v3-8 ...\")\n    else:\n        print(\"Using GPU/CPU ...\")\n\n    return TPU, STRATEGY, BATCH_SIZE\n\n\nif __name__ == '__main__':\n    TPU, STRATEGY, BATCH_SIZE = show_information()\n    \n    cfg.COMPRESS_DATASET_DIR = KaggleDatasets().get_gcs_path(\"bertuncased-2model-v1\")\n    \n    trainer = Trainer(\n        tpu_status=TPU,\n        strategy=STRATEGY,\n        epochs=5,\n        batchsize=BATCH_SIZE,\n        dataset_dir=cfg.COMPRESS_DATASET_DIR,\n        checkpoint_dir=cfg.CHECKPOINT_DIR,\n        lr=3e-5\n    )\n\n    trainer.training_fold_k1()\n","metadata":{"execution":{"iopub.status.busy":"2022-07-25T10:01:13.931187Z","iopub.execute_input":"2022-07-25T10:01:13.931517Z","iopub.status.idle":"2022-07-25T10:04:17.673335Z","shell.execute_reply.started":"2022-07-25T10:01:13.931474Z","shell.execute_reply":"2022-07-25T10:04:17.671759Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#######################################################################","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2022-07-05T13:48:52.472249Z","iopub.execute_input":"2022-07-05T13:48:52.472455Z","iopub.status.idle":"2022-07-05T13:48:58.558641Z","shell.execute_reply.started":"2022-07-05T13:48:52.47243Z","shell.execute_reply":"2022-07-05T13:48:58.557629Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}