{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip uninstall typing -y\n!pip install neptune-client \n!pip install audiomentations==0.11.0","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-05-30T18:27:47.199998Z","iopub.execute_input":"2021-05-30T18:27:47.200565Z","iopub.status.idle":"2021-05-30T18:28:25.236844Z","shell.execute_reply.started":"2021-05-30T18:27:47.200506Z","shell.execute_reply":"2021-05-30T18:28:25.235975Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf kaggle-birdclef-2021\n!git clone https://github.com/fraank/kaggle-birdclef-2021.git","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2021-05-30T18:28:25.238630Z","iopub.execute_input":"2021-05-30T18:28:25.238969Z","iopub.status.idle":"2021-05-30T18:28:27.821399Z","shell.execute_reply.started":"2021-05-30T18:28:25.238940Z","shell.execute_reply":"2021-05-30T18:28:27.820592Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%load_ext autoreload\n%autoreload 2","metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","execution":{"iopub.status.busy":"2021-05-30T18:28:27.822770Z","iopub.execute_input":"2021-05-30T18:28:27.823107Z","iopub.status.idle":"2021-05-30T18:28:27.850291Z","shell.execute_reply.started":"2021-05-30T18:28:27.823072Z","shell.execute_reply":"2021-05-30T18:28:27.849328Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%cd kaggle-birdclef-2021/src","metadata":{"execution":{"iopub.status.busy":"2021-05-30T18:28:27.851787Z","iopub.execute_input":"2021-05-30T18:28:27.852410Z","iopub.status.idle":"2021-05-30T18:28:27.876337Z","shell.execute_reply.started":"2021-05-30T18:28:27.852369Z","shell.execute_reply":"2021-05-30T18:28:27.875405Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Environment Variables\n\nSet these environment variables if you plan on adding slack notifications at the end of training loops or plan on using neptune for logging\n","metadata":{}},{"cell_type":"code","source":"# %env SLACK_URL=\"\"\n# %env NEPTUNE_API_TOKEN=\"\"","metadata":{"execution":{"iopub.status.busy":"2021-05-29T18:30:15.031811Z","iopub.execute_input":"2021-05-29T18:30:15.03235Z","iopub.status.idle":"2021-05-29T18:30:15.051914Z","shell.execute_reply.started":"2021-05-29T18:30:15.032312Z","shell.execute_reply":"2021-05-29T18:30:15.051291Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nfrom sklearn.model_selection import train_test_split\nimport pandas as pd\nimport numpy as np\n\nTRAIN = pd.read_csv('../../../input/birdclef-2021/train_metadata.csv',)\nprint('Complete Train Data: %s' % (len(TRAIN)))\n\nTRAIN = TRAIN.query('rating>=4')\ntrain, test = train_test_split(TRAIN, test_size=0.2, stratify=TRAIN['primary_label'])\n\n# print(TRAIN.groupby('primary_label').size())\n# print(train.groupby('primary_label').size())\n# print(test.groupby('primary_label').size())\n\nprint('Train Data with rating >4: %s' % (len(train)))\nprint('Test Data with rating >4: %s' % (len(test)))\n\n# save CSVs\nif not os.path.exists('../../csv'):\n    os.mkdir('../../csv')\ntrain.to_csv('../../csv/train.csv')\ntest.to_csv('../../csv/test.csv')\n\n# TODO accept ogg\n# TODO ","metadata":{"execution":{"iopub.status.busy":"2021-05-30T18:28:27.877762Z","iopub.execute_input":"2021-05-30T18:28:27.878206Z","iopub.status.idle":"2021-05-30T18:28:30.188754Z","shell.execute_reply.started":"2021-05-30T18:28:27.878167Z","shell.execute_reply":"2021-05-30T18:28:30.187950Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate Labels for Config\nLABELS = {}\ni = 0\nfor item in sorted(TRAIN.primary_label.unique()):\n    LABELS[item] = i\n    i += 1\nprint(LABELS)","metadata":{"execution":{"iopub.status.busy":"2021-05-30T18:32:32.653710Z","iopub.execute_input":"2021-05-30T18:32:32.654142Z","iopub.status.idle":"2021-05-30T18:32:32.692692Z","shell.execute_reply.started":"2021-05-30T18:32:32.654105Z","shell.execute_reply":"2021-05-30T18:32:32.691754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Generate Names for Config\nNAMES = {}\nlabel_name = TRAIN.groupby(['primary_label', 'common_name']).count()\nfor label, name in label_name.index:\n    NAMES[label] = name\nprint(NAMES)","metadata":{"execution":{"iopub.status.busy":"2021-05-30T18:32:36.683019Z","iopub.execute_input":"2021-05-30T18:32:36.683362Z","iopub.status.idle":"2021-05-30T18:32:36.758170Z","shell.execute_reply.started":"2021-05-30T18:32:36.683326Z","shell.execute_reply":"2021-05-30T18:32:36.757316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%writefile config_params/example_config.py\nfrom pathlib import Path\nimport torch\nfrom config_params.configs import get_dict_value, BIRD_CODE, INV_EBIRD_LABEL\nimport os\n\nclass Parameters(object):\n    def __init__(self, hparams=None):\n        self.fold = 0\n        self.name = os.path.basename(__file__).replace(\".py\",\"\")\n        \n        self.aug_name = \"secondary_default\"\n        self.apply_mixup = False\n        self.model_name = \"sed_dense121att\"\n\n        self.model_config =  {\n            \"sample_rate\": 32000,\n            \"window_size\": 1024,\n            \"hop_size\": 320,\n            \"mel_bins\": 64,\n            \"fmin\": 50,\n            \"fmax\": 14000,\n            \"classes_num\": 397,\n            \"apply_aug\": True,\n            \"top_db\": None\n        }\n        self.pretrained_path = None\n\n        self.bckgrd_aug_dir = \"../../../input/pinknoise\" \n        self.secondary_bckgrd_aug_dir = \"../../../input/pinknoise\"\n\n        self.optimizer_name = \"adamw\"\n        self.weight_decay = 0.01\n\n        self.criterion_name = \"sed_scaled_pos_neg_focal_loss\"\n        self.criterion_params = {\n            \"gamma\" : 0.0,\n            \"alpha_0\" : 1.0,\n            \"alpha_1\": 1.0,\n            \"secondary_factor\": 1.0\n        }\n        \n\n        self.scheduler_name = \"warmup_with_cosine\"\n        self.lr_scale_factor = 0.01\n        self.lr = 0.001\n\n        self.logger_name = \"print\" # set to \"neptune\" if using neptune logger instead\n        \n        self.PERIOD = 30\n\n        self.train_ds_params = {\n            \"root_dir\": get_dict_value(Path(\"../../../input/birdclef-2021/train_short_audio/\")), # Set to Path(\"data/\") if not using Kaggle Kernel if the data is in data/\n            \"csv_dir\": Path(f\"../../csv/train.csv\"),\n            \"period\": self.PERIOD,\n            \"bird_code\": BIRD_CODE,\n            \"inv_ebird_label\":INV_EBIRD_LABEL,\n            \"isTraining\": True,\n            \"num_test_samples\": 1,\n        }\n        \n        self.valid_ds_params = {\n            \"root_dir\": get_dict_value(Path(\"../../../input/birdclef-2021/train_short_audio/\")), # Set to Path(\"data/\") if not using Kaggle Kernel if the data is in data/\n            \"csv_dir\": Path(f\"../../csv/test.csv\"),\n            \"period\": self.PERIOD,\n            \"bird_code\": BIRD_CODE,\n            \"inv_ebird_label\":INV_EBIRD_LABEL,\n            \"background_audio_dir\": None,\n            \"isTraining\": False,\n            \"num_test_samples\": 2,\n        }\n\n        self.test_ds_params = {\n            \"root_dir\": get_dict_value(Path(\"../../../input/birdclef-2021/train_short_audio/\")), # Set to Path(\"data/\") if not using Kaggle Kernel if the data is in data/\n             \"csv_dir\": Path(f\"../../csv/test.csv\"),\n            \"background_audio_dir\":  None,\n            \"period\": self.PERIOD,\n            \"bird_code\": BIRD_CODE,\n            \"inv_ebird_label\":INV_EBIRD_LABEL,\n            \"isTraining\": False,\n            \"num_test_samples\": 2,\n        }\n\n        self.checkpoint_params = {\n            \"save_dir\":f\"../../saved_models/{self.name}\", # Path to save the checkpoints\n            \"n_saved\":2,\n            \"prefix_name\":self.name,\n        }\n        \n        self.train_bs = 28\n        self.train_num_workers = 2 # changed to 2 workers\n        self.valid_bs = 32\n        self.valid_num_workers = 1 # changed to 1 workers\n        self.metrics = [\"lraps\", \"f1score_clip\", \"f1score_frame\"]\n        \n        self.track_metric = \"f1score_clip\"\n        self.metric_factor = 1\n        \n        self.checkpoint_dir = None\n        self.add_pbar = True\n        \n        self.run_params = {\n            \"max_epochs\": 2, # 50,\n            \"epoch_length\": None\n        }\n        \n        self.logger_params = {\n            \"project_name\": \"bird-song\",\n            \"log_every\": 1, # 10,\n            \"name\": self.name,\n            \"prefix_name\": f\"{self.name}_best\",\n            \"tags\": [self.fold, self.name, self.model_name, self.criterion_name],\n            \"params\": {\n                \"bs\": self.train_bs,\n                \"lr\": self.lr,\n                \"name\": self.name,\n                \"aug_name\": self.aug_name,\n                \"model_name\": self.model_name,\n                \"weight_decay\": self.weight_decay,\n                \"apply_mixup\": self.apply_mixup,\n                \"optimizer_name\": self.optimizer_name,\n                \"criterion_name\": self.criterion_name,\n                \"scheduler_name\": self.scheduler_name,\n                \"fold\": self.fold,\n                \"lr_scale_factor\": self.lr_scale_factor,\n                \"period\": self.PERIOD,\n                **self.model_config,\n                **self.criterion_params\n            }\n        }\n        \n        self.dist_params = {\n        }\n\n        self.val_length = None \n        self.eval_every = 2 # Validates every 2 epochs\n        self.load_model_only = True\n        self.accumulation_steps = 1\n        self.gradient_clip_val = 0","metadata":{"execution":{"iopub.status.busy":"2021-05-30T19:42:25.650609Z","iopub.execute_input":"2021-05-30T19:42:25.650959Z","iopub.status.idle":"2021-05-30T19:42:25.685340Z","shell.execute_reply.started":"2021-05-30T19:42:25.650925Z","shell.execute_reply":"2021-05-30T19:42:25.684445Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!python sed_train.py --config \"config_params.example_config\"","metadata":{"execution":{"iopub.status.busy":"2021-05-30T19:42:38.522139Z","iopub.execute_input":"2021-05-30T19:42:38.522521Z","iopub.status.idle":"2021-05-30T21:54:53.685842Z","shell.execute_reply.started":"2021-05-30T19:42:38.522485Z","shell.execute_reply":"2021-05-30T21:54:53.682994Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls ../../saved_models/example_config","metadata":{"execution":{"iopub.status.busy":"2021-05-30T19:40:42.438910Z","iopub.execute_input":"2021-05-30T19:40:42.439248Z","iopub.status.idle":"2021-05-30T19:40:43.094580Z","shell.execute_reply.started":"2021-05-30T19:40:42.439215Z","shell.execute_reply":"2021-05-30T19:40:43.093331Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]}]}