{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Approach 2: Using 1D CNN on raw Audio waveforms\n\nThe idea is:\n1. Convert Audio to numpy after performing a STFT.\n2. Clip 15 second of random audio to perform STFT\n3. Use features from clipped STFT as an input to 1D CNN\n\nResources:\n\n[Resampled Dataset](https://www.kaggle.com/harveenchadha/pogmusicclassification)","metadata":{}},{"cell_type":"code","source":"\nimport os\nos.environ[\"KMP_AFFINITY\"] = \"noverbose\"\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\n\nimport numpy as np\nimport tensorflow as tf\ntf.get_logger().setLevel('ERROR')\n\nimport glob\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\nfrom joblib import Parallel, delayed\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, classification_report\nimport warnings\nimport soundfile as sf\n\nimport librosa\nimport gc\nwarnings.filterwarnings('ignore')\nimport wandb\nfrom wandb.keras import WandbCallback\nfrom kaggle_secrets import UserSecretsClient","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-17T20:34:14.933229Z","iopub.execute_input":"2022-03-17T20:34:14.934219Z","iopub.status.idle":"2022-03-17T20:34:22.817482Z","shell.execute_reply.started":"2022-03-17T20:34:14.934092Z","shell.execute_reply":"2022-03-17T20:34:22.816369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = {\n    'SEED' : 42,\n    'DEBUG': False,\n    'test_size':0.1,\n    'batch_size':16,\n    'num_labels':0,\n    'epochs':10,\n    'device':'GPU',\n    'max_duration' : 15,\n    'sample_rate': 16000,\n    'USE_WANDB':True,\n    'WANDB_MODE': 'online',\n    'WANDB_PROJECT': 'POG-Music',\n    \n}\nconfig['target_size'] = config['max_duration'] * config['sample_rate']","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:40:46.370461Z","iopub.execute_input":"2022-03-17T18:40:46.370801Z","iopub.status.idle":"2022-03-17T18:40:46.377617Z","shell.execute_reply.started":"2022-03-17T18:40:46.37075Z","shell.execute_reply":"2022-03-17T18:40:46.376619Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(SEED):\n    os.environ['PYTHONHASHSEED'] = str(SEED)\n    np.random.seed(SEED)\n    tf.random.set_seed(SEED)\n    \nset_seed(config['SEED'])\n\ndef get_device(device):\n    if device == 'TPU':\n        try: # detect TPUs\n            tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect() # TPU detection\n            strategy = tf.distribute.TPUStrategy(tpu)\n        except ValueError: # detect GPUs\n            print('Cannot initialize TPU')\n    if device == 'GPU':\n        strategy = tf.distribute.MirroredStrategy() \n\n    print(\"Number of accelerators: \", strategy.num_replicas_in_sync)\n    return strategy\n\nstrategy= get_device(config['device'])\nconfig['batch_size'] = config['batch_size'] * strategy.num_replicas_in_sync\n\n\ndef use_wandb():\n    if config['WANDB_MODE'] == 'offline':\n        os.environ[\"WANDB_MODE\"] = \"offline\"\n        key='X'*40\n        wandb.login(key=key)\n    else:\n        user_secrets = UserSecretsClient()\n        wandb_api = user_secrets.get_secret(\"wandb_api\")\n        wandb.login(key=wandb_api)\n\n    run = wandb.init(project=config['WANDB_PROJECT'], \n                     job_type='train',\n                     config = config)\n\n    return run\n\nif config['USE_WANDB']:\n    run = use_wandb()\n","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-03-17T18:40:46.379387Z","iopub.execute_input":"2022-03-17T18:40:46.38001Z","iopub.status.idle":"2022-03-17T18:40:49.432456Z","shell.execute_reply.started":"2022-03-17T18:40:46.379962Z","shell.execute_reply":"2022-03-17T18:40:49.431324Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Preprocessing","metadata":{}},{"cell_type":"code","source":"df_train = pd.read_csv('../input/kaggle-pog-series-s01e02/train.csv')\ndf_test = pd.read_csv('../input/kaggle-pog-series-s01e02/test.csv')\n\ndf_train['ID'] = df_train['filename'].str.split('.').str[0]\ndf_test['ID'] = df_test['filename'].str.split('.').str[0]\n\ntrain_files = pd.DataFrame(glob.glob('../input/pogmusicclassification/resampled_train/*.wav'), columns=['resampled_path'])\ntest_files = pd.DataFrame(glob.glob('../input/pogmusicclassification/resampled_test/*.wav'), columns=['resampled_path'])\n\ntrain_files['ID'] = train_files['resampled_path'].str.split('/').str[-1].str.split('.').str[0].str.split('_').str[0]\ntest_files['ID'] = test_files['resampled_path'].str.split('/').str[-1].str.split('.').str[0].str.split('_').str[0]\n\ndf_train = pd.merge(df_train, train_files, how='right', on='ID')\ndf_test = pd.merge(df_test, test_files, how='right', on='ID')\n\ntrain_song_ids_to_drop = [17400, 18390, 1975, 8114, 13437]\ndf_train = df_train[~df_train.song_id.isin(train_song_ids_to_drop)]","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:40:49.434742Z","iopub.execute_input":"2022-03-17T18:40:49.434992Z","iopub.status.idle":"2022-03-17T18:40:51.128962Z","shell.execute_reply.started":"2022-03-17T18:40:49.434959Z","shell.execute_reply":"2022-03-17T18:40:51.127934Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training Distribution","metadata":{}},{"cell_type":"code","source":"config['num_labels'] = df_train['genre_id'].nunique()\ndf_train.genre_id.value_counts(normalize=True) * 100","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:40:51.130538Z","iopub.execute_input":"2022-03-17T18:40:51.130882Z","iopub.status.idle":"2022-03-17T18:40:51.148009Z","shell.execute_reply.started":"2022-03-17T18:40:51.130838Z","shell.execute_reply":"2022-03-17T18:40:51.146788Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train Valid Split","metadata":{}},{"cell_type":"code","source":"X_train, X_valid = train_test_split(df_train, test_size = config['test_size'], random_state=config['SEED'], shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:40:51.150217Z","iopub.execute_input":"2022-03-17T18:40:51.150969Z","iopub.status.idle":"2022-03-17T18:40:51.16701Z","shell.execute_reply.started":"2022-03-17T18:40:51.150919Z","shell.execute_reply":"2022-03-17T18:40:51.165809Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataloaders","metadata":{}},{"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:40:52.054886Z","iopub.execute_input":"2022-03-17T18:40:52.055574Z","iopub.status.idle":"2022-03-17T18:40:52.066235Z","shell.execute_reply.started":"2022-03-17T18:40:52.055522Z","shell.execute_reply":"2022-03-17T18:40:52.064835Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"##randomly cropping out 15s of the audio\n\nclass POGDataset:\n    def crop_to_max_size(self, wav):\n        size = len(wav)\n#         if size < config['target_size']:\n#             return None\n\n        diff = size - config['target_size']\n        if diff <= 0:\n            return wav\n\n        start = np.random.randint(0, diff + 1)\n        end = size - diff + start\n        return wav[start:end]\n\n    def process_file(self, file, label):\n        y, sr = sf.read(file)\n        y = self.crop_to_max_size(y)\n\n        D = librosa.stft(y)\n        del y, sr\n        spect, phase = librosa.magphase(D)\n        spect = np.log1p(spect)\n        del D, phase\n        return spect, label\n    \n    def process_file_test(self, file, id):\n        y, sr = sf.read(file)\n        y = self.crop_to_max_size(y)\n        D = librosa.stft(y)\n        del y, sr\n        spect, phase = librosa.magphase(D)\n        spect = np.log1p(spect)\n        del D, phase\n        return spect, id\n    \n    \n    \n    def get_tf_data_pipeline(self, paths, labels):\n        ds = tf.data.Dataset.from_tensor_slices((paths, labels))\n        ds = ds.map(lambda x , y : tf.numpy_function(func=self.process_file, inp=[x, y], Tout=[tf.float64, tf.int64]), num_parallel_calls=2)\n        return ds\n    \n    def get_tf_data_pipeline_test(self, paths, ids):\n        ds = tf.data.Dataset.from_tensor_slices((paths, ids))\n        ds = ds.map(lambda x, y : tf.numpy_function(func=self.process_file_test, inp=[x, y], Tout=[tf.float64, tf.string]), num_parallel_calls=AUTOTUNE)\n        return ds","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:56:41.873066Z","iopub.execute_input":"2022-03-17T18:56:41.87339Z","iopub.status.idle":"2022-03-17T18:56:41.889417Z","shell.execute_reply.started":"2022-03-17T18:56:41.873355Z","shell.execute_reply":"2022-03-17T18:56:41.888314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dataset = POGDataset()\ntrain_ds = dataset.get_tf_data_pipeline(X_train.resampled_path.values, X_train.genre_id.values)\nvalid_ds = dataset.get_tf_data_pipeline(X_valid.resampled_path.values, X_valid.genre_id.values)\n","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:56:45.572101Z","iopub.execute_input":"2022-03-17T18:56:45.572746Z","iopub.status.idle":"2022-03-17T18:56:45.614732Z","shell.execute_reply.started":"2022-03-17T18:56:45.572709Z","shell.execute_reply":"2022-03-17T18:56:45.613695Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def configure_for_performance(ds, batch_size = 32):\n    ds = ds.cache('/kaggle/dump.tfcache') \n    ds = ds.shuffle(buffer_size=32)\n    ds = ds.batch(batch_size)\n    ds = ds.prefetch(buffer_size=AUTOTUNE)\n    return ds\n\ntrain_ds_batch = configure_for_performance(train_ds)\nvalid_ds_batch = valid_ds.batch(config['batch_size'])\n","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:56:47.257331Z","iopub.execute_input":"2022-03-17T18:56:47.258214Z","iopub.status.idle":"2022-03-17T18:56:47.27302Z","shell.execute_reply.started":"2022-03-17T18:56:47.258177Z","shell.execute_reply":"2022-03-17T18:56:47.271903Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_train_batch = next(iter(train_ds))\ninput_dim_1 = test_train_batch[0].shape[0]\ninput_dim_2 = test_train_batch[0].shape[1] \n\ntest_train_batch","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:43:03.377098Z","iopub.execute_input":"2022-03-17T18:43:03.377414Z","iopub.status.idle":"2022-03-17T18:43:03.705872Z","shell.execute_reply.started":"2022-03-17T18:43:03.37738Z","shell.execute_reply":"2022-03-17T18:43:03.704696Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modelling","metadata":{}},{"cell_type":"code","source":"with strategy.scope():\n    model = tf.keras.models.Sequential(\n        [\n            tf.keras.layers.Conv1D(16, kernel_size=5,  activation='relu', input_shape = [input_dim_1, input_dim_2]),\n            tf.keras.layers.Conv1D(32, kernel_size=3, activation='relu'),\n            tf.keras.layers.Conv1D(64, kernel_size=3, strides= 2, activation='relu'),\n            tf.keras.layers.Conv1D(128, kernel_size=3,  activation='relu'),\n            tf.keras.layers.Conv1D(256, kernel_size=3, strides= 2,  activation='relu'),\n            tf.keras.layers.Conv1D(512, kernel_size=3, strides= 2, activation='relu'),\n            tf.keras.layers.GlobalAveragePooling1D(),\n            tf.keras.layers.Flatten(),\n            tf.keras.layers.Dense(128, activation='relu'),\n            tf.keras.layers.Dense(64, activation='relu'),\n            tf.keras.layers.Dropout(0.2),\n            tf.keras.layers.Dense(config['num_labels'], activation='softmax')\n        ]\n    )\n\n    model.compile(loss = tf.keras.losses.SparseCategoricalCrossentropy(),\n                 optimizer=tf.keras.optimizers.Adam(learning_rate=0.001),\n                 metrics='sparse_categorical_accuracy')","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:43:07.793346Z","iopub.execute_input":"2022-03-17T18:43:07.793807Z","iopub.status.idle":"2022-03-17T18:43:09.009209Z","shell.execute_reply.started":"2022-03-17T18:43:07.793732Z","shell.execute_reply":"2022-03-17T18:43:09.00816Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:43:09.124241Z","iopub.execute_input":"2022-03-17T18:43:09.124538Z","iopub.status.idle":"2022-03-17T18:43:09.147Z","shell.execute_reply.started":"2022-03-17T18:43:09.124505Z","shell.execute_reply":"2022-03-17T18:43:09.145699Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Callbacks","metadata":{}},{"cell_type":"code","source":"best_weight_path = 'best_model.hdf5'\nlast_weight_path = 'last_model.hdf5'\n\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(best_weight_path, \n                             monitor= 'val_loss', \n                             verbose=1, \n                             save_best_only=True, \n                             mode= 'min', \n                             save_weights_only = False)\ncheckpoint_last = tf.keras.callbacks.ModelCheckpoint(last_weight_path, \n                             monitor= 'val_loss', \n                             verbose=1, \n                             save_best_only=False, \n                             mode= 'min', \n                             save_weights_only = False)\n\n\nearly = tf.keras.callbacks.EarlyStopping(monitor= 'val_loss', \n                      mode= 'min', \n                      patience=4)\n\nreduceLROnPlat = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.8, patience=2, verbose=1, mode='auto', epsilon=0.0001, cooldown=5, min_lr=0.00001)\ncallbacks_list = [checkpoint, checkpoint_last, early, reduceLROnPlat]\n\nif config['USE_WANDB']:\n    callbacks_list.append(WandbCallback())","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:43:15.782217Z","iopub.execute_input":"2022-03-17T18:43:15.782512Z","iopub.status.idle":"2022-03-17T18:43:15.79128Z","shell.execute_reply.started":"2022-03-17T18:43:15.78248Z","shell.execute_reply":"2022-03-17T18:43:15.790262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!nvidia-smi","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:43:17.991281Z","iopub.execute_input":"2022-03-17T18:43:17.991786Z","iopub.status.idle":"2022-03-17T18:43:18.835668Z","shell.execute_reply.started":"2022-03-17T18:43:17.991749Z","shell.execute_reply":"2022-03-17T18:43:18.83451Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training","metadata":{}},{"cell_type":"code","source":"history = model.fit(train_ds_batch, callbacks=callbacks_list, epochs=config['epochs'], validation_data=valid_ds_batch)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:43:43.86105Z","iopub.execute_input":"2022-03-17T18:43:43.861741Z","iopub.status.idle":"2022-03-17T18:44:39.40017Z","shell.execute_reply.started":"2022-03-17T18:43:43.861694Z","shell.execute_reply":"2022-03-17T18:44:39.396988Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_hist(hist):\n    plt.figure(figsize=(15,5))\n    local_epochs = len(hist.history[\"sparse_categorical_accuracy\"])\n    plt.plot(np.arange(local_epochs, step=1), hist.history[\"sparse_categorical_accuracy\"], '-o', label='Train Accuracy',color='#ff7f0e')\n    plt.plot(np.arange(local_epochs, step=1), hist.history[\"val_sparse_categorical_accuracy\"], '-o',label='Val Accuracy',color='#1f77b4')\n    plt.xlabel('Epoch',size=14)\n    plt.ylabel('Accuracy',size=14)\n    plt.legend(loc=2)\n    \n    plt2 = plt.gca().twinx()\n    plt2.plot(np.arange(local_epochs, step=1) ,history.history['loss'],'-o',label='Train Loss',color='#2ca02c')\n    plt2.plot(np.arange(local_epochs, step=1) ,history.history['val_loss'],'-o',label='Val Loss',color='#d62728')\n    plt.legend(loc=3)\n    plt.ylabel('Loss',size=14)\n    plt.title(\"Model Accuracy and loss\")\n    \n    plt.savefig('loss.png')\n    plt.show()\n    \nplot_hist(history)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluation on valid set","metadata":{}},{"cell_type":"code","source":"model.load_weights('./best_model.hdf5')\npred_valid_y = model.predict(valid_ds_batch, workers=4, verbose = True)\npred_valid_y_labels = np.argmax(pred_valid_y, axis=-1)\n\nvalid_labels = np.concatenate([y.numpy() for x, y in valid_ds_batch], axis=0)\nprint(classification_report(valid_labels, pred_valid_y_labels ))","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:48:29.354437Z","iopub.execute_input":"2022-03-17T18:48:29.355035Z","iopub.status.idle":"2022-03-17T18:48:34.671355Z","shell.execute_reply.started":"2022-03-17T18:48:29.354996Z","shell.execute_reply":"2022-03-17T18:48:34.669679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction on test set","metadata":{}},{"cell_type":"code","source":"del valid_ds_batch, train_ds_batch\ngc.collect()","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_ds = dataset.get_tf_data_pipeline_test(df_test.resampled_path.values, df_test.ID.values)\ntest_ds_batch = test_ds.batch(config['batch_size']*2)\n\npreds = []\nfor spect, id  in tqdm(test_ds_batch):\n    local_preds = model.predict(spect, workers=4)\n    \n    for idx, local_id in enumerate(id.numpy()):\n        preds.append({'song_id':local_id.decode(\"utf-8\") , 'genre_id': np.argmax(local_preds[idx], axis=-1)})","metadata":{"execution":{"iopub.status.busy":"2022-03-17T18:56:57.965885Z","iopub.execute_input":"2022-03-17T18:56:57.966192Z","iopub.status.idle":"2022-03-17T19:04:19.941677Z","shell.execute_reply.started":"2022-03-17T18:56:57.966158Z","shell.execute_reply":"2022-03-17T19:04:19.939596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Adding missing values","metadata":{}},{"cell_type":"code","source":"preds.append({'song_id':'022612', 'genre_id':1})\npreds.append({'song_id':'024013', 'genre_id':0})\n\ntest_df = pd.DataFrame.from_dict(preds)\ntest_df['song_id'] = test_df['song_id'].astype('int')\n\nassert(len(test_df) == 5078)\ntest_df.to_csv('submission.csv', index=False)\n\n\nif config['USE_WANDB']:\n    run.finish()","metadata":{"execution":{"iopub.status.busy":"2022-03-17T19:04:23.697181Z","iopub.execute_input":"2022-03-17T19:04:23.697502Z","iopub.status.idle":"2022-03-17T19:04:23.773157Z","shell.execute_reply.started":"2022-03-17T19:04:23.697468Z","shell.execute_reply":"2022-03-17T19:04:23.772107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Notes:\n\n1. You can try and train the model of your choice.\n2. Feel free to change hyperparameters in the config section\n\n\n**Please do upvote if this was useful for you**","metadata":{}}]}