{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Approach 1: Using Spectograms with 2D CNN\n\nThe idea is:\n1. Convert Audio to spectrograms.\n2. Use spectogram with 2D CNN.\n\n\nUpdates:\n\n17/03 : Remove all tensorflow warnings\n\nResources:\n\n[Spectrogram Generator Notebook](https://www.kaggle.com/harveenchadha/pog-spectogram-generator)\n\n[Spectrogram Dataset](https://www.kaggle.com/harveenchadha/pog-train-and-test-spectograms)","metadata":{}},{"cell_type":"code","source":"import os\nos.environ[\"KMP_AFFINITY\"] = \"noverbose\"\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3'\n\n\nimport numpy as np\nimport tensorflow as tf\ntf.get_logger().setLevel('ERROR')\n\nimport glob\nimport pandas as pd\nimport matplotlib.pyplot as plt\nfrom tqdm.notebook import tqdm\n\nfrom sklearn.model_selection import train_test_split\nfrom sklearn.metrics import confusion_matrix, classification_report\nimport warnings\n\nwarnings.filterwarnings('ignore')","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-17T17:18:37.407348Z","iopub.execute_input":"2022-03-17T17:18:37.407842Z","iopub.status.idle":"2022-03-17T17:18:37.414421Z","shell.execute_reply.started":"2022-03-17T17:18:37.407805Z","shell.execute_reply":"2022-03-17T17:18:37.413568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"config = {\n    'SEED' : 42,\n    'DEBUG': False,\n    'test_size':0.1,\n    'img_size':256,\n    'batch_size':16,\n    'num_labels':0,\n    'epochs':20,\n    'device':'GPU'\n}\n","metadata":{"execution":{"iopub.status.busy":"2022-03-17T17:00:58.889792Z","iopub.execute_input":"2022-03-17T17:00:58.890547Z","iopub.status.idle":"2022-03-17T17:00:58.897139Z","shell.execute_reply.started":"2022-03-17T17:00:58.890505Z","shell.execute_reply":"2022-03-17T17:00:58.896164Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def set_seed(SEED):\n    os.environ['PYTHONHASHSEED'] = str(SEED)\n    np.random.seed(SEED)\n    tf.random.set_seed(SEED)\n    \nset_seed(config['SEED'])\n\ndef get_device(device):\n    if device == 'TPU':\n        try: # detect TPUs\n            tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect() # TPU detection\n            strategy = tf.distribute.TPUStrategy(tpu)\n        except ValueError: # detect GPUs\n            print('Cannot initialize TPU')\n    if device == 'GPU':\n        strategy = tf.distribute.MirroredStrategy() \n\n    print(\"Number of accelerators: \", strategy.num_replicas_in_sync)\n    return strategy\n\nstrategy= get_device(config['device'])\nconfig['batch_size'] = config['batch_size'] * strategy.num_replicas_in_sync","metadata":{"_kg_hide-input":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-03-17T17:01:09.200487Z","iopub.execute_input":"2022-03-17T17:01:09.200746Z","iopub.status.idle":"2022-03-17T17:01:11.428352Z","shell.execute_reply.started":"2022-03-17T17:01:09.200715Z","shell.execute_reply":"2022-03-17T17:01:11.427535Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Data Preprocessing","metadata":{}},{"cell_type":"code","source":"## Reading spectograms and ignoring files not present\n\ndf_train = pd.read_csv('../input/kaggle-pog-series-s01e02/train.csv')\ndf_test = pd.read_csv('../input/kaggle-pog-series-s01e02/test.csv')\n\ndf_train['ID'] = df_train['filename'].str.split('.').str[0]\ndf_test['ID'] = df_test['filename'].str.split('.').str[0]\n\ntrain_files = pd.DataFrame(glob.glob('../input/pog-train-and-test-spectograms/kaggle/spectograms/train/*.jpg'), columns=['spec_path'])\ntest_files = pd.DataFrame(glob.glob('../input/pog-train-and-test-spectograms/kaggle/spectograms/test/*.jpg'), columns=['spec_path'])\n\ntrain_files['ID'] = train_files['spec_path'].str.split('/').str[-1].str.split('.').str[0]\ntest_files['ID'] = test_files['spec_path'].str.split('/').str[-1].str.split('.').str[0]\n\ndf_train_spec = pd.merge(df_train, train_files, how='right', on='ID')\ndf_test_spec = pd.merge(df_test, test_files, how='right', on='ID')","metadata":{"execution":{"iopub.status.busy":"2022-03-17T17:01:11.660719Z","iopub.execute_input":"2022-03-17T17:01:11.661402Z","iopub.status.idle":"2022-03-17T17:01:13.446257Z","shell.execute_reply.started":"2022-03-17T17:01:11.661354Z","shell.execute_reply":"2022-03-17T17:01:13.445497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training Distribution","metadata":{}},{"cell_type":"code","source":"config['num_labels'] = df_train_spec['genre_id'].nunique()\ndf_train_spec.genre_id.value_counts(normalize=True) * 100","metadata":{"execution":{"iopub.status.busy":"2022-03-17T17:01:18.131188Z","iopub.execute_input":"2022-03-17T17:01:18.131707Z","iopub.status.idle":"2022-03-17T17:01:18.146010Z","shell.execute_reply.started":"2022-03-17T17:01:18.131668Z","shell.execute_reply":"2022-03-17T17:01:18.145364Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Train Valid Split","metadata":{}},{"cell_type":"code","source":"X_train, X_valid = train_test_split(df_train_spec, test_size = config['test_size'], random_state=config['SEED'], shuffle=True)","metadata":{"execution":{"iopub.status.busy":"2022-03-17T17:01:19.808711Z","iopub.execute_input":"2022-03-17T17:01:19.809238Z","iopub.status.idle":"2022-03-17T17:01:19.823023Z","shell.execute_reply.started":"2022-03-17T17:01:19.809196Z","shell.execute_reply":"2022-03-17T17:01:19.822368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Dataloaders","metadata":{}},{"cell_type":"code","source":"AUTOTUNE = tf.data.experimental.AUTOTUNE","metadata":{"execution":{"iopub.status.busy":"2022-03-17T17:01:20.681739Z","iopub.execute_input":"2022-03-17T17:01:20.683034Z","iopub.status.idle":"2022-03-17T17:01:20.687164Z","shell.execute_reply.started":"2022-03-17T17:01:20.682989Z","shell.execute_reply":"2022-03-17T17:01:20.686481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def process_data_train(image_path, label):\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.random_brightness(img, 0.1)\n    img = tf.image.resize(img,[config['img_size'], config['img_size']])\n    return img, label\n\ndef process_data_valid(image_path, label):\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    img = tf.image.resize(img,[config['img_size'], config['img_size']])\n    return img, label\n\ndef configure_for_performance(ds, batch_size = 32):\n    ds = ds.cache('/kaggle/dump.tfcache') \n    ds = ds.shuffle(buffer_size=32)\n    ds = ds.batch(batch_size)\n    ds = ds.prefetch(buffer_size=AUTOTUNE)\n    return ds\n\n\ntrain_ds = tf.data.Dataset.from_tensor_slices((X_train.spec_path.values, X_train.genre_id.values))\nvalid_ds = tf.data.Dataset.from_tensor_slices((X_valid.spec_path.values, X_valid.genre_id.values))\n\n\ntrain_ds = train_ds.map(process_data_train, num_parallel_calls=AUTOTUNE)\nvalid_ds = valid_ds.map(process_data_valid, num_parallel_calls=AUTOTUNE)\n\ntrain_ds_batch = configure_for_performance(train_ds, config['batch_size'])\nvalid_ds_batch = valid_ds.batch(config['batch_size']*2)","metadata":{"execution":{"iopub.status.busy":"2022-03-17T17:03:39.408821Z","iopub.execute_input":"2022-03-17T17:03:39.409515Z","iopub.status.idle":"2022-03-17T17:03:39.452511Z","shell.execute_reply.started":"2022-03-17T17:03:39.409476Z","shell.execute_reply":"2022-03-17T17:03:39.451831Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image_batch, label_batch = next(iter(train_ds_batch))\nplt.figure(figsize=(14, 14))\nfor i in range(8):\n    ax = plt.subplot(4, 4, i + 1)\n    plt.imshow(image_batch[i].numpy().astype(\"uint8\"))\n    label = label_batch[i].numpy()\n    plt.title(label)\n    plt.axis(\"off\")","metadata":{"execution":{"iopub.status.busy":"2022-03-17T17:03:40.544511Z","iopub.execute_input":"2022-03-17T17:03:40.544760Z","iopub.status.idle":"2022-03-17T17:03:57.719563Z","shell.execute_reply.started":"2022-03-17T17:03:40.544730Z","shell.execute_reply":"2022-03-17T17:03:57.718700Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Modelling","metadata":{}},{"cell_type":"code","source":"with strategy.scope():\n    model = tf.keras.models.Sequential(\n        [\n            tf.keras.layers.Rescaling(1./255, input_shape = (config['img_size'], config['img_size'], 3)),\n            tf.keras.layers.Conv2D(16, kernel_size=5,  activation='relu'),\n            tf.keras.layers.Conv2D(32, kernel_size=3, activation='relu'),\n            tf.keras.layers.Conv2D(64, kernel_size=3, strides= 2, activation='relu'),\n            tf.keras.layers.Conv2D(128, kernel_size=3,  activation='relu'),\n            tf.keras.layers.Conv2D(256, kernel_size=3, strides= 2,  activation='relu'),\n            tf.keras.layers.Conv2D(512, kernel_size=3, strides= 2, activation='relu'),\n            tf.keras.layers.GlobalAveragePooling2D(),\n            tf.keras.layers.Flatten(),\n            tf.keras.layers.Dense(128, activation='relu'),\n            tf.keras.layers.Dense(64, activation='relu'),\n            tf.keras.layers.Dropout(0.2),\n            tf.keras.layers.Dense(config['num_labels'], activation='softmax')\n        ]\n    )\n\n    model.compile(loss = tf.keras.losses.SparseCategoricalCrossentropy(),\n                 optimizer='adam',\n                 metrics='sparse_categorical_accuracy')","metadata":{"execution":{"iopub.status.busy":"2022-03-17T17:01:29.351651Z","iopub.execute_input":"2022-03-17T17:01:29.351885Z","iopub.status.idle":"2022-03-17T17:01:30.532523Z","shell.execute_reply.started":"2022-03-17T17:01:29.351839Z","shell.execute_reply":"2022-03-17T17:01:30.531806Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.summary()","metadata":{"execution":{"iopub.status.busy":"2022-03-17T17:01:30.533661Z","iopub.execute_input":"2022-03-17T17:01:30.533925Z","iopub.status.idle":"2022-03-17T17:01:30.548350Z","shell.execute_reply.started":"2022-03-17T17:01:30.533873Z","shell.execute_reply":"2022-03-17T17:01:30.545830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Callbacks","metadata":{}},{"cell_type":"code","source":"best_weight_path = 'best_model.hdf5'\nlast_weight_path = 'last_model.hdf5'\n\ncheckpoint = tf.keras.callbacks.ModelCheckpoint(best_weight_path, \n                             monitor= 'val_loss', \n                             verbose=1, \n                             save_best_only=True, \n                             mode= 'min', \n                             save_weights_only = False)\ncheckpoint_last = tf.keras.callbacks.ModelCheckpoint(last_weight_path, \n                             monitor= 'val_loss', \n                             verbose=1, \n                             save_best_only=False, \n                             mode= 'min', \n                             save_weights_only = False)\n\n\nearly = tf.keras.callbacks.EarlyStopping(monitor= 'val_loss', \n                      mode= 'min', \n                      patience=4)\n\nreduceLROnPlat = tf.keras.callbacks.ReduceLROnPlateau(monitor='val_loss', factor=0.8, patience=2, verbose=1, mode='auto', epsilon=0.0001, cooldown=5, min_lr=0.00001)\ncallbacks_list = [checkpoint, checkpoint_last, early, reduceLROnPlat]","metadata":{"execution":{"iopub.status.busy":"2022-03-17T17:01:36.528459Z","iopub.execute_input":"2022-03-17T17:01:36.529105Z","iopub.status.idle":"2022-03-17T17:01:36.538674Z","shell.execute_reply.started":"2022-03-17T17:01:36.529058Z","shell.execute_reply":"2022-03-17T17:01:36.537979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!nvidia-smi","metadata":{"execution":{"iopub.status.busy":"2022-03-17T17:01:37.288005Z","iopub.execute_input":"2022-03-17T17:01:37.288674Z","iopub.status.idle":"2022-03-17T17:01:38.003752Z","shell.execute_reply.started":"2022-03-17T17:01:37.288636Z","shell.execute_reply":"2022-03-17T17:01:38.002918Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Training","metadata":{}},{"cell_type":"code","source":"history = model.fit(train_ds_batch, callbacks=callbacks_list, epochs=config['epochs'], validation_data=valid_ds_batch)","metadata":{"execution":{"iopub.status.busy":"2022-03-17T17:01:39.093184Z","iopub.execute_input":"2022-03-17T17:01:39.093896Z","iopub.status.idle":"2022-03-17T17:02:03.660938Z","shell.execute_reply.started":"2022-03-17T17:01:39.093827Z","shell.execute_reply":"2022-03-17T17:02:03.658117Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_hist(hist):\n    plt.figure(figsize=(15,5))\n    local_epochs = len(hist.history[\"sparse_categorical_accuracy\"])\n    plt.plot(np.arange(local_epochs, step=1), hist.history[\"sparse_categorical_accuracy\"], '-o', label='Train Accuracy',color='#ff7f0e')\n    plt.plot(np.arange(local_epochs, step=1), hist.history[\"val_sparse_categorical_accuracy\"], '-o',label='Val Accuracy',color='#1f77b4')\n    plt.xlabel('Epoch',size=14)\n    plt.ylabel('Accuracy',size=14)\n    plt.legend(loc=2)\n    \n    plt2 = plt.gca().twinx()\n    plt2.plot(np.arange(local_epochs, step=1) ,history.history['loss'],'-o',label='Train Loss',color='#2ca02c')\n    plt2.plot(np.arange(local_epochs, step=1) ,history.history['val_loss'],'-o',label='Val Loss',color='#d62728')\n    plt.legend(loc=3)\n    plt.ylabel('Loss',size=14)\n    plt.title(\"Model Accuracy and loss\")\n    \n    plt.savefig('loss.png')\n    plt.show()\n    \nplot_hist(history)","metadata":{"execution":{"iopub.status.busy":"2022-03-17T17:02:03.664077Z","iopub.status.idle":"2022-03-17T17:02:03.666442Z","shell.execute_reply.started":"2022-03-17T17:02:03.666161Z","shell.execute_reply":"2022-03-17T17:02:03.666192Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Evaluation on valid set","metadata":{}},{"cell_type":"code","source":"model.load_weights('./best_model.hdf5')\npred_valid_y = model.predict(valid_ds_batch, workers=4, verbose = True)\npred_valid_y_labels = np.argmax(pred_valid_y, axis=-1)\n\nvalid_labels = np.concatenate([y.numpy() for x, y in valid_ds_batch], axis=0)\nprint(classification_report(valid_labels, pred_valid_y_labels ))","metadata":{"execution":{"iopub.status.busy":"2022-03-15T07:22:38.100998Z","iopub.execute_input":"2022-03-15T07:22:38.101296Z","iopub.status.idle":"2022-03-15T07:22:44.354773Z","shell.execute_reply.started":"2022-03-15T07:22:38.101262Z","shell.execute_reply":"2022-03-15T07:22:44.353235Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Prediction on test set","metadata":{}},{"cell_type":"code","source":"test_ds = tf.data.Dataset.from_tensor_slices((df_test_spec.ID.values, df_test_spec.spec_path.values))\n\n\ndef process_test(id, image_path):\n    img = tf.io.read_file(image_path)\n    img = tf.image.decode_jpeg(img, channels=3)\n    \n    img = tf.image.resize(img, size=[config['img_size'], config['img_size']])\n    return id, img\n    \ntest_ds = test_ds.map(process_test, num_parallel_calls=AUTOTUNE).batch(config['batch_size']*2)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T05:52:24.676852Z","iopub.execute_input":"2022-03-15T05:52:24.678866Z","iopub.status.idle":"2022-03-15T05:52:24.750633Z","shell.execute_reply.started":"2022-03-15T05:52:24.678829Z","shell.execute_reply":"2022-03-15T05:52:24.749947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"preds = []\nfor id, imgl in tqdm(test_ds):\n    local_preds = model.predict(imgl, workers=4)\n    \n    for idx, local_id in enumerate(id.numpy()):\n        preds.append({'song_id':local_id.decode(\"utf-8\") , 'genre_id': np.argmax(local_preds[idx], axis=-1)})","metadata":{"execution":{"iopub.status.busy":"2022-03-15T06:16:36.241592Z","iopub.execute_input":"2022-03-15T06:16:36.241874Z","iopub.status.idle":"2022-03-15T06:17:58.197517Z","shell.execute_reply.started":"2022-03-15T06:16:36.241833Z","shell.execute_reply":"2022-03-15T06:17:58.19685Z"},"collapsed":true,"jupyter":{"outputs_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Adding missing values","metadata":{}},{"cell_type":"code","source":"preds.append({'song_id':'022612', 'genre_id':1})\npreds.append({'song_id':'024013', 'genre_id':0})\n\ntest_df = pd.DataFrame.from_dict(preds)\ntest_df['song_id'] = test_df['song_id'].astype('int')\n\nassert(len(test_df) == 5078)\ntest_df.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-15T06:23:07.070663Z","iopub.execute_input":"2022-03-15T06:23:07.071078Z","iopub.status.idle":"2022-03-15T06:23:07.091557Z","shell.execute_reply.started":"2022-03-15T06:23:07.071047Z","shell.execute_reply":"2022-03-15T06:23:07.090744Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Notes:\n\n1. You can try and train the model of your choice.\n2. Feel free to change hyperparameters in the config section\n\n\n**Please do upvote if this was useful for you**","metadata":{}}]}