{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport matplotlib.pyplot as plt\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-31T16:33:52.569491Z","iopub.execute_input":"2022-03-31T16:33:52.570087Z","iopub.status.idle":"2022-03-31T16:33:52.574851Z","shell.execute_reply.started":"2022-03-31T16:33:52.570042Z","shell.execute_reply":"2022-03-31T16:33:52.574167Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nfrom IPython.display import Audio\nimport time\nimport torchaudio\nimport torch\n\npath = \"../input/kaggle-pog-series-s01e02/train\"\ntest_path = \"../input/kaggle-pog-series-s01e02/test\"\n\ndf = pd.read_csv(path+\".csv\")\ntest_df = pd.read_csv(test_path+\".csv\")\n\nprint(df.shape)\ndef load_data(path, df,savefile=False,visualize=False,test=False):\n    df[\"file_exists\"] = True\n    data = list()\n    directory = path\n    print(path)\n    transformer = torchaudio.transforms.MFCC( sample_rate=22050, n_mfcc=20,melkwargs={\"n_fft\": 512, \"hop_length\": 256})\n    #transformer = torchaudio.transforms.MelSpectrogram(sample_rate=22050,n_fft=2048, hop_length=512,normalized =True)\n    #powertodb = torchaudio.transforms.AmplitudeToDB()\n    #display(df)\n    #w_rate, w_data = read(path + \"/_background_noise_/white_noise.wav\")\n    size = 32000\n    duration = 10\n    for i,item in enumerate(df.itertuples()):\n        try:\n            f = item.filename\n            file = path+\"/\"+str(f)\n            \n            r_data, samplingrate = librosa.load(path+\"/\"+f, duration=duration)\n            logmel = librosa.feature.mfcc(y=r_data, n_fft=1024, hop_length=256, n_mels=80,n_mfcc=20)\n            #logmel = librosa.power_to_db(logmel, ref=np.max)\n            logmel = librosa.util.normalize(logmel)\n            #x_test = librosa.util.normalize(x_test)\n            \n            #r_data, samplingrate = torchaudio.load(path+\"/\"+f)\n            #r_data = torch.mean(r_data, dim=0).unsqueeze(0)\n            #logmel = transformer(r_data[:20000])#.flatten()[:size])  \n            \n            size=r_data.shape[0]\n            if visualize:\n                if i%5==1:\n                    r_data1,_ = librosa.effects.trim(r_data,top_db=5)\n                    print(r_data1.shape,r_data.shape)\n                    plt.plot(range(len(r_data)),r_data)\n                    plt.show()\n                    plt.pcolormesh(logmel)\n                    plt.colorbar()\n                    plt.show()\n                    display(Audio(r_data, rate=samplingrate))\n                    display(Audio(r_data1, rate=samplingrate))\n                          \n            if not test:\n                genre = item.genre\n            else:\n                genre = -1\n                \n            data.append({\n                \"filename\": f,\n                #\"samplingrate\": samplingrate,\n                \"genre\": genre,\n                #\"samples\": narray,\n                \"mfcc\": logmel\n                #\"lmfcc\": lmfcc\n            })\n            if i%1000 == 999:\n                print(i, genre)\n                \n        except FileNotFoundError:\n            print(\"FileNotFoundError\" + item.filename)\n            df.loc[df.filename == item.filename, 'file_exists'] = False\n            continue\n        except RuntimeError:\n            print(\"FileNotFoundError\" + item.filename)\n            df.loc[df.filename == item.filename, 'file_exists'] = False\n            continue\n        except ValueError:\n            print(\"[ValueError] Could not save:\" + item.filename)\n            continue\n\n    if savefile:\n        np.savez('melspectrograms.npz', data=data)\n\n    return data\n\"\"\"start = time.perf_counter()\ndata = load_data(path,savefile=True)\nend = time.perf_counter()\nprint((end-start)/60)\"\"\"","metadata":{"execution":{"iopub.status.busy":"2022-03-31T16:33:52.580174Z","iopub.execute_input":"2022-03-31T16:33:52.580781Z","iopub.status.idle":"2022-03-31T16:33:54.977393Z","shell.execute_reply.started":"2022-03-31T16:33:52.580747Z","shell.execute_reply":"2022-03-31T16:33:54.976654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os.path\nimport numpy as np\n#exists = os.path.exists('dataset/TTS-lmfcc.npz')\nload = True\ndata = None\ntest_data = None\nstart = time.perf_counter()\nif load:\n    print(\"reading files...\")\n    data = np.load('../input/melspectrograms-train-test/melspectrograms.npz', allow_pickle=True)['data']\n    test_data = np.load('../input/melspectrograms-train-test/melspectrograms_test.npz', allow_pickle=True)['data']\n\n    print(\"lmfcc's loaded!\")\nelse:\n    print(\"Loading data from wav files...\")\n    #data = load_data(path,df,savefile=False)\n    test_data = load_data(test_path,test_df, savefile=False,test=True)\n    np.savez('melspectrograms_test.npz', data=test_data)\n    #np.savez('melspectrograms.npz', data=data)\n\n    print(\"lmfcc data saved to: dataset/TTS-lmfcc.npz\")\nend = time.perf_counter()\nprint(end-start)","metadata":{"execution":{"iopub.status.busy":"2022-03-31T16:33:54.979295Z","iopub.execute_input":"2022-03-31T16:33:54.979720Z","iopub.status.idle":"2022-03-31T16:34:13.437125Z","shell.execute_reply.started":"2022-03-31T16:33:54.979684Z","shell.execute_reply":"2022-03-31T16:34:13.436268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sys import getsizeof\n\nprint(getsizeof(data))\nx_train, y_train = [], []\nx_test = []\nshape = data[0]['mfcc'].T.shape\n#x_train = np.zeros((len(data),data[0]['mfcc'].shape[1],data[0]['mfcc'].shape[0]))\nz_train = []\nz_test = []\nfor i,d in enumerate(data):\n    if d['mfcc'].T.shape == shape:\n        x_train.append(d['mfcc'].T)\n        z_train.append(d['filename'])\n        y_train.append(df.loc[d['genre']==df.genre]['genre_id'].values[0])\n        \n        \nfor i,d in enumerate(test_data):\n    if d['mfcc'].T.shape == shape:\n        x_test.append(d['mfcc'].T)\n        z_test.append(d['filename'])\n        \nx_train = np.array(x_train)\nx_test = np.array(x_test)\n\nprint(x_train.shape)\nprint(x_test.shape)\n","metadata":{"execution":{"iopub.status.busy":"2022-03-31T16:34:13.438525Z","iopub.execute_input":"2022-03-31T16:34:13.438789Z","iopub.status.idle":"2022-03-31T16:35:18.930276Z","shell.execute_reply.started":"2022-03-31T16:34:13.438753Z","shell.execute_reply":"2022-03-31T16:35:18.929567Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Imports:\nimport tensorflow as tf\nimport numpy as np\nimport os\nimport sklearn\nimport matplotlib.pyplot as plt\nfrom tensorflow import keras as keras\n# TODO: do we need maxpooling2D and Upsampling2D?\nfrom tensorflow.keras.layers import Input, Dense, Conv1D,Conv2DTranspose,Conv2D, MaxPooling2D, UpSampling2D, Flatten, GlobalAveragePooling2D,Dense, Reshape, Lambda,BatchNormalization,ReLU,MaxPool2D,Dropout,AveragePooling2D,GlobalAveragePooling2D\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.callbacks import TensorBoard, ReduceLROnPlateau, EarlyStopping\nfrom tensorflow.keras import backend as K\nfrom tensorflow.keras.datasets import mnist\nfrom sklearn.preprocessing import OneHotEncoder\nfrom tensorflow.keras.layers import Concatenate\nfrom tensorflow.keras.optimizers import Adam, SGD\nfrom tensorflow.keras.losses import CategoricalCrossentropy\nfrom tensorflow.keras.metrics import AUC","metadata":{"execution":{"iopub.status.busy":"2022-03-31T16:35:18.932182Z","iopub.execute_input":"2022-03-31T16:35:18.932923Z","iopub.status.idle":"2022-03-31T16:35:24.158873Z","shell.execute_reply.started":"2022-03-31T16:35:18.932881Z","shell.execute_reply":"2022-03-31T16:35:24.158124Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\nfrom sklearn.preprocessing import MinMaxScaler\nprint(np.max(x_train))\nprint(np.min(x_train))\nprint(x_train.shape)\nplt.pcolormesh(x_train[0])\nplt.colorbar()\n\nplt.show()\n\"\"\"x_train/=80\nx_train*=-1\nx_test/=80\nx_test*=-1\"\"\"\n#scaler = MinMaxScaler()\n#x_train = librosa.util.normalize(x_train)\n#x_test = librosa.util.normalize(x_test)\n\nprint(np.max(x_train))\nprint(np.min(x_train))\nplt.pcolormesh(x_train[0])\nplt.colorbar()\n","metadata":{"execution":{"iopub.status.busy":"2022-03-31T16:35:24.160055Z","iopub.execute_input":"2022-03-31T16:35:24.160987Z","iopub.status.idle":"2022-03-31T16:35:25.317158Z","shell.execute_reply.started":"2022-03-31T16:35:24.160955Z","shell.execute_reply":"2022-03-31T16:35:25.316492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"oh_encoder = OneHotEncoder(categories='auto', sparse=False)\ny_train_oh = oh_encoder.fit_transform(np.array(y_train).reshape(-1, 1))\nprint(np.count_nonzero(y_train_oh,axis=0))\nprint(np.unique(y_train, return_counts=True))\n\n# Parameters:\nn_batch = 256\nn_classes = y_train_oh.shape[1]\nn_latent = 25\nn_channels = 1\nn_epoch = 50","metadata":{"execution":{"iopub.status.busy":"2022-03-31T16:35:25.321060Z","iopub.execute_input":"2022-03-31T16:35:25.322979Z","iopub.status.idle":"2022-03-31T16:35:25.350203Z","shell.execute_reply.started":"2022-03-31T16:35:25.322942Z","shell.execute_reply":"2022-03-31T16:35:25.349497Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Source: https://www.kaggle.com/ashishpatel26/best-trick-for-audio-data","metadata":{}},{"cell_type":"code","source":"from keras.applications.vgg16 import VGG16\nfrom tensorflow.keras.applications import ResNet50\n\n# load model\nx_train = x_train.reshape(-1, x_train.shape[1], x_train.shape[2], 1)\nx_test = x_test.reshape(-1, x_test.shape[1], x_test.shape[2], 1)\n\ninput_sound = Input(shape=x_train.shape[1:])\ninput_cond = Input(shape=(n_classes,))\nkernel_size = 3\nstride_size = (2,2)\nfilters = 32\nlearning_rate = 0.01\n\n\ndef get_2d_conv_model(n_epoch=50,LR=2e-3):\n    \n    nclass = n_classes\n    \n    inp = Input(shape=x_train.shape[1:])\n    x = Conv2D(16, (3,3), padding=\"same\")(inp)\n    x = BatchNormalization()(x)\n    x = ReLU()(x)\n    x = Dropout(0.25)(x)\n    \n    x = Conv2D(32, (3,3), padding=\"same\")(x)\n    x = BatchNormalization()(x)\n    x = ReLU()(x)\n    x = Dropout(0.25)(x)\n\n    \n    x = Conv2D(64, (3,3), padding=\"same\")(x)\n    x = BatchNormalization()(x)\n    x = ReLU()(x)\n    x = Dropout(0.25)(x)\n    \n    x = Conv2D(128, (3,3), padding=\"same\")(x)\n    x = BatchNormalization()(x)\n    x = ReLU()(x)\n    x = Dropout(0.25)(x)\n    \n    x = GlobalAveragePooling2D()(x)\n    x = Dense(64)(x)\n    x = ReLU()(x)\n    out = Dense(n_classes, activation='softmax')(x)\n\n    model = Model(inputs=inp, outputs=out)\n    opt = Adam(LR)\n\n    model.compile(optimizer=opt, loss=CategoricalCrossentropy(), metrics=[AUC()])\n    return model\n\ndef get_resnet(n_epoch=50,LR=1e-1):\n    input_tensor = Input(shape=x_train.shape[1:])\n    x = Conv2D(3,(3,3),padding='same')(input_tensor)    # x has a dimension of (IMG_SIZE,IMG_SIZE,3)\n    baseModel = ResNet50(weights=None,\n                         include_top=False,\n                         input_shape=(x_train.shape[1],x_train.shape[2],3))\n    # construct the head of the model that will be placed on top of the\n    # the base model\n    headModel = baseModel(x,training=True)\n    headModel = GlobalAveragePooling2D()(headModel)\n    headModel = Flatten(name=\"flatten\")(headModel)\n    headModel = Dense(256, activation=\"relu\")(headModel)\n    headModel = Dropout(0.5)(headModel)\n    headModel = Dense(n_classes, activation=\"softmax\")(headModel)\n    # place the head FC model on top of the base model (this will become\n    # the actual model we will train)\n    model = Model(inputs=input_tensor, outputs=headModel)\n    # loop over all layers in the base model and freeze them so they will\n    # *not* be updated during the training process\n    for layer in baseModel.layers:\n        layer.trainable = False\n    lr_schedule = tf.keras.optimizers.schedules.ExponentialDecay(\n        LR,\n        decay_steps=100000,\n        decay_rate=0.92,\n        staircase=True)\n    opt = SGD(lr_schedule,momentum=0.96)\n    model.compile(optimizer=opt, loss=CategoricalCrossentropy(), metrics=[AUC()])\n    return model","metadata":{"execution":{"iopub.status.busy":"2022-03-31T16:35:25.354099Z","iopub.execute_input":"2022-03-31T16:35:25.356064Z","iopub.status.idle":"2022-03-31T16:35:25.428888Z","shell.execute_reply.started":"2022-03-31T16:35:25.356026Z","shell.execute_reply":"2022-03-31T16:35:25.428078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from keras.callbacks import History \nn_epoch = 200\nn_batch = 16\n\nmodel = get_2d_conv_model(n_epoch)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2022-03-31T16:35:25.433763Z","iopub.execute_input":"2022-03-31T16:35:25.435772Z","iopub.status.idle":"2022-03-31T16:35:30.816952Z","shell.execute_reply.started":"2022-03-31T16:35:25.435731Z","shell.execute_reply":"2022-03-31T16:35:30.816276Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history =  model.fit(x_train, y_train_oh,shuffle=True, epochs=n_epoch,batch_size=n_batch, callbacks=[EarlyStopping(patience=10)])","metadata":{"execution":{"iopub.status.busy":"2022-03-31T16:35:30.818273Z","iopub.execute_input":"2022-03-31T16:35:30.818710Z","iopub.status.idle":"2022-03-31T21:20:56.456575Z","shell.execute_reply.started":"2022-03-31T16:35:30.818670Z","shell.execute_reply":"2022-03-31T21:20:56.455840Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.title(\"Reconstruction Loss\")\nepochs=len(history.history['loss'])\n#plt.plot(np.arange(0,epochs),history.history['val_loss'],label=\"Reconstruction Loss on Validation Data\", marker='^',color='red')\nplt.plot(np.arange(0,epochs),history.history['loss'],label=\"Reconstruction Loss on Train Data\", marker='*',color='green')\n\nplt.xlabel(\"Epochs\")\nplt.ylabel(\"Reconstruction Loss\")\nplt.xticks(np.arange(0,epochs,5),np.arange(0,epochs,5))\nplt.legend()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-31T21:20:56.459247Z","iopub.execute_input":"2022-03-31T21:20:56.459528Z","iopub.status.idle":"2022-03-31T21:20:56.730053Z","shell.execute_reply.started":"2022-03-31T21:20:56.459490Z","shell.execute_reply":"2022-03-31T21:20:56.729398Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import seaborn as sns\n\n\ny_pred = model.predict(x_test)\nsubm = []\ntemp_subm = []\nfor idx, i in enumerate(test_df.itertuples()):\n    temp_subm.append([0,i.song_id])\n    #print(test_df.loc[test_df.filename == z_test[i]].song_id.values[0])\nresults_df = pd.DataFrame(temp_subm, columns=['genre_id','song_id'])\n\nfor i,_ in enumerate(z_test):\n    #print(test_df.loc[test_df.filename == z_test[i]].song_id.values[0])\n    results_df.loc[test_df.loc[test_df.filename == z_test[i]].song_id.values[0]==results_df.song_id,'genre_id'] = np.argmax(y_pred[i])\n    \ndisplay(results_df)\nax = sns.countplot(x = results_df['genre_id'])\nplt.show()\nax = sns.countplot(x = df['genre_id'])\nplt.show()\nresults_df.to_csv(\"submission.csv\", index=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-31T21:20:56.731224Z","iopub.execute_input":"2022-03-31T21:20:56.731945Z","iopub.status.idle":"2022-03-31T21:21:16.134138Z","shell.execute_reply.started":"2022-03-31T21:20:56.731908Z","shell.execute_reply":"2022-03-31T21:21:16.133414Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#1.3934 ","metadata":{"execution":{"iopub.status.busy":"2022-03-31T21:21:16.135442Z","iopub.execute_input":"2022-03-31T21:21:16.135780Z","iopub.status.idle":"2022-03-31T21:21:16.140160Z","shell.execute_reply.started":"2022-03-31T21:21:16.135741Z","shell.execute_reply":"2022-03-31T21:21:16.139307Z"},"trusted":true},"execution_count":null,"outputs":[]}]}