{"cells":[{"metadata":{"trusted":true,"_uuid":"da47cf37ddf9463ee0320709018e229d1f2fa76c"},"cell_type":"code","source":"batch_size = 64\nepochs = 150","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"d42696d24681de355e32c2964ce1b748b0855810"},"cell_type":"code","source":"%matplotlib inline  \nimport numpy as np\nimport pandas as pd\nfrom keras import optimizers, losses, activations, models\nfrom keras.callbacks import ModelCheckpoint, EarlyStopping, LearningRateScheduler\nfrom keras import layers\nimport librosa\nimport numpy as np\nimport glob\nimport os\nimport pandas as pd\nfrom PIL import Image\nimport numpy as np\nimport matplotlib.pyplot as plt\nfrom random import shuffle\nfrom sklearn.model_selection import train_test_split\nfrom tqdm import tqdm\nimport h5py\naudio_dir = '../input/freesound-audio-tagging/'\nfeature_dir = '../input/spectrogram-data-preparation/'","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"aff02f930ecbc461353c67a9032360b0ed4d9119"},"cell_type":"code","source":"train_labels = pd.read_csv(os.path.join(audio_dir, \"train.csv\"))\nprint(len(train_labels), 'training')\ntrain_labels.groupby(['label']).size().plot.bar()\ntrain_labels.sample(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"ac426a0c2779ff103228913c36133cce2575885e"},"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfull_train_ds_pointer = h5py.File(os.path.join(feature_dir, 'spectrograms.h5'))\nfor k in full_train_ds_pointer.keys():\n    print(k, full_train_ds_pointer[k].shape)\ninput_length = full_train_ds_pointer['spectrograms'].shape[1]\nN_MEL_COUNT = full_train_ds_pointer['spectrograms'].shape[2]\nlab_enc = LabelEncoder()\nall_labels = lab_enc.fit_transform(full_train_ds_pointer['label'].value)\nnclass = len(lab_enc.classes_)","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"579d26030d99b62d86e013ad7df195c028ab9709"},"cell_type":"markdown","source":"# Show a few example spectrograms\nWe can see what sort of patterns the model should be looking for"},{"metadata":{"trusted":true,"_uuid":"399ddfdc31f46f31ec186c7fbd2c170fd38aea57"},"cell_type":"code","source":"fig, m_axs = plt.subplots(15, 8, figsize = (10, 30))\nfor c_class, c_axs in zip(np.random.permutation(range(nclass)), m_axs):\n    idxs = np.where(all_labels==c_class)[0]\n    c_axs[0].set_title(lab_enc.classes_[c_class].decode())\n    c_axs[0].set_xlabel('Time')\n    c_axs[1].set_ylabel('Frequency')\n    for c_idx, c_ax in zip(np.random.permutation(idxs), c_axs):\n        c_ax.axis('off')\n        c_ax.imshow(full_train_ds_pointer['spectrograms'][c_idx][:, :, 0].swapaxes(0, 1))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"2cc5af81124e9f4802ebed0c2722a59ce8041d89"},"cell_type":"code","source":"from sklearn.model_selection import train_test_split\ntrain_X, valid_X, train_y, valid_y = train_test_split(\n    full_train_ds_pointer['spectrograms'].value,\n    all_labels,\n    test_size = 0.2, \n    random_state = 2018)\nprint('Training', train_X.shape, train_y.shape)\nprint('Validation', valid_X.shape, valid_y.shape)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"e2d724618d1f5cabfeb86298ba69972bd997432f"},"cell_type":"markdown","source":"## Setup the Model"},{"metadata":{"trusted":true,"_uuid":"c11230c47d8062447adc0eaad660c540fd6bbc98"},"cell_type":"code","source":"from keras import layers, metrics, applications\n# this is what the results are scored on, so we should keep this\ndef top_3_accuracy(x, y): \n    return metrics.sparse_top_k_categorical_accuracy(x,y,3)\ndef create_model():\n    model = applications.mobilenet.MobileNet(input_shape=train_X.shape[1:],\n                                    classes=nclass,\n                                    weights=None)\n    opt = optimizers.Adam(lr=4e-4)\n    \n    model.compile(optimizer=opt, \n                  loss=losses.sparse_categorical_crossentropy, \n                  metrics=[metrics.sparse_categorical_accuracy,\n                           top_3_accuracy])\n    model.summary()\n    return model\nmodel = create_model()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"b130ece37b2afb4950e22e1ab9cf030f3a578d66"},"cell_type":"code","source":"from keras.callbacks import ModelCheckpoint, ReduceLROnPlateau, EarlyStopping\nweight_path=\"{}_weights.best.hdf5\".format('spectro_sound_model')\ncheckpoint = ModelCheckpoint(weight_path, monitor='val_loss', verbose=1, \n                             save_best_only=True, mode='min', save_weights_only = True)\nreduceLROnPlat = ReduceLROnPlateau(monitor='val_loss', \n                                   factor=0.8, patience=5, \n                                   verbose=1, mode='auto', \n                                   epsilon=0.0001, cooldown=5, \n                                   min_lr=0.0001)\nearly = EarlyStopping(monitor=\"val_loss\", \n                      mode=\"min\", \n                      patience=15) # probably needs to be more patient, but kaggle time is limited\ncallbacks_list = [checkpoint, early, reduceLROnPlat]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"df0d6f9ed3d667c83ffbf9665023b36e3d26ede4"},"cell_type":"code","source":"from IPython.display import clear_output\nfit_results = model.fit(train_X, train_y,\n          epochs=epochs,\n          batch_size=batch_size,\n          validation_data=(valid_X, valid_y), \n          callbacks=callbacks_list)\nclear_output()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"05955ac654405b440e440d18eee6b6621e0a2405"},"cell_type":"code","source":"fig, (ax1, ax2, ax3) = plt.subplots(1, 3, figsize = (20, 10))\nax1.plot(fit_results.history['loss'], label='Training')\nax1.plot(fit_results.history['val_loss'], label='Validation')\nax1.legend()\nax1.set_title('Loss')\nax2.plot(fit_results.history['sparse_categorical_accuracy'], label='Training')\nax2.plot(fit_results.history['val_sparse_categorical_accuracy'], label='Validation')\nax2.legend()\nax2.set_title('Top 1 Accuracy')\nax2.set_ylim(0, 1)\nax3.plot(fit_results.history['top_3_accuracy'], label='Training')\nax3.plot(fit_results.history['val_top_3_accuracy'], label='Validation')\nax3.legend()\nax3.set_title('Top 3 Accuracy')\nax3.set_ylim(0, 1);","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"5daf147daa76f464dd1a4084fbb17d8d2b3604c9"},"cell_type":"code","source":"model.load_weights(weight_path)\nfor k, v in zip(model.metrics_names, \n        model.evaluate(valid_X, valid_y)):\n    if k!='loss':\n        print('{:40s}:\\t{:2.1f}%'.format(k, 100*v))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"32e8642a50854e907ee36111fdd72eb0ecd07033"},"cell_type":"code","source":"model.save(\"spectro_baseline_cnn.h5\")","execution_count":null,"outputs":[]},{"metadata":{"collapsed":true,"trusted":false,"_uuid":"71787d953933838626f108c92d33d03f3a75d1d5"},"cell_type":"markdown","source":"# Run Predictions on Test Data"},{"metadata":{"trusted":true,"_uuid":"9bfc5f2e1d6aea9597f42263453ec21a1d6ca62d"},"cell_type":"code","source":"full_test_ds_pointer = h5py.File(os.path.join(feature_dir, 'test_spectrograms.h5'))\nfor k in full_test_ds_pointer.keys():\n    print(k, full_test_ds_pointer[k].shape)\ntest_files = [x.decode() for x in full_test_ds_pointer['fname'].value]\ntest_preds = model.predict(full_test_ds_pointer['spectrograms'].value, \n                           batch_size=batch_size, verbose=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"bc0d0ac8010a457e99bccdfa66a8f008c6c98701"},"cell_type":"code","source":"top_3 = lab_enc.classes_[np.argsort(-test_preds, axis=1)[:, :3]] #https://www.kaggle.com/inversion/freesound-starter-kernel\npred_labels = [' '.join([cat.decode() for cat in row]) for row in top_3]\npred_labels[0:2]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"f8b5c9594cf2555a3633adf341951ab96e0d5ec6"},"cell_type":"code","source":"df = pd.DataFrame(test_files, columns=[\"fname\"])\ndf['label'] = pred_labels\ndf['fname'] = df.fname.apply(lambda x: x.split(\"/\")[-1])\ndf.sample(3)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"0d67a88f61bcdca1891bc571e48c0d97e95d089e"},"cell_type":"code","source":"df.to_csv(\"baseline.csv\", index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true,"_uuid":"1964381b66d2d5d35c9b328c3f17ed23f702ae57"},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"name":"python","version":"3.6.6","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat":4,"nbformat_minor":1}