{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"%config Completer.use_jedi = False","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2021-05-24T15:54:43.822521Z","iopub.execute_input":"2021-05-24T15:54:43.822923Z","iopub.status.idle":"2021-05-24T15:54:43.841526Z","shell.execute_reply.started":"2021-05-24T15:54:43.822807Z","shell.execute_reply":"2021-05-24T15:54:43.840797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Importing standard libraries\nimport os\nimport time\nimport numpy as np\nimport pandas as pd\nimport plotly.express as px\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom pprint import pprint\nfrom glob import glob\nfrom plotly.offline import init_notebook_mode, iplot\ninit_notebook_mode(connected=True)\n\n# Importing Libraries for Audio file reading\nimport soundfile as sf\nimport librosa\nimport librosa.display\nimport IPython.display as display\n\n# Importing Libraries to build the neural network\nimport tensorflow as tf\nfrom tensorflow.keras.utils import Sequence, plot_model\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras.optimizers import *\nfrom tensorflow.keras.callbacks import *\n\n# For data preparation and model evaluation\nfrom sklearn.metrics import f1_score, accuracy_score, precision_score, recall_score, confusion_matrix\nfrom sklearn.model_selection import train_test_split\n\n\n\nimport warnings\nwarnings.filterwarnings(\"ignore\")","metadata":{"execution":{"iopub.status.busy":"2021-05-24T16:47:42.051777Z","iopub.execute_input":"2021-05-24T16:47:42.052142Z","iopub.status.idle":"2021-05-24T16:47:42.060735Z","shell.execute_reply.started":"2021-05-24T16:47:42.052110Z","shell.execute_reply":"2021-05-24T16:47:42.059815Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Helper functions\ndef read_wav_file(path):\n    return sf.read(path)\n\ndef displayWaveform(data, sr):\n    plt.figure(figsize = (14, 5))\n    librosa.display.waveplot(data, sr = sr)\n    plt.grid()\n    plt.show()\n\ndef plot_spectrogram(data, sr):\n    spectrogram = librosa.feature.melspectrogram(data, sr)\n    log_spec = librosa.power_to_db(spectrogram, ref = np.max)\n    librosa.display.specshow(log_spectrogram, sr = sr, x_axis = 'time', y_axis = 'mel')\n    ","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:54:52.890318Z","iopub.execute_input":"2021-05-24T15:54:52.890603Z","iopub.status.idle":"2021-05-24T15:54:52.900275Z","shell.execute_reply.started":"2021-05-24T15:54:52.890575Z","shell.execute_reply":"2021-05-24T15:54:52.899460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# CONFIG\nparams = {}\nparams['train_csv'] = '../input/birdclef-2021/train_soundscape_labels.csv'\nparams['test_csv'] = '../input/birdclef-2021/test.csv'\nparams['train_metadata'] = '../input/birdclef-2021/train_metadata.csv'\nparams['train_short_audio'] = '../input/birdclef-2021/train_short_audio'\nparams['train_soundscapes'] = '../input/birdclef-2021/train_soundscapes'\nparams['sample_csv'] = '../input/birdclef-2021/sample_submission.csv'\nparams['labels'] = '../input/birdclef-2021/train_soundscape_labels.csv'\nparams['test_soundscapes'] = \"../input/birdclef-2021/test_soundscapes\"\npprint(params)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:54:52.902449Z","iopub.execute_input":"2021-05-24T15:54:52.902699Z","iopub.status.idle":"2021-05-24T15:54:52.914010Z","shell.execute_reply.started":"2021-05-24T15:54:52.902675Z","shell.execute_reply":"2021-05-24T15:54:52.912144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Reading csv files\ntrain_csv = pd.read_csv(params['train_csv'])\ntest_csv = pd.read_csv(params['test_csv'])\ntrain_meta = pd.read_csv(params['train_metadata'])\nsample_sub = pd.read_csv(params['sample_csv'])\ntrain_labels = pd.read_csv(params['labels'])","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:54:52.915547Z","iopub.execute_input":"2021-05-24T15:54:52.916042Z","iopub.status.idle":"2021-05-24T15:54:53.317372Z","shell.execute_reply.started":"2021-05-24T15:54:52.915982Z","shell.execute_reply":"2021-05-24T15:54:53.316449Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_csv.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:54:53.318546Z","iopub.execute_input":"2021-05-24T15:54:53.318879Z","iopub.status.idle":"2021-05-24T15:54:53.343161Z","shell.execute_reply.started":"2021-05-24T15:54:53.318835Z","shell.execute_reply":"2021-05-24T15:54:53.342250Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels.head(3)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:54:53.344510Z","iopub.execute_input":"2021-05-24T15:54:53.344855Z","iopub.status.idle":"2021-05-24T15:54:53.354631Z","shell.execute_reply.started":"2021-05-24T15:54:53.344820Z","shell.execute_reply":"2021-05-24T15:54:53.353608Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_sub.sample(3)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:54:53.356282Z","iopub.execute_input":"2021-05-24T15:54:53.356730Z","iopub.status.idle":"2021-05-24T15:54:53.369548Z","shell.execute_reply.started":"2021-05-24T15:54:53.356658Z","shell.execute_reply":"2021-05-24T15:54:53.368557Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_meta.sample(3)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:54:53.372926Z","iopub.execute_input":"2021-05-24T15:54:53.373229Z","iopub.status.idle":"2021-05-24T15:54:53.395014Z","shell.execute_reply.started":"2021-05-24T15:54:53.373196Z","shell.execute_reply":"2021-05-24T15:54:53.394263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Len of train data : {len(train_csv)}\")\nprint(f\"Len of test data : {len(test_csv)}\")\nprint(f\"Len of train meta : {len(train_meta)}\")\nprint(f\"Len of train labels : {len(train_labels)}\")","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:54:53.397174Z","iopub.execute_input":"2021-05-24T15:54:53.397518Z","iopub.status.idle":"2021-05-24T15:54:53.404723Z","shell.execute_reply.started":"2021-05-24T15:54:53.397482Z","shell.execute_reply":"2021-05-24T15:54:53.403686Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets hear some voices from soundscapes\nsoundscapes = glob(\"../input/birdclef-2021/train_soundscapes/*.ogg\")\n\ndisplay.Audio(soundscapes[np.random.randint(len(soundscapes))])","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:54:53.406286Z","iopub.execute_input":"2021-05-24T15:54:53.406871Z","iopub.status.idle":"2021-05-24T15:54:53.879696Z","shell.execute_reply.started":"2021-05-24T15:54:53.406823Z","shell.execute_reply":"2021-05-24T15:54:53.876170Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets hear some short sounds\nshort_sounds = glob(\"../input/birdclef-2021/train_short_audio/*/*.ogg\")\n\ndisplay.Audio(short_sounds[np.random.randint(len(short_sounds))])","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:54:53.880938Z","iopub.execute_input":"2021-05-24T15:54:53.881317Z","iopub.status.idle":"2021-05-24T15:55:02.238511Z","shell.execute_reply.started":"2021-05-24T15:54:53.881276Z","shell.execute_reply":"2021-05-24T15:55:02.237721Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA","metadata":{}},{"cell_type":"code","source":"# Distribution of labels\nfig = px.histogram(train_labels, x = 'birds', color = 'birds')\nfig.update_layout(\n    title = 'Distribution of Birds calls/labels',\n    title_x = 0.5\n)\nfig.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-24T16:46:14.050934Z","iopub.execute_input":"2021-05-24T16:46:14.051273Z","iopub.status.idle":"2021-05-24T16:46:14.508986Z","shell.execute_reply.started":"2021-05-24T16:46:14.051243Z","shell.execute_reply":"2021-05-24T16:46:14.505636Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Distribution of audio_ids\ntrain_labels.groupby(by=['audio_id']).count()['birds']","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:03.455664Z","iopub.execute_input":"2021-05-24T15:55:03.456182Z","iopub.status.idle":"2021-05-24T15:55:03.466809Z","shell.execute_reply.started":"2021-05-24T15:55:03.456144Z","shell.execute_reply":"2021-05-24T15:55:03.465766Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_labels","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:03.468346Z","iopub.execute_input":"2021-05-24T15:55:03.468711Z","iopub.status.idle":"2021-05-24T15:55:03.486343Z","shell.execute_reply.started":"2021-05-24T15:55:03.468672Z","shell.execute_reply":"2021-05-24T15:55:03.485620Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# There are labels with multiplt birds list in them\n# Lets list down unique labels first and observe their count\n\nuniq_labels = []\nfor bs in train_labels['birds'].values:\n    uniq_labels += bs.split()\n\n# '-1' for \"no call\"\nprint(f\"Num of unique birds : {len(set(uniq_labels)) - 1}\")\nfig = px.histogram(uniq_labels, color = uniq_labels)\nfig.show();\n\n# Finally storing only unique values\nuniq_labels = list(set(uniq_labels))","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:03.487516Z","iopub.execute_input":"2021-05-24T15:55:03.487890Z","iopub.status.idle":"2021-05-24T15:55:03.708054Z","shell.execute_reply.started":"2021-05-24T15:55:03.487854Z","shell.execute_reply":"2021-05-24T15:55:03.707121Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train_labels = pd.DataFrame(\n    index = train_labels.index,\n    columns = uniq_labels\n)\n\nfor row in train_labels.index:\n    birds = train_labels.loc[row, 'birds'].split()\n    for bird in birds:\n        df_train_labels.loc[row, bird] = 1\n        \ndf_train_labels.fillna(0, inplace = True)\n\ntest_csv['birds'] = 'nocall'\n\ndf_test_labels = pd.DataFrame(index = test_csv.index,\n                              columns = uniq_labels)\nfor row in test_csv.index:\n    birds = test_csv.loc[row, 'birds'].split()\n    for bird in birds:\n        df_test_labels.loc[row, bird] = 1\n\ndf_test_labels.fillna(0, inplace = True)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:03.709472Z","iopub.execute_input":"2021-05-24T15:55:03.709814Z","iopub.status.idle":"2021-05-24T15:55:04.001524Z","shell.execute_reply.started":"2021-05-24T15:55:03.709778Z","shell.execute_reply":"2021-05-24T15:55:04.000607Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Merging the table with the original data\n\ntrain_labels = pd.concat([train_labels, df_train_labels], axis = 1)\ntest_csv = pd.concat([test_csv, df_test_labels], axis = 1)\n","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:04.002834Z","iopub.execute_input":"2021-05-24T15:55:04.003226Z","iopub.status.idle":"2021-05-24T15:55:04.013874Z","shell.execute_reply.started":"2021-05-24T15:55:04.003187Z","shell.execute_reply":"2021-05-24T15:55:04.013079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"eg = \"../input/birdclef-2021/train_soundscapes/10534_SSW_20170429.ogg\"\ndata, sr = read_wav_file(eg)\nprint(len(data))\nprint(data[:5]) # first 5 entries\nprint(sr)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:04.015909Z","iopub.execute_input":"2021-05-24T15:55:04.016230Z","iopub.status.idle":"2021-05-24T15:55:04.611259Z","shell.execute_reply.started":"2021-05-24T15:55:04.016205Z","shell.execute_reply":"2021-05-24T15:55:04.610326Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"audio_id, site, _ = eg.split(\"/\")[-1].split(\"_\")\ntrain_labels[(train_labels['audio_id']==int(audio_id)) & (train_labels['site']==site) & (train_labels['birds']!='nocall')].head(5)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:04.612662Z","iopub.execute_input":"2021-05-24T15:55:04.613213Z","iopub.status.idle":"2021-05-24T15:55:04.648707Z","shell.execute_reply.started":"2021-05-24T15:55:04.613174Z","shell.execute_reply":"2021-05-24T15:55:04.647921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Lets hear this bird for index 1450\nsub_data = data[int(50/5)*160000 : int(55/5)*160000]\n\nplt.figure(figsize=(14, 5))\nlibrosa.display.waveplot(sub_data, sr=sr)\nplt.grid()\nplt.show();\n\ndisplay.Audio(sub_data, rate=sr)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:04.649869Z","iopub.execute_input":"2021-05-24T15:55:04.650214Z","iopub.status.idle":"2021-05-24T15:55:04.877135Z","shell.execute_reply.started":"2021-05-24T15:55:04.650179Z","shell.execute_reply":"2021-05-24T15:55:04.870270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"This voice sounds more like fighter plane crash landing 😂","metadata":{}},{"cell_type":"code","source":"params['data_len'] = 160000\nparams['audio_len'] = 5\nparams['num_labels'] = len(uniq_labels)\nparams['for_training'] = {\n    'bs' : 16,\n    'epochs' : 50,\n}\npprint(params)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:04.878689Z","iopub.execute_input":"2021-05-24T15:55:04.879073Z","iopub.status.idle":"2021-05-24T15:55:04.887717Z","shell.execute_reply.started":"2021-05-24T15:55:04.879031Z","shell.execute_reply":"2021-05-24T15:55:04.886470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Preparing Dataset","metadata":{}},{"cell_type":"code","source":"train_ids, val_ids = train_test_split(\n                            list(train_labels.index),\n                            test_size = 0.3,\n                            random_state = 2021)\ntest_ids = list(sample_sub.index)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:04.889427Z","iopub.execute_input":"2021-05-24T15:55:04.889897Z","iopub.status.idle":"2021-05-24T15:55:04.899972Z","shell.execute_reply.started":"2021-05-24T15:55:04.889861Z","shell.execute_reply":"2021-05-24T15:55:04.899153Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Creating a custom data loader using [sequence](http://https://www.tensorflow.org/api_docs/python/tf/keras/utils/Sequence) class","metadata":{}},{"cell_type":"code","source":"class DataLoader(Sequence):\n    def __init__(self, path : str, list_ids : list, data : \"Dataframe\", batch_size : int) -> \"Data for Training\":\n        self.path = path\n        self.list_IDs = list_ids\n        self.data = data\n        self.batch_size = batch_size\n        self.indexes = np.arange(len(self.list_IDs))\n    \n    def __len__(self):\n        len_ = int(len(self.list_IDs)/self.batch_size)\n        if len_*self.batch_size < len(self.list_IDs):\n            len_ += 1\n        return len_\n    \n    def __getitem__(self, index):\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        list_IDs_temp = [self.list_IDs[k] for k in indexes]\n        X, y = self.__data_generation(list_IDs_temp)\n        X = X.reshape((self.batch_size, 100, 1600//2))\n        return X, y\n    \n    def __data_generation(self, list_IDs_temp):\n        X = np.zeros((self.batch_size, params['data_len']//2))\n        y = np.zeros((self.batch_size, params['num_labels']))\n        for i, ID in enumerate(list_IDs_temp):\n            prefix = str(self.data.loc[ID, 'audio_id']) + '_' + self.data.loc[ID, 'site']\n            file_list = [s for s in os.listdir(self.path) if prefix in s]\n            if len(file_list) == 0:\n                # Dummy for missing test audio files\n                audio_file_fft = np.zeros((params['data_len']//2))\n            else:\n                file = file_list[0]\n                audio_file, audio_sr = read_wav_file(os.path.join(self.path, file))\n                audio_file = audio_file[int((self.data.loc[ID, 'seconds']-5)/params['audio_len'])*params['data_len']:int(self.data.loc[ID, 'seconds']/params['audio_len'])*params['data_len']]\n                audio_file_fft = np.abs(np.fft.fft(audio_file)[: len(audio_file)//2])\n                # scale data\n                audio_file_fft = (audio_file_fft-audio_file_fft.mean())/audio_file_fft.std()\n            X[i, ] = audio_file_fft\n            y[i, ] = self.data.loc[ID, self.data.columns[5:]].values\n        return X, y","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:04.901260Z","iopub.execute_input":"2021-05-24T15:55:04.901624Z","iopub.status.idle":"2021-05-24T15:55:04.915086Z","shell.execute_reply.started":"2021-05-24T15:55:04.901587Z","shell.execute_reply":"2021-05-24T15:55:04.914213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we have our Data Loader ready lets build our train_gen, val_gen and test_gen\ntrain_gen = DataLoader(params['train_soundscapes'],\n                       train_ids,\n                       train_labels,\n                       params['for_training']['bs']\n                      )\n\nval_gen = DataLoader(params['train_soundscapes'],\n                     val_ids,\n                     train_labels,\n                     params['for_training']['bs']\n                      )\n\ntest_gen = DataLoader(params['test_soundscapes'],\n                      test_ids,\n                      test_csv,         \n                      params['for_training']['bs']\n                      )","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:04.916303Z","iopub.execute_input":"2021-05-24T15:55:04.916897Z","iopub.status.idle":"2021-05-24T15:55:04.928221Z","shell.execute_reply.started":"2021-05-24T15:55:04.916847Z","shell.execute_reply":"2021-05-24T15:55:04.927360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Building Model","metadata":{}},{"cell_type":"markdown","source":"* Input_shape : (16, 100, 800)\n* Output_shape : (16, 49)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:08:22.161188Z","iopub.execute_input":"2021-05-24T15:08:22.161732Z","iopub.status.idle":"2021-05-24T15:08:22.16713Z","shell.execute_reply.started":"2021-05-24T15:08:22.161672Z","shell.execute_reply":"2021-05-24T15:08:22.166285Z"}}},{"cell_type":"code","source":"def build_model():\n    model = Sequential()\n    model.add(LSTM(128,input_shape=(100, 800)))\n    model.add(Dropout(0.2))\n    model.add(Dense(128, activation='relu'))\n    model.add(Dense(64, activation='relu'))\n    model.add(Dropout(0.25))\n    model.add(Dense(params['num_labels'], activation='sigmoid'))\n    \n    return model","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:04.929448Z","iopub.execute_input":"2021-05-24T15:55:04.929858Z","iopub.status.idle":"2021-05-24T15:55:04.940857Z","shell.execute_reply.started":"2021-05-24T15:55:04.929820Z","shell.execute_reply":"2021-05-24T15:55:04.940097Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"clf = build_model()\nclf.summary()","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:04.944324Z","iopub.execute_input":"2021-05-24T15:55:04.944689Z","iopub.status.idle":"2021-05-24T15:55:07.520315Z","shell.execute_reply.started":"2021-05-24T15:55:04.944654Z","shell.execute_reply":"2021-05-24T15:55:07.519546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"tf.keras.utils.plot_model(clf, show_layer_names = True, show_shapes = True)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:55:07.521920Z","iopub.execute_input":"2021-05-24T15:55:07.522454Z","iopub.status.idle":"2021-05-24T15:55:07.926717Z","shell.execute_reply.started":"2021-05-24T15:55:07.522416Z","shell.execute_reply":"2021-05-24T15:55:07.925843Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Compiling model \nclf.compile(optimizer = 'adam',\n            loss = 'binary_crossentropy',\n            metrics = ['binary_accuracy', 'accuracy', 'AUC']\n           )","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:57:10.700918Z","iopub.execute_input":"2021-05-24T15:57:10.701292Z","iopub.status.idle":"2021-05-24T15:57:10.715750Z","shell.execute_reply.started":"2021-05-24T15:57:10.701264Z","shell.execute_reply":"2021-05-24T15:57:10.714817Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_hist = clf.fit_generator(generator = train_gen,\n                              validation_data = val_gen,\n                              epochs = 2,\n                              workers = 4)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T15:57:11.190714Z","iopub.execute_input":"2021-05-24T15:57:11.191061Z","iopub.status.idle":"2021-05-24T16:42:53.865471Z","shell.execute_reply.started":"2021-05-24T15:57:11.191029Z","shell.execute_reply":"2021-05-24T16:42:53.864568Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_hist.history.keys()","metadata":{"execution":{"iopub.status.busy":"2021-05-24T16:45:27.971001Z","iopub.execute_input":"2021-05-24T16:45:27.971368Z","iopub.status.idle":"2021-05-24T16:45:27.977832Z","shell.execute_reply.started":"2021-05-24T16:45:27.971338Z","shell.execute_reply":"2021-05-24T16:45:27.976670Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"fig, axs = plt.subplots(1, 2, figsize=(16, 4))\nfig.subplots_adjust(hspace = .2, wspace=.2)\naxs = axs.ravel()\nloss = train_hist.history['loss']\nloss_val = train_hist.history['val_loss']\nepochs = range(1, len(loss)+1)\naxs[0].plot(epochs, loss, 'b', label='loss_train')\naxs[0].plot(epochs, loss_val, 'r', label='loss_val')\naxs[0].set_title('Value of the loss function')\naxs[0].set_xlabel('epochs')\naxs[0].set_ylabel('value of the loss function')\naxs[0].legend()\naxs[0].grid()\nacc = train_hist.history['auc']\nacc_val = train_hist.history['val_auc']\naxs[1].plot(epochs, acc, 'b', label='AUC_train')\naxs[1].plot(epochs, acc_val, 'r', label='AUC_val')\naxs[1].set_title('Accuracy')\naxs[1].set_xlabel('Epochs')\naxs[1].set_ylabel('Value of accuracy')\naxs[1].legend()\naxs[1].grid()\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2021-05-24T16:45:37.951933Z","iopub.execute_input":"2021-05-24T16:45:37.952290Z","iopub.status.idle":"2021-05-24T16:45:38.391714Z","shell.execute_reply.started":"2021-05-24T16:45:37.952260Z","shell.execute_reply":"2021-05-24T16:45:38.390799Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"y_pred = clf.predict_generator(test_gen, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T16:53:10.210726Z","iopub.execute_input":"2021-05-24T16:53:10.211060Z","iopub.status.idle":"2021-05-24T16:53:10.595881Z","shell.execute_reply.started":"2021-05-24T16:53:10.211026Z","shell.execute_reply":"2021-05-24T16:53:10.595129Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_test = np.where(y_pred > 0.5, 1, 0)\nfor row in sample_sub.index:\n    string = ''\n    for col in range(len(y_test[row])):\n        if y_test[row][col] == 1:\n            if string == '':\n                string += uniq_labels[col]\n            else:\n                string += ' ' + uniq_labels[col]\n    if string == '':\n        string = 'nocall'\n    sample_sub.loc[row, 'birds'] = string","metadata":{"execution":{"iopub.status.busy":"2021-05-24T16:56:46.131616Z","iopub.execute_input":"2021-05-24T16:56:46.131932Z","iopub.status.idle":"2021-05-24T16:56:46.138040Z","shell.execute_reply.started":"2021-05-24T16:56:46.131901Z","shell.execute_reply":"2021-05-24T16:56:46.137213Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Saving submission\n\nres = sample_sub\nres.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2021-05-24T16:57:34.690626Z","iopub.execute_input":"2021-05-24T16:57:34.690973Z","iopub.status.idle":"2021-05-24T16:57:34.706012Z","shell.execute_reply.started":"2021-05-24T16:57:34.690942Z","shell.execute_reply":"2021-05-24T16:57:34.705280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Saving model weights","metadata":{}},{"cell_type":"code","source":"!mkdir ./baseline\nclf.save(\"./baseline/baseline.h5\")","metadata":{"execution":{"iopub.status.busy":"2021-05-24T17:01:28.610886Z","iopub.execute_input":"2021-05-24T17:01:28.611255Z","iopub.status.idle":"2021-05-24T17:01:29.340008Z","shell.execute_reply.started":"2021-05-24T17:01:28.611222Z","shell.execute_reply":"2021-05-24T17:01:29.339033Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install kaggle","metadata":{"execution":{"iopub.status.busy":"2021-05-24T17:03:41.406910Z","iopub.execute_input":"2021-05-24T17:03:41.407270Z","iopub.status.idle":"2021-05-24T17:03:48.584934Z","shell.execute_reply.started":"2021-05-24T17:03:41.407238Z","shell.execute_reply":"2021-05-24T17:03:48.584010Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !cp ../input/kaggle-token/kaggle_token.json ./\n# !mv ./kaggle_token.json ./kaggle.json\n\n# !ls -l ../../root\n# !cp ./kaggle.json ../../root/\n# !ls ../../root\n\n# !mkdir ../../root/.kaggle\n# !mv ../../root/kaggle.json ../../root/.kaggle/kaggle.json\n\n# !chmod 600 /root/.kaggle/kaggle.json\n# !kaggle datasets init -p ./baseline","metadata":{"execution":{"iopub.status.busy":"2021-05-24T17:04:28.970873Z","iopub.execute_input":"2021-05-24T17:04:28.971248Z","iopub.status.idle":"2021-05-24T17:04:35.539404Z","shell.execute_reply.started":"2021-05-24T17:04:28.971214Z","shell.execute_reply":"2021-05-24T17:04:35.538437Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !cat ./baseline/dataset-metadata.json\n\n# import json\n# with open(\"./baseline/dataset-metadata.json\", 'r+') as file_:\n#     meta_data = json.load(file_)\n#     meta_data['title'] = 'baseline_BirdCLEF'\n#     meta_data['id'] = 'hotsonhonet/BirdCLEF'\n#     file_.seek(0)        \n#     json.dump(meta_data, file_, indent=4)\n#     file_.truncate()\n    \n# print(meta_data['title'], meta_data['id'])\n# print(\"\\nAfter editing\\n\")\n# !cat ./baseline/dataset-metadata.json","metadata":{"execution":{"iopub.status.busy":"2021-05-24T17:06:39.650968Z","iopub.execute_input":"2021-05-24T17:06:39.651341Z","iopub.status.idle":"2021-05-24T17:06:41.004341Z","shell.execute_reply.started":"2021-05-24T17:06:39.651308Z","shell.execute_reply":"2021-05-24T17:06:41.003395Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !kaggle datasets create -p ./baseline","metadata":{"execution":{"iopub.status.busy":"2021-05-24T17:07:37.436185Z","iopub.execute_input":"2021-05-24T17:07:37.436551Z","iopub.status.idle":"2021-05-24T17:07:43.296046Z","shell.execute_reply.started":"2021-05-24T17:07:37.436518Z","shell.execute_reply":"2021-05-24T17:07:43.295073Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}