{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n\n#for dirname, _, filenames in os.walk('/kaggle/input'):\n    #for filename in filenames:\n        #print(os.path.join(dirname, filename))\n        \n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-02-27T11:59:12.429281Z","iopub.execute_input":"2022-02-27T11:59:12.429677Z","iopub.status.idle":"2022-02-27T11:59:12.436135Z","shell.execute_reply.started":"2022-02-27T11:59:12.429630Z","shell.execute_reply":"2022-02-27T11:59:12.434503Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install noisereduce","metadata":{"execution":{"iopub.status.busy":"2022-02-27T11:59:17.701834Z","iopub.execute_input":"2022-02-27T11:59:17.702500Z","iopub.status.idle":"2022-02-27T11:59:28.862591Z","shell.execute_reply.started":"2022-02-27T11:59:17.702458Z","shell.execute_reply":"2022-02-27T11:59:28.861265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\nimport noisereduce as nr\nimport glob\n#Just a sample of birds (from the scored_birds.json file)\nbird_types = ['akiapo', 'aniani', 'apapan', 'barpet', 'crehon', 'elepai', 'ercfra', 'hawama',\n              'hawcre', 'hawgoo', 'hawhaw', 'hawpet1', 'houfin', 'iiwi', 'jabwar', 'maupar',\n              'omao', 'puaioh', 'skylar', 'warwhe1', 'yefcan']\ndata = []\nlabels = []\nfor i in range(len(bird_types)):\n    count = 0\n    for j in glob.glob('/kaggle/input/birdclef-2022/train_audio/'+bird_types[i]+'/*.ogg'):\n        count += 1\n        if (count > 10): \n            break\n        y, sr = librosa.load(j)\n        #Noise reduction\n        reduced = nr.reduce_noise(y=y, sr=sr)\n        dur = len(reduced) / sr\n        #Split into 5 second intervals\n        for k in range(int(dur/5)):\n            data.append(reduced[5*k*sr:5*(k+1)*sr])\n            labels.append(i)","metadata":{"execution":{"iopub.status.busy":"2022-02-27T11:59:32.013436Z","iopub.execute_input":"2022-02-27T11:59:32.013830Z","iopub.status.idle":"2022-02-27T12:06:05.013030Z","shell.execute_reply.started":"2022-02-27T11:59:32.013790Z","shell.execute_reply":"2022-02-27T12:06:05.012263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\nprint(len(data))","metadata":{"execution":{"iopub.status.busy":"2022-02-27T12:07:15.896535Z","iopub.execute_input":"2022-02-27T12:07:15.896886Z","iopub.status.idle":"2022-02-27T12:07:15.903496Z","shell.execute_reply.started":"2022-02-27T12:07:15.896851Z","shell.execute_reply":"2022-02-27T12:07:15.902508Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#FFT\nffts = []\nsr = 22050\nfrom scipy.fft import fft, fftfreq\nfor i in range(len(data)):\n    yf = fft(data[i])\n    xf = fftfreq(len(data[i]), 1.0/sr)\n    print(xf[:int(len(yf)/2)])\n    yf = yf[range(0, int(len(yf)/2), 50)]\n    ffts.append(yf)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import matplotlib.pyplot as plt\nfig, axs = plt.subplots(4, 2)\nfor i in range(100,108):\n    x = i % 4\n    y = i // 4 - 25\n    yf = ffts[i]\n    xf = np.arange(0, 11030, 10)\n    axs[x, y].plot(xf, np.abs(yf))","metadata":{"execution":{"iopub.status.busy":"2022-02-27T12:10:46.812323Z","iopub.execute_input":"2022-02-27T12:10:46.812812Z","iopub.status.idle":"2022-02-27T12:10:47.999042Z","shell.execute_reply.started":"2022-02-27T12:10:46.812768Z","shell.execute_reply":"2022-02-27T12:10:47.997964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.preprocessing import OrdinalEncoder\nfrom sklearn.model_selection import train_test_split\nenc = OrdinalEncoder()\nenc.fit_transform(np.array(labels).reshape(-1, 1))\nfft_mags = np.abs(ffts)\nX_train, X_test, y_train, y_test = train_test_split(\n    fft_mags, labels, test_size=0.33)","metadata":{"execution":{"iopub.status.busy":"2022-02-27T12:12:33.932605Z","iopub.execute_input":"2022-02-27T12:12:33.933363Z","iopub.status.idle":"2022-02-27T12:12:33.972494Z","shell.execute_reply.started":"2022-02-27T12:12:33.933314Z","shell.execute_reply":"2022-02-27T12:12:33.971133Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#SVC\nfrom sklearn.pipeline import make_pipeline\nfrom sklearn.preprocessing import StandardScaler\nfrom sklearn.svm import LinearSVC\nfrom sklearn.metrics import f1_score\n\nclf = make_pipeline(StandardScaler(), LinearSVC(dual=False, C=0.1))\nclf.fit(X_train, y_train)\n\ny_pred = clf.predict(X_test)\ny_train_pred = clf.predict(X_train)\nprint('Train score:', f1_score(y_train, y_train_pred, average='micro'))\nprint('Test score:', f1_score(y_test, y_pred, average='micro'))","metadata":{"execution":{"iopub.status.busy":"2022-02-27T12:18:36.630389Z","iopub.execute_input":"2022-02-27T12:18:36.630848Z","iopub.status.idle":"2022-02-27T12:22:48.895395Z","shell.execute_reply.started":"2022-02-27T12:18:36.630807Z","shell.execute_reply":"2022-02-27T12:22:48.894304Z"},"trusted":true},"execution_count":null,"outputs":[]}]}