{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"Import The Libraries📚","metadata":{}},{"cell_type":"code","source":"import os\nimport pandas as pd\nimport torch\nimport json","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-05-02T07:43:52.793857Z","iopub.execute_input":"2022-05-02T07:43:52.794676Z","iopub.status.idle":"2022-05-02T07:43:54.143985Z","shell.execute_reply.started":"2022-05-02T07:43:52.794563Z","shell.execute_reply":"2022-05-02T07:43:54.143015Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport ast\nimport json\nimport numpy as np\nimport pandas as pd\nimport matplotlib.pyplot as plt\nimport soundfile as sf\nimport librosa\nimport librosa.display\nimport IPython.display as display","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:43:54.145915Z","iopub.execute_input":"2022-05-02T07:43:54.146238Z","iopub.status.idle":"2022-05-02T07:43:56.674947Z","shell.execute_reply.started":"2022-05-02T07:43:54.14619Z","shell.execute_reply":"2022-05-02T07:43:56.673594Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import keras\nimport tensorflow as tf\nfrom tensorflow.keras.models import Sequential\nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import StandardScaler","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:43:56.676428Z","iopub.execute_input":"2022-05-02T07:43:56.676667Z","iopub.status.idle":"2022-05-02T07:44:03.187548Z","shell.execute_reply.started":"2022-05-02T07:43:56.676638Z","shell.execute_reply":"2022-05-02T07:44:03.18664Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# model1 = tf.keras.models.load_model(\"../input/l-a-save-model/modelv0-birdclef.h5\")\nmodelv1 = tf.keras.models.load_model(\"../input/modelv1/modelv1-birdclef.h5\")","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:44:03.18998Z","iopub.execute_input":"2022-05-02T07:44:03.190335Z","iopub.status.idle":"2022-05-02T07:44:03.72261Z","shell.execute_reply.started":"2022-05-02T07:44:03.190278Z","shell.execute_reply":"2022-05-02T07:44:03.721476Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"os.getcwd()","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:44:03.725959Z","iopub.execute_input":"2022-05-02T07:44:03.726294Z","iopub.status.idle":"2022-05-02T07:44:03.734005Z","shell.execute_reply.started":"2022-05-02T07:44:03.726259Z","shell.execute_reply":"2022-05-02T07:44:03.733365Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"SEE THE DATA 🐔:","metadata":{}},{"cell_type":"code","source":"path = '/kaggle/input/birdclef-2022/'\nos.listdir(path)\n\ntrain_meta = pd.read_csv(path+'train_metadata.csv')\ntest_data = pd.read_csv(path+'test.csv')\nebird_data = pd.read_csv(path+'eBird_Taxonomy_v2021.csv')\nsamp_subm = pd.read_csv(path+'sample_submission.csv')\n\nwith open(path+'scored_birds.json') as f:\n    scored_birds = json.load(f)","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:44:03.735182Z","iopub.execute_input":"2022-05-02T07:44:03.735844Z","iopub.status.idle":"2022-05-02T07:44:03.963947Z","shell.execute_reply.started":"2022-05-02T07:44:03.735806Z","shell.execute_reply":"2022-05-02T07:44:03.963189Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class_list = train_meta.iloc[:,0]\nconverter = LabelEncoder()\ny = converter.fit_transform(class_list)","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:44:03.965091Z","iopub.execute_input":"2022-05-02T07:44:03.965341Z","iopub.status.idle":"2022-05-02T07:44:03.975251Z","shell.execute_reply.started":"2022-05-02T07:44:03.965313Z","shell.execute_reply":"2022-05-02T07:44:03.974402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**TEST MODEL**","metadata":{}},{"cell_type":"code","source":"import csv","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:44:03.976717Z","iopub.execute_input":"2022-05-02T07:44:03.977589Z","iopub.status.idle":"2022-05-02T07:44:03.987732Z","shell.execute_reply.started":"2022-05-02T07:44:03.977542Z","shell.execute_reply":"2022-05-02T07:44:03.986754Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"2. Features extraction\nIn the preprocessing of the data, feature extraction is necessary before running the training. The purpose is to define the inputs and outputs of the neural network.\n\nOUTPUT (y): last column which is the label.\nYou cannot use text directly for training. You will encode these labels with the LabelEncoder() function of sklearn.preprocessing.\n\nBefore running a model, you need to convert this type of categorical text data into numerical data that the model can understand.","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import LabelEncoder\nfrom sklearn.preprocessing import StandardScaler","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:44:03.988957Z","iopub.execute_input":"2022-05-02T07:44:03.989208Z","iopub.status.idle":"2022-05-02T07:44:03.999147Z","shell.execute_reply.started":"2022-05-02T07:44:03.989146Z","shell.execute_reply":"2022-05-02T07:44:03.998404Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"**Format submission : un fichier + 5 sec + un oiseau --> TRUE or FALSE**","metadata":{}},{"cell_type":"markdown","source":"**PREPARE TEST DATA**","metadata":{}},{"cell_type":"code","source":"# # Create empty csv file \n# header = \"filename length chroma_stft_mean chroma_stft_var rms_mean rms_var spectral_centroid_mean spectral_centroid_var spectral_bandwidth_mean spectral_bandwidth_var rolloff_mean rolloff_var zero_crossing_rate_mean zero_crossing_rate_var harmony_mean harmony_var perceptr_mean perceptr_var tempo mfcc1_mean mfcc1_var mfcc2_mean mfcc2_var mfcc3_mean mfcc3_var mfcc4_mean mfcc4_var label\".split()\n\n# file = open('datatest.csv', 'w', newline = '')\n# with file:\n#     writer = csv.writer(file)\n#     writer.writerow(header)\n    \n# # Transform each .wav file into a .csv row:\n# for line in test_data.index:\n#     sound_name = path+'test_soundscapes/'+test_data['file_id'][line]\n#     y, sr = librosa.load(sound_name, mono = True, duration = 30)\n#     chroma_stft = librosa.feature.chroma_stft(y = y, sr = sr)\n#     rmse = librosa.feature.rms(y = y)\n#     spec_cent = librosa.feature.spectral_centroid(y = y, sr = sr)\n#     spec_bw = librosa.feature.spectral_bandwidth(y = y, sr = sr)\n#     rolloff = librosa.feature.spectral_rolloff(y = y, sr = sr)\n#     zcr = librosa.feature.zero_crossing_rate(y)\n#     mfcc = librosa.feature.mfcc(y = y, sr = sr)\n#     to_append = f'{filename} {np.mean(chroma_stft)} {np.mean(rmse)} {np.mean(spec_cent)} {np.mean(spec_bw)} {np.mean(rolloff)} {np.mean(zcr)}'\n\n#     for e in mfcc:\n#         to_append += f' {np.mean(e)}'\n        \n#     to_append += str(\" \") + str(test_data['bird'][line])\n#     file = open('datatest.csv', 'a', newline = '')\n    \n#     with file:\n#         writer = csv.writer(file)\n#         writer.writerow(to_append.split())    \n    \n# df = pd.read_csv('datatest.csv')","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:44:04.001516Z","iopub.execute_input":"2022-05-02T07:44:04.002194Z","iopub.status.idle":"2022-05-02T07:44:04.012838Z","shell.execute_reply.started":"2022-05-02T07:44:04.002132Z","shell.execute_reply":"2022-05-02T07:44:04.011889Z"},"jupyter":{"source_hidden":true},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Create empty csv file \nheader = \"filename length chroma_stft_mean chroma_stft_var rms_mean rms_var spectral_centroid_mean spectral_centroid_var spectral_bandwidth_mean spectral_bandwidth_var rolloff_mean rolloff_var zero_crossing_rate_mean zero_crossing_rate_var harmony_mean harmony_var perceptr_mean perceptr_var tempo mfcc1_mean mfcc1_var mfcc2_mean mfcc2_var mfcc3_mean mfcc3_var mfcc4_mean mfcc4_var\".split()\n\nfile = open('datatest.csv', 'w', newline = '')\nwith file:\n    writer = csv.writer(file)\n    writer.writerow(header)\n\nfiles = sorted(os.listdir(path+'test_soundscapes/'))\n\nwith open('../input/birdclef-2022/scored_birds.json') as fp:\n    SCORED_BIRDS = json.load(fp)","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:44:04.014181Z","iopub.execute_input":"2022-05-02T07:44:04.014455Z","iopub.status.idle":"2022-05-02T07:44:04.03185Z","shell.execute_reply.started":"2022-05-02T07:44:04.014422Z","shell.execute_reply":"2022-05-02T07:44:04.031187Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import math","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:44:04.036092Z","iopub.execute_input":"2022-05-02T07:44:04.036607Z","iopub.status.idle":"2022-05-02T07:44:04.043169Z","shell.execute_reply.started":"2022-05-02T07:44:04.036568Z","shell.execute_reply":"2022-05-02T07:44:04.042367Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_df_test_from_path():\n    files = sorted(os.listdir(path+'test_soundscapes/'))\n    data = []\n    for f in files:\n        wv, sr = librosa.load(path+'test_soundscapes/' + f)\n        n_chunks = math.ceil(len(wv) / sr / 5)\n        filename = f\n        row_prefix = f[:-4]\n#         bird = SCORED_BIRDS[0]\n        for chunk in range(1, n_chunks + 1):\n            for bird in SCORED_BIRDS:\n                row_id = f\"{f[:-4]}_{bird}_{chunk*5}\"\n            \n                ending_second = chunk*5\n                data.append((filename, row_prefix, ending_second, bird))\n            \n    return  pd.DataFrame(data, columns=['filename', 'row_prefix', 'ending_second', 'birds'])\n        \ntest_df = create_df_test_from_path()\n","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:44:12.263673Z","iopub.execute_input":"2022-05-02T07:44:12.264099Z","iopub.status.idle":"2022-05-02T07:44:15.000094Z","shell.execute_reply.started":"2022-05-02T07:44:12.264066Z","shell.execute_reply":"2022-05-02T07:44:14.999101Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:44:15.001356Z","iopub.execute_input":"2022-05-02T07:44:15.001582Z","iopub.status.idle":"2022-05-02T07:44:15.025242Z","shell.execute_reply.started":"2022-05-02T07:44:15.001555Z","shell.execute_reply":"2022-05-02T07:44:15.024386Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Transform each .wav file into a .csv row:\n# for line in test_df.index:\n#     sound_name = path+'test_soundscapes/'+test_df['filename'][line]\nfor f in files:\n    sound_name = path+'test_soundscapes/'+f\n    print(sound_name)\n    y, sr = librosa.load(sound_name, mono = True, duration = 60)\n    chroma_stft = librosa.feature.chroma_stft(y = y, sr = sr)\n    rmse = librosa.feature.rms(y = y)\n    spec_cent = librosa.feature.spectral_centroid(y = y, sr = sr)\n    spec_bw = librosa.feature.spectral_bandwidth(y = y, sr = sr)\n    rolloff = librosa.feature.spectral_rolloff(y = y, sr = sr)\n    zcr = librosa.feature.zero_crossing_rate(y)\n    mfcc = librosa.feature.mfcc(y = y, sr = sr)\n    to_append = f'{f} {np.mean(chroma_stft)} {np.mean(rmse)} {np.mean(spec_cent)} {np.mean(spec_bw)} {np.mean(rolloff)} {np.mean(zcr)}'\n\n    for e in mfcc:\n        to_append += f' {np.mean(e)}'\n        \n#     to_append += str(\" \") + str(test_data['bird'][line])\n    file = open('datatest.csv', 'a', newline = '')\n    \n    with file:\n        writer = csv.writer(file)\n        writer.writerow(to_append.split())    \n    \ndf = pd.read_csv('datatest.csv')","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:44:42.84452Z","iopub.execute_input":"2022-05-02T07:44:42.844806Z","iopub.status.idle":"2022-05-02T07:44:45.597349Z","shell.execute_reply.started":"2022-05-02T07:44:42.844776Z","shell.execute_reply":"2022-05-02T07:44:45.596284Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ICI UNE PREDICTION PAR FICHIER ET PAS PAR 5SEC --> IL FAUT INTEGRER LA POSSIBILIT2 QU4IL N4Y AI AUCUN OISEAU","metadata":{"execution":{"iopub.status.busy":"2022-04-21T12:00:16.179198Z","iopub.execute_input":"2022-04-21T12:00:16.179564Z","iopub.status.idle":"2022-04-21T12:00:16.185664Z","shell.execute_reply.started":"2022-04-21T12:00:16.179528Z","shell.execute_reply":"2022-04-21T12:00:16.18481Z"}}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nfit = StandardScaler()\nX = fit.fit_transform(np.array(df.iloc[:, 2:27], dtype=float))\nX","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:44:55.838081Z","iopub.execute_input":"2022-05-02T07:44:55.838841Z","iopub.status.idle":"2022-05-02T07:44:55.851919Z","shell.execute_reply.started":"2022-05-02T07:44:55.838797Z","shell.execute_reply":"2022-05-02T07:44:55.850971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"y_pred  = modelv1.predict(X)\nclasses = np.argmax(y_pred, axis = 1)","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:44:57.926558Z","iopub.execute_input":"2022-05-02T07:44:57.927169Z","iopub.status.idle":"2022-05-02T07:44:58.281603Z","shell.execute_reply.started":"2022-05-02T07:44:57.9271Z","shell.execute_reply":"2022-05-02T07:44:58.28059Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# transform classes number into classes name\nresult = converter.inverse_transform(classes)\nprint(result)","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:45:00.6916Z","iopub.execute_input":"2022-05-02T07:45:00.691881Z","iopub.status.idle":"2022-05-02T07:45:00.699176Z","shell.execute_reply.started":"2022-05-02T07:45:00.691852Z","shell.execute_reply":"2022-05-02T07:45:00.69822Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# insert preditcion is the dataframe\ndf.insert(0, 'result', result)","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:45:02.433086Z","iopub.execute_input":"2022-05-02T07:45:02.433623Z","iopub.status.idle":"2022-05-02T07:45:02.440527Z","shell.execute_reply.started":"2022-05-02T07:45:02.433582Z","shell.execute_reply":"2022-05-02T07:45:02.439671Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df = pd.merge(test_df, df[['filename','result']], how='left', left_on = 'filename', right_on = 'filename')","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:45:03.765233Z","iopub.execute_input":"2022-05-02T07:45:03.766098Z","iopub.status.idle":"2022-05-02T07:45:03.786533Z","shell.execute_reply.started":"2022-05-02T07:45:03.766036Z","shell.execute_reply":"2022-05-02T07:45:03.785705Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:45:05.092412Z","iopub.execute_input":"2022-05-02T07:45:05.092729Z","iopub.status.idle":"2022-05-02T07:45:05.113823Z","shell.execute_reply.started":"2022-05-02T07:45:05.092695Z","shell.execute_reply":"2022-05-02T07:45:05.112854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tqdm.auto import tqdm","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:45:35.35629Z","iopub.execute_input":"2022-05-02T07:45:35.356585Z","iopub.status.idle":"2022-05-02T07:45:35.360687Z","shell.execute_reply.started":"2022-05-02T07:45:35.356554Z","shell.execute_reply":"2022-05-02T07:45:35.359765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import json\n\nsubmission = []\nfor i, row in tqdm(test_df.iterrows(), total=len(test_df)):\n    preds = True if test_df['birds'][i] == test_df['result'][i] else False\n    submission.append({\n        \"row_id\": f\"{row['row_prefix']}_{row['birds']}_{row['ending_second']}\",\n        \"target\": preds#[bird] > 1. / len(model.labels),\n    })\n\ndf_submission = pd.DataFrame(submission).set_index(\"row_id\")\ndf_submission.to_csv(\"submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2022-05-02T07:45:38.702615Z","iopub.execute_input":"2022-05-02T07:45:38.703088Z","iopub.status.idle":"2022-05-02T07:45:38.775615Z","shell.execute_reply.started":"2022-05-02T07:45:38.703033Z","shell.execute_reply":"2022-05-02T07:45:38.774745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}