{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2023-08-02T22:42:37.886914Z","iopub.execute_input":"2023-08-02T22:42:37.887319Z","iopub.status.idle":"2023-08-02T22:42:44.621814Z","shell.execute_reply.started":"2023-08-02T22:42:37.887288Z","shell.execute_reply":"2023-08-02T22:42:44.620845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd /kaggle/input/birdsong-recognition/","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:44.623522Z","iopub.execute_input":"2023-08-02T22:42:44.624102Z","iopub.status.idle":"2023-08-02T22:42:44.630062Z","shell.execute_reply.started":"2023-08-02T22:42:44.624066Z","shell.execute_reply":"2023-08-02T22:42:44.628993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 音声が聞けるか確認","metadata":{}},{"cell_type":"code","source":"import librosa\nimport IPython.display as ipd\n\n#チェロの音声データのうちの一つを読み込む\ndata, rate = librosa.load('/kaggle/input/birdsong-recognition/train_audio/aldfly/XC78890.mp3')\n\n#読み込んだデータを再生する\nipd.Audio(data = data, rate = rate)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:44.631594Z","iopub.execute_input":"2023-08-02T22:42:44.632204Z","iopub.status.idle":"2023-08-02T22:42:58.336490Z","shell.execute_reply.started":"2023-08-02T22:42:44.632161Z","shell.execute_reply":"2023-08-02T22:42:58.335068Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 訓練データのファイル名とラベル（鳥の種類）を取り出す","metadata":{}},{"cell_type":"code","source":"# csvファイル読み込み\ntrain = pd.read_csv('train.csv') # 訓練データの情報\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:58.339132Z","iopub.execute_input":"2023-08-02T22:42:58.339821Z","iopub.status.idle":"2023-08-02T22:42:58.951807Z","shell.execute_reply.started":"2023-08-02T22:42:58.339781Z","shell.execute_reply":"2023-08-02T22:42:58.950369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 鳥の名前がそれぞれ何個あるか\ntrain['ebird_code'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:58.953645Z","iopub.execute_input":"2023-08-02T22:42:58.954204Z","iopub.status.idle":"2023-08-02T22:42:58.972345Z","shell.execute_reply.started":"2023-08-02T22:42:58.954152Z","shell.execute_reply":"2023-08-02T22:42:58.970837Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 簡単のために、ファイル名と鳥の名前だけにする\ntrain_df = train[['filename', 'ebird_code']]","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:58.974017Z","iopub.execute_input":"2023-08-02T22:42:58.974829Z","iopub.status.idle":"2023-08-02T22:42:58.991962Z","shell.execute_reply.started":"2023-08-02T22:42:58.974779Z","shell.execute_reply":"2023-08-02T22:42:58.990589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:58.993521Z","iopub.execute_input":"2023-08-02T22:42:58.993991Z","iopub.status.idle":"2023-08-02T22:42:59.024686Z","shell.execute_reply.started":"2023-08-02T22:42:58.993947Z","shell.execute_reply":"2023-08-02T22:42:59.022099Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ebird_codeをlabelに変更\ntrain_df = train_df.rename(columns={'ebird_code':'label'})\ntrain_df = train_df.rename(columns={'filename':'fname'})\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:59.026203Z","iopub.execute_input":"2023-08-02T22:42:59.026936Z","iopub.status.idle":"2023-08-02T22:42:59.046927Z","shell.execute_reply.started":"2023-08-02T22:42:59.026897Z","shell.execute_reply":"2023-08-02T22:42:59.045781Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# カテゴリー変数を数値に変換\nlabels = pd.factorize(train_df['label'])[0]","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:59.048447Z","iopub.execute_input":"2023-08-02T22:42:59.049121Z","iopub.status.idle":"2023-08-02T22:42:59.055748Z","shell.execute_reply.started":"2023-08-02T22:42:59.049086Z","shell.execute_reply":"2023-08-02T22:42:59.054855Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:59.060965Z","iopub.execute_input":"2023-08-02T22:42:59.061726Z","iopub.status.idle":"2023-08-02T22:42:59.072102Z","shell.execute_reply.started":"2023-08-02T22:42:59.061674Z","shell.execute_reply":"2023-08-02T22:42:59.070828Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 鳥の種類を数字に置き換えたcode_dfをデータフレームに\nlabel_df = pd.DataFrame(labels)\nlabel_df","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:59.073979Z","iopub.execute_input":"2023-08-02T22:42:59.074786Z","iopub.status.idle":"2023-08-02T22:42:59.091905Z","shell.execute_reply.started":"2023-08-02T22:42:59.074740Z","shell.execute_reply":"2023-08-02T22:42:59.090677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 列名を「label」に変更\nlabel_df = label_df.rename(columns={0:'label'})\nlabel_df","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:59.093285Z","iopub.execute_input":"2023-08-02T22:42:59.093921Z","iopub.status.idle":"2023-08-02T22:42:59.107695Z","shell.execute_reply.started":"2023-08-02T22:42:59.093873Z","shell.execute_reply":"2023-08-02T22:42:59.106470Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:59.109321Z","iopub.execute_input":"2023-08-02T22:42:59.109671Z","iopub.status.idle":"2023-08-02T22:42:59.127176Z","shell.execute_reply.started":"2023-08-02T22:42:59.109642Z","shell.execute_reply":"2023-08-02T22:42:59.126009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_dfにlabel_dfを追加\ntrain_df['num'] = label_df\ntrain_df","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:59.129105Z","iopub.execute_input":"2023-08-02T22:42:59.129587Z","iopub.status.idle":"2023-08-02T22:42:59.144757Z","shell.execute_reply.started":"2023-08-02T22:42:59.129501Z","shell.execute_reply":"2023-08-02T22:42:59.143748Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 訓練データを作る","metadata":{}},{"cell_type":"code","source":"#ファイル名から音声データを読み込む関数を定義\nsampling_rate = rate\n\ndef _load_files(df):\n  result = []\n\n  for index, data in df.iterrows():\n        file_path = '/kaggle/input/birdsong-recognition/train_audio/' + data['label']+ '/' + data['fname']\n        data, _ = librosa.load(file_path, sr=sampling_rate)\n        result.append(data)\n        if index % 100 == 0:\n            print(index)\n\n  return result","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:59.146320Z","iopub.execute_input":"2023-08-02T22:42:59.146787Z","iopub.status.idle":"2023-08-02T22:42:59.181367Z","shell.execute_reply.started":"2023-08-02T22:42:59.146753Z","shell.execute_reply":"2023-08-02T22:42:59.180233Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:59.182942Z","iopub.execute_input":"2023-08-02T22:42:59.183835Z","iopub.status.idle":"2023-08-02T22:42:59.198848Z","shell.execute_reply.started":"2023-08-02T22:42:59.183800Z","shell.execute_reply":"2023-08-02T22:42:59.197787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 全てのデータを読み込むと、時間がかかったり、メモリが足りないというエラーが出てしまったため、とりあえず800個読みこむ\nresult = _load_files(train_df.head(800))","metadata":{"execution":{"iopub.status.busy":"2023-08-02T22:42:59.200350Z","iopub.execute_input":"2023-08-02T22:42:59.200915Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#読み込んだデータを再生する\nipd.Audio(data = result[0], rate = rate)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(rate)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ファイルによって長さが異なると予測されるため、最短の長さがどれくらいなのか求める\nminLength = 1000\nminIndex = -1\nfor i, data in enumerate(result):\n    length = len(data) / rate # 1秒にrate個分データがある\n    if length < minLength:\n        # print(\"i: \" + str(i) + \", 長さ: \" + str(length))\n        minLength = length\n        minIndex = i\n\nprint(\"i: \" + str(minIndex) + \", 長さ: \" + str(minLength))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(result[160])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#読み込んだデータを再生する\nipd.Audio(data = result[160], rate = rate)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1秒に満たないデータもあることがわかった。\nよって、AI教科書第9章同様、3秒で揃えることにした。","metadata":{}},{"cell_type":"code","source":"# データを3秒で揃える\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n\naudio_duration = 3\naudio_length = sampling_rate * audio_duration\nX_train = pad_sequences(result, dtype='float32', maxlen=audio_length, padding='pre', truncating='pre', value=0.0).tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"また、データの標準化（平均値を0、分散を１に補正）も行う","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\nscaler = scaler.fit(X_train) # 平均μと分散σを計算\nX_train = scaler.transform(X_train) # 平均0、分散1になるよう変換（標準化）","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ラベルをone-hot encordingする","metadata":{}},{"cell_type":"code","source":"train_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.utils import to_categorical\nY_train = to_categorical(train_df['num'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# one-hot encordingされているか確認\nY_train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# いくつクラスがあるのか確認\nlen(Y_train[0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## モデルを構築","metadata":{}},{"cell_type":"markdown","source":"今回はAI教科書第9章に倣い、まずはLSTMを構築","metadata":{}},{"cell_type":"code","source":"# TPUを使う\n# detect and init the TPU\n# tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\n# instantiate a distribution strategy\n# tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import Dense, LSTM, Dropout,Bidirectional\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\n\ndef create_lstm_model():\n  input_shape = (audio_length, 1)\n\n  #モデルの構築\n  model_lstm = Sequential()\n  model_lstm.add(LSTM(64, return_sequences=True, dropout=0.3 ,input_shape=input_shape))\n  model_lstm.add(LSTM(64, return_sequences=False, dropout=0.3))\n  model_lstm.add(Dense(units=264, activation=\"softmax\")) # 264個クラスがある\n\n  # 損失関数はcategorical_crossentropy（多クラス分類問題なので）\n  model_lstm.compile(loss=\"categorical_crossentropy\", optimizer=Adam(0.001), metrics=[\"acc\"])\n    \n  return model_lstm\n\nmodel_lstm = create_lstm_model()\n#モデルの構造を表示する\nmodel_lstm.summary()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#学習開始(とりあえずエポック数3)\nhistory = model_lstm.fit(X_train, Y_train, batch_size=16, epochs=3, validation_split=0.1, verbose=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lossは下がっているが、あまり正解率accが上がらないので、batch_sizeを増やす(しかし、バッチサイズを大きくしすぎる（batch_size=64など)と、GPUのメモリが足りなくなるので注意！)\nmodel_lstm2 = create_lstm_model()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = model_lstm2.fit(X_train, Y_train, batch_size=32, epochs=3, validation_split=0.1, verbose=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#モデルの重みの保存\nmodel_lstm.save_weights('/kaggle/working/saved_models/model_lstm_weights')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## LSTMで予測する","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n#評価関数と精度のグラフ表示\nfig, ax = plt.subplots(2,1)\nax[0].plot(history.history[\"loss\"], color=\"b\", label=\"Training Loss\")\nax[0].plot(history.history[\"val_loss\"], color=\"g\", label=\"Validation Loss\")\nax[0].legend()\n\nax[1].plot(history.history[\"acc\"], color=\"b\", label=\"Training Accuracy\")\nax[1].plot(history.history[\"val_acc\"], color=\"g\", label=\"Validation Accuracy\")\nax[1].legend()\n\nplt.show()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# csvファイル読み込み\ntest = pd.read_csv('test.csv') # テストデータの情報\ntest.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# モデルを読み込む\nmodel_lstm = create_lstm_model()\nmodel_lstm.load_weights(\"/kaggle/working/saved_models/model_lstm_weights\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# テストデータを読み込む(まずはサンプル)\nimport pandas as pd\n\naudio_file_path = \"/kaggle/input/birdsong-recognition/example_test_audio\"\nexample_df = pd.read_csv(\"/kaggle/input/birdsong-recognition/example_test_audio_summary.csv\")\nexample_df[\"filename\"] = [ \"BLKFR-10-CPL_20190611_093000.pt540\" if filename==\"BLKFR-10-CPL\" else \"ORANGE-7-CAP_20190606_093000.pt623\" for filename in example_df[\"filename\"]]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\n\nexample_result = []\n\nfor index,data in example_df.iterrows():\n    filename = '{}/{}.mp3'.format(audio_file_path, data.filename)\n    data, _ = librosa.load(filename)\n    example_result.append(data)\n    if index == 10: # 今回は10個だけ読み込む\n        break","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# データを3秒で揃える\n\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n\naudio_duration = 3\nsampling_rate = 22050\naudio_length = sampling_rate * audio_duration\nX_test = pad_sequences(example_result, dtype='float32', maxlen=audio_length, padding='pre', truncating='pre', value=0.0).tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# データの標準化\nfrom sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\nscaler = scaler.fit(X_test) # 平均μと分散σを計算\nX_test = scaler.transform(X_test) # 平均0、分散1になるよう変換（標準化）","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model_lstm.predict(X_test, verbose=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"全て同じ値になっている？","metadata":{}},{"cell_type":"code","source":"pred_labels_num = np.array([np.argmax(pred) for pred in predictions])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_labels_num","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"予測がうまくいっていなさそう","metadata":{}},{"cell_type":"code","source":"# 数字のラベルを鳥の種類に戻す\ntrain['ebird_code'].value_counts().index","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_labels = train['ebird_code'].value_counts().index[pred_labels_num]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_labels","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"actual_labels = []\nfor lab in example_df['birds']:\n    label = str(lab).split()[0]\n    actual_labels.append(label)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"actual_labels = actual_labels[0:11]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#正答率の算出\ntmp = actual_labels == pred_labels\ntmp.sum()/len(tmp)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submissionしてみる\naudio_file_path = \"/kaggle/input/birdsong-recognition/test_audio\"\ntest_df = pd.read_csv(\"/kaggle/input/birdsong-recognition/test.csv\")\nsubmission_df = pd.read_csv(\"/kaggle/input/birdsong-recognition/sample_submission.csv\")","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df[[\"row_id\",\"birds\"]].to_csv('/kaggle/working/submission.csv', index=False)\nsubmission_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}