{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nfor dirname, _, filenames in os.walk('/kaggle/input'):\n    for filename in filenames:\n        print(os.path.join(dirname, filename))\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cd /kaggle/input/birdsong-recognition/","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 音声が聞けるか確認","metadata":{}},{"cell_type":"code","source":"import librosa\nimport IPython.display as ipd\n\n#チェロの音声データのうちの一つを読み込む\ndata, rate = librosa.load('/kaggle/input/birdsong-recognition/train_audio/aldfly/XC78890.mp3')\n\n#読み込んだデータを再生する\nipd.Audio(data = data, rate = rate)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 訓練データのファイル名とラベル（鳥の種類）を取り出す","metadata":{}},{"cell_type":"code","source":"# csvファイル読み込み\ntrain = pd.read_csv('train.csv') # 訓練データの情報\ntrain.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 鳥の名前がそれぞれ何個あるか\ntrain['ebird_code'].value_counts()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 簡単のために、ファイル名と鳥の名前だけにする\ntrain_df = train[['filename', 'ebird_code']]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ebird_codeをlabelに変更\ntrain_df = train_df.rename(columns={'ebird_code':'label'})\ntrain_df = train_df.rename(columns={'filename':'fname'})\ntrain_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# カテゴリー変数を数値に変換\nlabels = pd.factorize(train_df['label'])[0]","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"labels","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 鳥の種類を数字に置き換えたcode_dfをデータフレームに\nlabel_df = pd.DataFrame(labels)\nlabel_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 列名を「label」に変更\nlabel_df = label_df.rename(columns={0:'label'})\nlabel_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_dfにlabel_dfを追加\ntrain_df['num'] = label_df\ntrain_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## 訓練データを作る","metadata":{}},{"cell_type":"code","source":"#ファイル名から音声データを読み込む関数を定義\nsampling_rate = rate\n\ndef _load_files(df):\n  result = []\n\n  for index, data in df.iterrows():\n        file_path = '/kaggle/input/birdsong-recognition/train_audio/' + data['label']+ '/' + data['fname']\n        data, _ = librosa.load(file_path, sr=sampling_rate)\n        result.append(data)\n        if index % 100 == 0:\n            print(index)\n\n  return result","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.shape","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# 全てのデータを読み込むと、時間がかかったり、メモリが足りないというエラーが出てしまったため、とりあえず800個読みこむ\nresult = _load_files(train_df.head(800))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#読み込んだデータを再生する\nipd.Audio(data = result[0], rate = rate)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(rate)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# ファイルによって長さが異なると予測されるため、最短の長さがどれくらいなのか求める\nminLength = 1000\nminIndex = -1\nfor i, data in enumerate(result):\n    length = len(data) / rate # 1秒にrate個分データがある\n    if length < minLength:\n        # print(\"i: \" + str(i) + \", 長さ: \" + str(length))\n        minLength = length\n        minIndex = i\n\nprint(\"i: \" + str(minIndex) + \", 長さ: \" + str(minLength))","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"len(result[160])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#読み込んだデータを再生する\nipd.Audio(data = result[160], rate = rate)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"1秒に満たないデータもあることがわかった。\nよって、AI教科書第9章同様、3秒で揃えることにした。","metadata":{}},{"cell_type":"code","source":"# データを3秒で揃える\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n\naudio_duration = 3\naudio_length = sampling_rate * audio_duration\nX_train = pad_sequences(result, dtype='float32', maxlen=audio_length, padding='pre', truncating='pre', value=0.0).tolist()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"また、データの標準化（平均値を0、分散を１に補正）も行う","metadata":{}},{"cell_type":"code","source":"from sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\nscaler = scaler.fit(X_train) # 平均μと分散σを計算\nX_train = scaler.transform(X_train) # 平均0、分散1になるよう変換（標準化）","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"ラベルをone-hot encordingする","metadata":{}},{"cell_type":"code","source":"train_df","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.utils import to_categorical\nY_train = to_categorical(train_df['num'])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# one-hot encordingされているか確認\nY_train","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# いくつクラスがあるのか確認\nlen(Y_train[0])","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## モデルを構築","metadata":{}},{"cell_type":"markdown","source":"今回はAI教科書第9章に倣い、次はCNNを構築","metadata":{}},{"cell_type":"code","source":"# TPUを使う\n# detect and init the TPU\n# tpu = tf.distribute.cluster_resolver.TPUClusterResolver.connect()\n\n# instantiate a distribution strategy\n# tpu_strategy = tf.distribute.experimental.TPUStrategy(tpu)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.layers import Dense, LSTM, Dropout,Bidirectional\nfrom tensorflow.keras.models import Sequential\nfrom tensorflow.keras.optimizers import Adam\nfrom tensorflow.keras.layers import Activation, Conv1D, MaxPooling1D, GlobalMaxPool1D,Dropout\n\ndef create_CNN_model():\n    \n  input_shape = (audio_length, 1)\n  model_cnn = Sequential()\n  model_cnn.add(Conv1D(filters=128, kernel_size=9, padding='valid', input_shape=input_shape, activation='relu'))\n  model_cnn.add(MaxPooling1D(pool_size=16))\n  model_cnn.add(Dropout(rate=0.2))\n  model_cnn.add(Conv1D(filters=64, kernel_size=3, padding='valid', activation='relu'))\n  model_cnn.add(GlobalMaxPool1D())\n  model_cnn.add(Dropout(rate=0.2))\n  model_cnn.add(Dense(264, activation=\"softmax\")) # 264個クラスがある\n  model_cnn.compile(optimizer=Adam(0.0001), loss=\"categorical_crossentropy\", metrics=['acc'])\n  return model_cnn\n\nmodel_CNN = create_CNN_model()\n#モデルの構造を表示する\nmodel_CNN.summary()","metadata":{"execution":{"iopub.status.busy":"2023-08-02T12:52:30.846245Z","iopub.execute_input":"2023-08-02T12:52:30.846609Z","iopub.status.idle":"2023-08-02T12:52:39.099345Z","shell.execute_reply.started":"2023-08-02T12:52:30.846579Z","shell.execute_reply":"2023-08-02T12:52:39.098009Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#学習開始(とりあえずエポック数3)\nhistory = model_CNN.fit(X_train, Y_train, batch_size=32, epochs=3, validation_split=0.1, verbose=1)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#モデルの重みの保存\nmodel_CNN.save_weights('/kaggle/working/saved_models/model_lstm_weights')","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## CNNで予測する","metadata":{}},{"cell_type":"code","source":"import matplotlib.pyplot as plt\n\n#評価関数と精度のグラフ表示\nfig, ax = plt.subplots(2,1)\nax[0].plot(history.history[\"loss\"], color=\"b\", label=\"Training Loss\")\nax[0].plot(history.history[\"val_loss\"], color=\"g\", label=\"Validation Loss\")\nax[0].legend()\n\nax[1].plot(history.history[\"acc\"], color=\"b\", label=\"Training Accuracy\")\nax[1].plot(history.history[\"val_acc\"], color=\"g\", label=\"Validation Accuracy\")\nax[1].legend()\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:19:37.131209Z","iopub.execute_input":"2023-08-02T10:19:37.131917Z","iopub.status.idle":"2023-08-02T10:19:37.585321Z","shell.execute_reply.started":"2023-08-02T10:19:37.131857Z","shell.execute_reply":"2023-08-02T10:19:37.584406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# csvファイル読み込み\ntest = pd.read_csv('test.csv') # テストデータの情報\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-02T08:23:58.760523Z","iopub.execute_input":"2023-08-02T08:23:58.761430Z","iopub.status.idle":"2023-08-02T08:23:58.777135Z","shell.execute_reply.started":"2023-08-02T08:23:58.761387Z","shell.execute_reply":"2023-08-02T08:23:58.775977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# モデルを読み込む\nmodel_cnn = create_CNN_model()\nmodel_cnn.load_weights(\"/kaggle/working/saved_models/model_lstm_weights\")","metadata":{"execution":{"iopub.status.busy":"2023-08-02T08:41:16.576342Z","iopub.execute_input":"2023-08-02T08:41:16.576742Z","iopub.status.idle":"2023-08-02T08:41:17.140598Z","shell.execute_reply.started":"2023-08-02T08:41:16.576711Z","shell.execute_reply":"2023-08-02T08:41:17.139605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# テストデータを読み込む(まずはサンプル)\nimport pandas as pd\n\naudio_file_path = \"/kaggle/input/birdsong-recognition/example_test_audio\"\nexample_df = pd.read_csv(\"/kaggle/input/birdsong-recognition/example_test_audio_summary.csv\")\nexample_df[\"filename\"] = [ \"BLKFR-10-CPL_20190611_093000.pt540\" if filename==\"BLKFR-10-CPL\" else \"ORANGE-7-CAP_20190606_093000.pt623\" for filename in example_df[\"filename\"]]","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:08:32.448992Z","iopub.execute_input":"2023-08-02T10:08:32.449349Z","iopub.status.idle":"2023-08-02T10:08:32.477515Z","shell.execute_reply.started":"2023-08-02T10:08:32.449319Z","shell.execute_reply":"2023-08-02T10:08:32.476614Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"example_df.shape","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:03:31.208609Z","iopub.execute_input":"2023-08-02T10:03:31.209000Z","iopub.status.idle":"2023-08-02T10:03:31.215797Z","shell.execute_reply.started":"2023-08-02T10:03:31.208969Z","shell.execute_reply":"2023-08-02T10:03:31.214662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import librosa\n\nexample_result = []\n\nfor index,data in example_df.iterrows():\n    filename = '{}/{}.mp3'.format(audio_file_path, data.filename)\n    data, _ = librosa.load(filename)\n    example_result.append(data)\n    if index == 10: # 今回は10個だけ読み込む\n        break","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:08:42.411824Z","iopub.execute_input":"2023-08-02T10:08:42.412181Z","iopub.status.idle":"2023-08-02T10:08:54.139034Z","shell.execute_reply.started":"2023-08-02T10:08:42.412151Z","shell.execute_reply":"2023-08-02T10:08:54.137877Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#読み込んだデータを再生する\nipd.Audio(data = example_result[0], rate = rate)","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# データを3秒で揃える\n\nfrom tensorflow.keras.preprocessing.sequence import pad_sequences\n\naudio_duration = 3\nsampling_rate = 22050\naudio_length = sampling_rate * audio_duration\nX_test = pad_sequences(example_result, dtype='float32', maxlen=audio_length, padding='pre', truncating='pre', value=0.0).tolist()","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:09:13.216918Z","iopub.execute_input":"2023-08-02T10:09:13.217592Z","iopub.status.idle":"2023-08-02T10:09:20.823612Z","shell.execute_reply.started":"2023-08-02T10:09:13.217554Z","shell.execute_reply":"2023-08-02T10:09:20.822516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# データの標準化\nfrom sklearn.preprocessing import StandardScaler\n\nscaler = StandardScaler()\nscaler = scaler.fit(X_test) # 平均μと分散σを計算\nX_test = scaler.transform(X_test) # 平均0、分散1になるよう変換（標準化）","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:09:22.641634Z","iopub.execute_input":"2023-08-02T10:09:22.642476Z","iopub.status.idle":"2023-08-02T10:09:22.901050Z","shell.execute_reply.started":"2023-08-02T10:09:22.642424Z","shell.execute_reply":"2023-08-02T10:09:22.900107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions = model_cnn.predict(X_test, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:19:42.479395Z","iopub.execute_input":"2023-08-02T10:19:42.479807Z","iopub.status.idle":"2023-08-02T10:19:44.719755Z","shell.execute_reply.started":"2023-08-02T10:19:42.479776Z","shell.execute_reply":"2023-08-02T10:19:44.718564Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"predictions","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:19:48.924419Z","iopub.execute_input":"2023-08-02T10:19:48.924818Z","iopub.status.idle":"2023-08-02T10:19:48.934208Z","shell.execute_reply.started":"2023-08-02T10:19:48.924788Z","shell.execute_reply":"2023-08-02T10:19:48.933031Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"全て同じ値になっている？","metadata":{}},{"cell_type":"code","source":"pred_labels_num = np.array([np.argmax(pred) for pred in predictions])","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:23:49.582020Z","iopub.execute_input":"2023-08-02T10:23:49.582385Z","iopub.status.idle":"2023-08-02T10:23:49.588764Z","shell.execute_reply.started":"2023-08-02T10:23:49.582356Z","shell.execute_reply":"2023-08-02T10:23:49.586996Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_labels_num","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:23:52.633433Z","iopub.execute_input":"2023-08-02T10:23:52.633846Z","iopub.status.idle":"2023-08-02T10:23:52.640841Z","shell.execute_reply.started":"2023-08-02T10:23:52.633816Z","shell.execute_reply":"2023-08-02T10:23:52.639566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"予測がうまくいっていなさそう","metadata":{}},{"cell_type":"code","source":"# 数字のラベルを鳥の種類に戻す\ntrain['ebird_code'].value_counts().index","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:23:32.029183Z","iopub.execute_input":"2023-08-02T10:23:32.029625Z","iopub.status.idle":"2023-08-02T10:23:32.040742Z","shell.execute_reply.started":"2023-08-02T10:23:32.029572Z","shell.execute_reply":"2023-08-02T10:23:32.039565Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_labels = train['ebird_code'].value_counts().index[pred_labels_num]","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:24:09.303744Z","iopub.execute_input":"2023-08-02T10:24:09.304118Z","iopub.status.idle":"2023-08-02T10:24:09.314613Z","shell.execute_reply.started":"2023-08-02T10:24:09.304088Z","shell.execute_reply":"2023-08-02T10:24:09.313550Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"pred_labels","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:24:12.964088Z","iopub.execute_input":"2023-08-02T10:24:12.964482Z","iopub.status.idle":"2023-08-02T10:24:12.971566Z","shell.execute_reply.started":"2023-08-02T10:24:12.964452Z","shell.execute_reply":"2023-08-02T10:24:12.970460Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submissionしてみる\naudio_file_path = \"/kaggle/input/birdsong-recognition/test_audio\"\ntest_df = pd.read_csv(\"/kaggle/input/birdsong-recognition/test.csv\")\nsubmission_df = pd.read_csv(\"/kaggle/input/birdsong-recognition/sample_submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:27:12.484917Z","iopub.execute_input":"2023-08-02T10:27:12.485277Z","iopub.status.idle":"2023-08-02T10:27:12.503934Z","shell.execute_reply.started":"2023-08-02T10:27:12.485249Z","shell.execute_reply":"2023-08-02T10:27:12.502329Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission_df[[\"row_id\",\"birds\"]].to_csv('/kaggle/working/submission.csv', index=False)\nsubmission_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-08-02T10:29:54.613765Z","iopub.execute_input":"2023-08-02T10:29:54.614141Z","iopub.status.idle":"2023-08-02T10:29:54.629535Z","shell.execute_reply.started":"2023-08-02T10:29:54.614094Z","shell.execute_reply":"2023-08-02T10:29:54.628107Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}