{"metadata":{"kernelspec":{"display_name":"Python 3","language":"python","name":"python3"},"language_info":{"codemirror_mode":{"name":"ipython","version":3},"file_extension":".py","mimetype":"text/x-python","name":"python","nbconvert_exporter":"python","pygments_lexer":"ipython3","version":"3.10.14"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":8900,"databundleVersionId":862232,"sourceType":"competition"}],"dockerImageVersionId":30786,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false},"papermill":{"default_parameters":{},"duration":8790.969155,"end_time":"2024-10-20T08:07:00.344654","environment_variables":{},"exception":null,"input_path":"__notebook__.ipynb","output_path":"__notebook__.ipynb","parameters":{},"start_time":"2024-10-20T05:40:29.375499","version":"2.6.0"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"**Описание**: попытка выполнить задачу с помощью mel-спектрограммы (librosa) и сверточной нейросети (keras tensorflow). \n\nС помощью librosa считывается аудиофайл и строится mel-спектрограмма. Далее модель обучается на полученных изображених.","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":0.006259,"end_time":"2024-10-20T05:40:32.101825","exception":false,"start_time":"2024-10-20T05:40:32.095566","status":"completed"},"tags":[]}},{"cell_type":"markdown","source":"Импорт необходимых библиотек","metadata":{"papermill":{"duration":0.005317,"end_time":"2024-10-20T05:40:32.112838","exception":false,"start_time":"2024-10-20T05:40:32.107521","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import os\nfrom tqdm import tqdm\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport librosa.display\nimport matplotlib.pyplot as plt\nfrom sklearn.model_selection import StratifiedKFold\nfrom keras.utils import to_categorical\n\nimport keras\nfrom keras import Sequential\nfrom keras.layers import Input, Convolution2D, MaxPooling2D, Dense, Dropout, Flatten\nfrom keras.optimizers import Adam","metadata":{"execution":{"iopub.status.busy":"2024-12-10T21:17:22.683432Z","iopub.execute_input":"2024-12-10T21:17:22.683887Z","iopub.status.idle":"2024-12-10T21:17:22.691670Z","shell.execute_reply.started":"2024-12-10T21:17:22.683850Z","shell.execute_reply":"2024-12-10T21:17:22.690333Z"},"papermill":{"duration":15.587942,"end_time":"2024-10-20T05:40:47.706286","exception":false,"start_time":"2024-10-20T05:40:32.118344","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Чтение и подготовка данных","metadata":{"papermill":{"duration":0.007367,"end_time":"2024-10-20T05:40:47.719509","exception":false,"start_time":"2024-10-20T05:40:47.712142","status":"completed"},"tags":[]}},{"cell_type":"code","source":"# Чтение тренировочных данных\ntrain_data = pd.read_csv(\"../input/freesound-audio-tagging/train.csv\")\ntest_data = pd.read_csv(\"../input/freesound-audio-tagging/test_post_competition.csv\")\n\n# Удаляем ненужные колонки\ntrain_data = train_data.drop([\"manually_verified\"], axis=1)\ntest_data = test_data.drop([\"usage\", \"freesound_id\", \"license\"], axis=1)\n\n# Преобразуем метки в числовой формат\nlabel_to_num = {l: n for n, l in enumerate(train_data.label.unique())}\nnum_to_label = {v: k for k, v in label_to_num.items()}\ntrain_data[\"label\"] = train_data[\"label\"].map(lambda x: label_to_num[x])\ntest_data[\"label\"] = test_data[\"label\"].map(lambda x: label_to_num[x] if x in label_to_num else -1 )\n\nsr = 22050\nn_mels = 90\nduration = 30\nn_fft = 4096\nhop_length = 512\nexpected_time_steps = 500\n\n# Функция для преобразования аудиофайла в мел-спектрограмму\ndef audio_to_mel_spectrogram(file_path, sr=sr, n_mels=n_mels, duration=duration, n_fft=n_fft, hop_length=hop_length):\n    data_audio, _ = librosa.load(file_path, sr=sr, duration=duration)\n    mel_spectrogram = librosa.feature.melspectrogram(y=data_audio, sr=sr, n_mels=n_mels, n_fft=n_fft, hop_length=hop_length)\n    log_mel_spec = librosa.power_to_db(mel_spectrogram)\n    log_mel_spec = normalize(log_mel_spec)\n    \n    if log_mel_spec.shape[1] < expected_time_steps:\n        pad_width = expected_time_steps - log_mel_spec.shape[1]\n        log_mel_spec = np.pad(log_mel_spec, pad_width=((0, 0), (0, pad_width)), mode='constant')\n    elif log_mel_spec.shape[1] > expected_time_steps:\n        log_mel_spec = log_mel_spec[:, :expected_time_steps]  \n\n    return log_mel_spec\n\ndef normalize(spec):\n    std = np.std(spec)\n    if std == 0:\n        std = 1e-10\n    return (spec - np.mean(spec)) / np.std(spec)\n\n# Подготовка данных для обучения\ndef prepare_data(df, data_dir):\n    X = []\n    y = []\n    for i, row in tqdm(df.iterrows()):\n        file_path = os.path.join(data_dir, row['fname'])\n        mel_spectrogram = audio_to_mel_spectrogram(file_path)\n        X.append(mel_spectrogram)\n        y.append(row['label'])  \n    return np.array(X), np.array(y)","metadata":{"execution":{"iopub.status.busy":"2024-12-10T21:17:23.235907Z","iopub.execute_input":"2024-12-10T21:17:23.236363Z","iopub.status.idle":"2024-12-10T21:17:23.327847Z","shell.execute_reply.started":"2024-12-10T21:17:23.236324Z","shell.execute_reply":"2024-12-10T21:17:23.326683Z"},"papermill":{"duration":0.09387,"end_time":"2024-10-20T05:40:47.827943","exception":false,"start_time":"2024-10-20T05:40:47.734073","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train_dir = \"../input/freesound-audio-tagging/audio_train/\"\nX_train, y_train = prepare_data(train_data, train_dir)\n\ntest_dir = \"../input/freesound-audio-tagging/audio_test/\"\nX_test, y_test = prepare_data(test_data, test_dir)","metadata":{"papermill":{"duration":1489.104351,"end_time":"2024-10-20T06:05:36.937841","exception":false,"start_time":"2024-10-20T05:40:47.833490","status":"completed"},"tags":[],"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T21:17:34.141855Z","iopub.execute_input":"2024-12-10T21:17:34.142338Z","iopub.status.idle":"2024-12-10T21:18:04.361199Z","shell.execute_reply.started":"2024-12-10T21:17:34.142295Z","shell.execute_reply":"2024-12-10T21:18:04.357403Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Перевод в категориальные признаки","metadata":{"papermill":{"duration":0.005959,"end_time":"2024-10-20T06:05:36.954784","exception":false,"start_time":"2024-10-20T06:05:36.948825","status":"completed"},"tags":[]}},{"cell_type":"code","source":"num_classes = len(train_data.label.unique())\ny_train_cat = to_categorical(y_train, num_classes)\ny_test_cat = to_categorical(y_test, num_classes)","metadata":{"execution":{"iopub.status.busy":"2024-12-10T21:18:04.368629Z","iopub.execute_input":"2024-12-10T21:18:04.370574Z","iopub.status.idle":"2024-12-10T21:18:04.387073Z","shell.execute_reply.started":"2024-12-10T21:18:04.370482Z","shell.execute_reply":"2024-12-10T21:18:04.385266Z"},"papermill":{"duration":0.032779,"end_time":"2024-10-20T06:05:36.993588","exception":false,"start_time":"2024-10-20T06:05:36.960809","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"X_train.shape, X_test.shape","metadata":{"execution":{"iopub.status.busy":"2024-12-10T21:18:04.388747Z","iopub.execute_input":"2024-12-10T21:18:04.389211Z","iopub.status.idle":"2024-12-10T21:18:04.413211Z","shell.execute_reply.started":"2024-12-10T21:18:04.389160Z","shell.execute_reply":"2024-12-10T21:18:04.411830Z"},"papermill":{"duration":0.023088,"end_time":"2024-10-20T06:05:37.040873","exception":false,"start_time":"2024-10-20T06:05:37.017785","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"2 примера mel-спектрограмм из тренировочной выборки (с дополнением до единой длины)","metadata":{"papermill":{"duration":0.005901,"end_time":"2024-10-20T06:05:37.053000","exception":false,"start_time":"2024-10-20T06:05:37.047099","status":"completed"},"tags":[]}},{"cell_type":"code","source":"plt.figure(figsize=(16, 5))\nlibrosa.display.specshow(X_train[0], x_axis=\"time\", y_axis=\"mel\", sr=sr)\nplt.colorbar(format=\"%+2.0f dB\")\nplt.xlabel(\"Время (сек.)\")\nplt.ylabel(\"Частота (Гц)\")","metadata":{"execution":{"iopub.status.busy":"2024-12-10T21:18:04.424677Z","iopub.execute_input":"2024-12-10T21:18:04.425347Z","iopub.status.idle":"2024-12-10T21:18:04.958991Z","shell.execute_reply.started":"2024-12-10T21:18:04.425279Z","shell.execute_reply":"2024-12-10T21:18:04.957766Z"},"papermill":{"duration":0.607208,"end_time":"2024-10-20T06:05:37.666365","exception":false,"start_time":"2024-10-20T06:05:37.059157","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"plt.figure(figsize=(16, 5))\nlibrosa.display.specshow(X_train[234], x_axis=\"time\", y_axis=\"mel\", sr=sr)\nplt.colorbar(format=\"%+2.0f dB\")\nplt.xlabel(\"Время (сек.)\")\nplt.ylabel(\"Частота (Гц)\")","metadata":{"execution":{"iopub.status.busy":"2024-12-10T21:18:04.960316Z","iopub.execute_input":"2024-12-10T21:18:04.960679Z","iopub.status.idle":"2024-12-10T21:18:05.172803Z","shell.execute_reply.started":"2024-12-10T21:18:04.960633Z","shell.execute_reply":"2024-12-10T21:18:05.171185Z"},"papermill":{"duration":0.506212,"end_time":"2024-10-20T06:05:38.184989","exception":false,"start_time":"2024-10-20T06:05:37.678777","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Добавление размерности для использования сверточной сети и разбиение на тренировочную и тестовые выборки","metadata":{"papermill":{"duration":0.008823,"end_time":"2024-10-20T06:05:38.203032","exception":false,"start_time":"2024-10-20T06:05:38.194209","status":"completed"},"tags":[]}},{"cell_type":"code","source":"X_train = X_train[..., np.newaxis]\nX_test = X_test[..., np.newaxis]","metadata":{"execution":{"iopub.status.busy":"2024-12-10T21:18:21.668434Z","iopub.execute_input":"2024-12-10T21:18:21.668820Z","iopub.status.idle":"2024-12-10T21:18:21.674574Z","shell.execute_reply.started":"2024-12-10T21:18:21.668789Z","shell.execute_reply":"2024-12-10T21:18:21.673142Z"},"papermill":{"duration":0.674429,"end_time":"2024-10-20T06:05:38.886791","exception":false,"start_time":"2024-10-20T06:05:38.212362","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Построение модели для обучения","metadata":{"papermill":{"duration":0.008291,"end_time":"2024-10-20T06:05:38.903683","exception":false,"start_time":"2024-10-20T06:05:38.895392","status":"completed"},"tags":[]}},{"cell_type":"code","source":"kernel_size = 3\npool_size = 2\nconv_depth_1 = 32\nconv_depth_2 = 64  \nconv_depth_3 = 128\ndrop_prob_1 = 0.25 \ndrop_prob_2 = 0.5 \nhidden_size = 512 \n\ndef get_model():\n    model = Sequential(\n        [\n            Input(shape=(n_mels, expected_time_steps, 1)),  \n            \n            Convolution2D(conv_depth_1, kernel_size, padding=\"same\", activation=\"relu\"),\n            MaxPooling2D(pool_size=pool_size),\n            Dropout(drop_prob_1),\n            \n            Convolution2D(conv_depth_2, kernel_size, padding=\"same\", activation=\"relu\"),\n            MaxPooling2D(pool_size=pool_size),\n            Dropout(drop_prob_1),\n    \n            Convolution2D(conv_depth_3, kernel_size, padding=\"same\", activation=\"relu\"),  # Новый слой\n            MaxPooling2D(pool_size=pool_size),\n            Dropout(drop_prob_1),\n            \n            Flatten(),\n            Dense(hidden_size, activation=\"relu\"),\n            Dropout(drop_prob_2),\n            Dense(num_classes, activation=\"softmax\"),\n        ]\n    )\n    \n    model.compile(\n        loss=\"categorical_crossentropy\",\n        optimizer=Adam(),\n        metrics=[\"accuracy\"],\n    )\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2024-12-10T21:18:22.113481Z","iopub.execute_input":"2024-12-10T21:18:22.114351Z","iopub.status.idle":"2024-12-10T21:18:22.121953Z","shell.execute_reply.started":"2024-12-10T21:18:22.114313Z","shell.execute_reply":"2024-12-10T21:18:22.120803Z"},"papermill":{"duration":0.409562,"end_time":"2024-10-20T06:05:39.321649","exception":false,"start_time":"2024-10-20T06:05:38.912087","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Обучение модели с кросс-валидацией (с уччетом распределений классов)","metadata":{"papermill":{"duration":0.009618,"end_time":"2024-10-20T06:05:39.340655","exception":false,"start_time":"2024-10-20T06:05:39.331037","status":"completed"},"tags":[]}},{"cell_type":"code","source":"batch_size = 32\nnum_epochs = 50\nnum_folds = 5\n\nkfold = StratifiedKFold(n_splits=num_folds, shuffle=True)\n\nfor num, (train, test) in enumerate(kfold.split(X_train, y_train)):\n    print(f\"Train №{num+1}/{num_folds}\")\n    X_val, y_val = X_train[test], y_train_cat[test]\n    X_trn, y_trn = X_train[train], y_train_cat[train]\n    print(X_trn.shape, y_trn.shape)\n    print(X_val.shape, y_val.shape)\n    \n    checkpoint_filepath = f'{num}_checkpoint.weights.h5'\n    model_checkpoint_callback = keras.callbacks.ModelCheckpoint(\n        filepath=checkpoint_filepath,\n        save_weights_only=True,\n        monitor='val_accuracy',\n        mode='max',\n        save_best_only=True,\n        verbose=1,\n    )\n    early_stoping = keras.callbacks.EarlyStopping(monitor='val_loss', patience=5, verbose=1)\n\n    model = get_model()\n    model.fit(\n        X_trn, y_trn,\n        batch_size=batch_size,\n        epochs=num_epochs,\n        validation_data=(X_val, y_val),\n        callbacks=[model_checkpoint_callback, early_stoping]\n    )\n    print()","metadata":{"execution":{"iopub.status.busy":"2024-12-10T21:21:21.401931Z","iopub.execute_input":"2024-12-10T21:21:21.402392Z","iopub.status.idle":"2024-12-10T21:21:46.984754Z","shell.execute_reply.started":"2024-12-10T21:21:21.402352Z","shell.execute_reply":"2024-12-10T21:21:46.982860Z"},"papermill":{"duration":7046.013658,"end_time":"2024-10-20T08:03:05.363575","exception":false,"start_time":"2024-10-20T06:05:39.349917","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Загрузка сохраненных весов моделей","metadata":{"papermill":{"duration":0.133039,"end_time":"2024-10-20T08:03:48.441831","exception":false,"start_time":"2024-10-20T08:03:48.308792","status":"completed"},"tags":[]}},{"cell_type":"code","source":"models = []\nfor i in range(num_folds):\n    model = get_model()\n    checkpoint_filepath = f\"{i}_checkpoint.weights.h5\"\n    model.load_weights(checkpoint_filepath, skip_mismatch=False)\n    models.append(model)","metadata":{"execution":{"iopub.status.busy":"2024-12-10T21:19:26.080071Z","iopub.execute_input":"2024-12-10T21:19:26.080514Z","iopub.status.idle":"2024-12-10T21:19:30.160393Z","shell.execute_reply.started":"2024-12-10T21:19:26.080475Z","shell.execute_reply":"2024-12-10T21:19:30.159151Z"},"papermill":{"duration":28.666548,"end_time":"2024-10-20T08:04:17.242580","exception":false,"start_time":"2024-10-20T08:03:48.576032","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"Получение топ-3 предсказания для тестовой выборки и сохранение в csv файл","metadata":{"papermill":{"duration":0.138395,"end_time":"2024-10-20T08:04:17.523201","exception":false,"start_time":"2024-10-20T08:04:17.384806","status":"completed"},"tags":[]}},{"cell_type":"code","source":"all_preds = []\nfor num, model in enumerate(models):\n    predictions = model.predict(X_test)\n    all_preds.append(predictions)\n    \n    top_3_indices = np.argsort(predictions, axis=1)[:, -3:] \n    \n    top_3_labels = []\n    for indices in top_3_indices:\n        labels = [num_to_label[idx] for idx in reversed(indices)]  \n        top_3_labels.append(\" \".join(labels)) \n    \n    results_df = pd.DataFrame({\n        \"fname\": test_data[\"fname\"],\n        \"label\": top_3_labels\n    })\n    \n    results_df.to_csv(f\"submission_{num}.csv\", index=False)\n    \n    results_df.head(10)","metadata":{"execution":{"iopub.status.busy":"2024-12-10T21:20:16.402369Z","iopub.execute_input":"2024-12-10T21:20:16.402886Z","iopub.status.idle":"2024-12-10T21:20:31.219777Z","shell.execute_reply.started":"2024-12-10T21:20:16.402846Z","shell.execute_reply":"2024-12-10T21:20:31.218156Z"},"papermill":{"duration":159.190555,"end_time":"2024-10-20T08:06:56.853918","exception":false,"start_time":"2024-10-20T08:04:17.663363","status":"completed"},"tags":[],"trusted":true},"outputs":[],"execution_count":null},{"cell_type":"code","source":"indices = [[i, j] for i in range(num_folds) for j in range(i + 1, num_folds)]\nindices.extend(\n    [\n        [i, j, k]\n        for i in range(num_folds)\n        for j in range(i + 1, num_folds)\n        for k in range(j + 1, num_folds)\n    ]\n)\n\nall_preds = np.array(all_preds)\n\nfor ind in indices:\n    predictions = np.mean(all_preds[ind], axis=0)\n    top_3_indices = np.argsort(predictions, axis=1)[:, -3:] \n    top_3_labels = []\n    for indices in top_3_indices:\n        labels = [num_to_label[idx] for idx in reversed(indices)]  \n        top_3_labels.append(\" \".join(labels)) \n    results_df = pd.DataFrame({\n        \"fname\": test_data[\"fname\"],\n        \"label\": top_3_labels\n    })\n    num = '_'.join([str(i) for i in ind])\n    results_df.to_csv(f\"submission_{num}.csv\", index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T21:20:31.221817Z","iopub.execute_input":"2024-12-10T21:20:31.222207Z","iopub.status.idle":"2024-12-10T21:20:31.268690Z","shell.execute_reply.started":"2024-12-10T21:20:31.222169Z","shell.execute_reply":"2024-12-10T21:20:31.267321Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"predictions = np.mean(all_preds, axis=0)\ntop_3_indices = np.argsort(predictions, axis=1)[:, -3:] \ntop_3_labels = []\nfor indices in top_3_indices:\n    labels = [num_to_label[idx] for idx in reversed(indices)]  \n    top_3_labels.append(\" \".join(labels)) \nresults_df = pd.DataFrame({\"fname\": test_data[\"fname\"], \"label\": top_3_labels})\nnum = \"_\".join([str(i) for i in ind])\nresults_df.to_csv(\"submission.csv\", index=False)\n\nresults_df.head()","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2024-12-10T21:20:36.427990Z","iopub.execute_input":"2024-12-10T21:20:36.428443Z","iopub.status.idle":"2024-12-10T21:20:36.453700Z","shell.execute_reply.started":"2024-12-10T21:20:36.428406Z","shell.execute_reply":"2024-12-10T21:20:36.451776Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}