{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.11.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"nvidiaTeslaT4","dataSources":[{"sourceId":8900,"databundleVersionId":862232,"sourceType":"competition"}],"dockerImageVersionId":31153,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":true}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport pandas as pd\nimport librosa\nimport tensorflow as tf\nfrom tensorflow.keras.applications import MobileNetV2\nfrom tensorflow.keras.applications.mobilenet_v2 import preprocess_input\nfrom tensorflow.keras.models import Model\nfrom tensorflow.keras.layers import Dense, GlobalAveragePooling2D, Dropout, Input\nfrom tensorflow.keras.utils import to_categorical\nfrom sklearn.model_selection import train_test_split\nfrom tqdm.notebook import tqdm\n\nTRAIN_CSV = '../input/freesound-audio-tagging/train.csv'\nTRAIN_DIR = '../input/freesound-audio-tagging/audio_train/'\nTEST_DIR = '../input/freesound-audio-tagging/audio_test/'\nSAMPLE_SUB = '../input/freesound-audio-tagging/sample_submission.csv'\n\nN_MELS = 128\nTIME_STEPS = 256\nSR = 32000\nBATCH_SIZE = 32\nEPOCHS = 35\nNUM_CLASSES = 41","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T23:34:25.560636Z","iopub.execute_input":"2025-12-15T23:34:25.560839Z","iopub.status.idle":"2025-12-15T23:34:25.566164Z","shell.execute_reply.started":"2025-12-15T23:34:25.560821Z","shell.execute_reply":"2025-12-15T23:34:25.565387Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"df_train = pd.read_csv(TRAIN_CSV)\ndf_test = pd.read_csv(SAMPLE_SUB)\n\nunique_labels = sorted(df_train['label'].unique())\nlabel2id = {label: i for i, label in enumerate(unique_labels)}\nid2label = {i: label for i, label in enumerate(unique_labels)}\n\ndf_train['label_idx'] = df_train['label'].map(label2id)\n\ndef read_audio(filename, audio_dir):\n    file_path = os.path.join(audio_dir, filename)\n    y, _ = librosa.load(file_path, sr=SR)\n    y, _ = librosa.effects.trim(y)\n    \n    if len(y) < TIME_STEPS * 512:\n        padding = TIME_STEPS * 512 - len(y)\n        y = np.pad(y, (0, padding), 'constant')\n        \n    mels = librosa.feature.melspectrogram(y=y, sr=SR, n_mels=N_MELS, n_fft=2048, hop_length=512)\n    mels = librosa.power_to_db(mels, ref=np.max)\n    \n    mels = (mels - mels.min()) / (mels.max() - mels.min() + 1e-6)\n    return mels.astype(np.float32)\n\nX_all = []\nfor fname in tqdm(df_train['fname'].values):\n    X_all.append(read_audio(fname, TRAIN_DIR))\n\ny_all = df_train['label_idx'].values\n\nX_train_raw, X_val_raw, y_train, y_val = train_test_split(X_all, y_all, test_size=0.2, random_state=42, stratify=y_all)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T23:34:29.425376Z","iopub.execute_input":"2025-12-15T23:34:29.425853Z","iopub.status.idle":"2025-12-15T23:41:10.508648Z","shell.execute_reply.started":"2025-12-15T23:34:29.425825Z","shell.execute_reply":"2025-12-15T23:41:10.507805Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"class DataGenerator(tf.keras.utils.Sequence):\n    def __init__(self, X_data, y_data, batch_size=32, dim=(128, 256), n_classes=41, shuffle=True, augment=False):\n        self.dim = dim\n        self.batch_size = batch_size\n        self.y_data = y_data\n        self.X_data = X_data\n        self.n_classes = n_classes\n        self.shuffle = shuffle\n        self.augment = augment\n        self.on_epoch_end()\n\n    def on_epoch_end(self):\n        self.indexes = np.arange(len(self.X_data))\n        if self.shuffle:\n            np.random.shuffle(self.indexes)\n\n    def __len__(self):\n        return int(np.floor(len(self.X_data) / self.batch_size))\n\n    def __getitem__(self, index):\n        indexes = self.indexes[index*self.batch_size:(index+1)*self.batch_size]\n        X = np.empty((self.batch_size, *self.dim, 3))\n        y = np.empty((self.batch_size, self.n_classes))\n\n        for i, idx in enumerate(indexes):\n            spec = self.X_data[idx]\n            label = self.y_data[idx]\n            \n            spec_len = spec.shape[1]\n            if spec_len > self.dim[1]:\n                if self.augment:\n                    start = np.random.randint(0, spec_len - self.dim[1])\n                else:\n                    start = (spec_len - self.dim[1]) // 2\n                crop = spec[:, start:start+self.dim[1]]\n            else:\n                padding = self.dim[1] - spec_len\n                crop = np.pad(spec, ((0,0), (0, padding)), 'constant')\n            \n            img = np.stack([crop, crop, crop], axis=-1)\n            \n            X[i,] = preprocess_input(img * 255) \n            y[i,] = to_categorical(label, num_classes=self.n_classes)\n\n        if self.augment and np.random.random() > 0.3:\n            lam = np.random.beta(0.2, 0.2, self.batch_size)\n            X2 = X[::-1]\n            y2 = y[::-1]\n            lam = lam.reshape(self.batch_size, 1, 1, 1)\n            X = lam * X + (1 - lam) * X2\n            lam = lam.reshape(self.batch_size, 1)\n            y = lam * y + (1 - lam) * y2\n\n        return X, y\n\ntrain_gen = DataGenerator(X_train_raw, y_train, BATCH_SIZE, (N_MELS, TIME_STEPS), NUM_CLASSES, augment=True)\nval_gen = DataGenerator(X_val_raw, y_val, BATCH_SIZE, (N_MELS, TIME_STEPS), NUM_CLASSES, augment=False, shuffle=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T23:41:59.187427Z","iopub.execute_input":"2025-12-15T23:41:59.188154Z","iopub.status.idle":"2025-12-15T23:41:59.198803Z","shell.execute_reply.started":"2025-12-15T23:41:59.188124Z","shell.execute_reply":"2025-12-15T23:41:59.197895Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"def build_model():\n    base_model = MobileNetV2(input_shape=(N_MELS, TIME_STEPS, 3), include_top=False, weights='imagenet')\n    base_model.trainable = True\n    \n    inp = Input(shape=(N_MELS, TIME_STEPS, 3))\n    x = base_model(inp)\n    x = GlobalAveragePooling2D()(x)\n    x = Dropout(0.2)(x)\n    out = Dense(NUM_CLASSES, activation='softmax')(x)\n    \n    model = Model(inp, out)\n    model.compile(optimizer=tf.keras.optimizers.Adam(learning_rate=3e-4),\n                  loss='categorical_crossentropy',\n                  metrics=['accuracy'])\n    return model\n\nmodel = build_model()\n\ncallbacks = [\n    tf.keras.callbacks.ReduceLROnPlateau(monitor='val_accuracy', factor=0.5, patience=3, verbose=1, min_lr=1e-6),\n    tf.keras.callbacks.EarlyStopping(monitor='val_accuracy', patience=8, restore_best_weights=True),\n    tf.keras.callbacks.ModelCheckpoint('best_model.h5', monitor='val_accuracy', save_best_only=True)\n]\n\nhistory = model.fit(train_gen, validation_data=val_gen, epochs=EPOCHS, callbacks=callbacks)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T23:42:10.153845Z","iopub.execute_input":"2025-12-15T23:42:10.154575Z","iopub.status.idle":"2025-12-15T23:54:12.241589Z","shell.execute_reply.started":"2025-12-15T23:42:10.154545Z","shell.execute_reply":"2025-12-15T23:54:12.240912Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"test_files = df_test['fname'].values\nX_test = []\n\nfor fname in tqdm(test_files):\n    X_test.append(read_audio(fname, TEST_DIR))\n\ntest_predictions = []\n\nfor spec in tqdm(X_test):\n    crops = []\n    spec_len = spec.shape[1]\n    \n    starts = []\n    if spec_len > TIME_STEPS:\n        starts = [0, (spec_len - TIME_STEPS)//2, spec_len - TIME_STEPS]\n    else:\n        starts = [0]\n        \n    for start in starts:\n        if spec_len > TIME_STEPS:\n            c = spec[:, start:start+TIME_STEPS]\n        else:\n            padding = TIME_STEPS - spec_len\n            c = np.pad(spec, ((0,0), (0, padding)), 'constant')\n            \n        img = np.stack([c, c, c], axis=-1)\n        crops.append(preprocess_input(img * 255))\n    \n    crops = np.array(crops)\n    preds = model.predict(crops, verbose=0)\n    avg_pred = preds.mean(axis=0)\n    test_predictions.append(avg_pred)\n\ntest_predictions = np.array(test_predictions)\n\ntop3_labels = []\nfor pred in test_predictions:\n    top3_idx = pred.argsort()[-3:][::-1]\n    top3_names = [id2label[idx] for idx in top3_idx]\n    top3_labels.append(\" \".join(top3_names))\n\ndf_test['label'] = top3_labels\ndf_test.to_csv('submission.csv', index=False)\n","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2025-12-15T23:54:47.346271Z","iopub.execute_input":"2025-12-15T23:54:47.347189Z","iopub.status.idle":"2025-12-16T00:12:57.661416Z","shell.execute_reply.started":"2025-12-15T23:54:47.347156Z","shell.execute_reply":"2025-12-16T00:12:57.660812Z"}},"outputs":[],"execution_count":null}]}