{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.12.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[],"dockerImageVersionId":28755,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Setup","metadata":{}},{"cell_type":"code","source":"import os\nimport librosa\nimport numpy as np\nimport matplotlib\nmatplotlib.use('Agg')\nimport matplotlib.pyplot as plt\nimport pandas as pd\nfrom tqdm import tqdm\n\ntrain = pd.read_csv('/kaggle/input/competitions/freesound-audio-tagging/train.csv')\n\nSR = 44100\nN_FFT = 2048\nHOP_LENGTH = 512\nN_MELS = 256          # doubled from 128 — finer frequency resolution\n\nOUT_TRAIN = '/kaggle/working/spectrograms/train'\nOUT_TEST  = '/kaggle/working/spectrograms/test/all'\nos.makedirs(OUT_TRAIN, exist_ok=True)\nos.makedirs(OUT_TEST, exist_ok=True)\nfor lbl in train['label'].unique():\n    os.makedirs(f'{OUT_TRAIN}/{lbl}', exist_ok=True)","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true,"execution":{"iopub.status.busy":"2026-08-09T23:18:50.513257Z","iopub.execute_input":"2026-08-09T23:18:50.513574Z","iopub.status.idle":"2026-08-09T23:18:50.535185Z","shell.execute_reply.started":"2026-08-09T23:18:50.513547Z","shell.execute_reply":"2026-08-09T23:18:50.534347Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Save Function","metadata":{}},{"cell_type":"code","source":"def save_melspec_png(audio_path, out_path,\n                     sr=SR, n_fft=N_FFT, hop_length=HOP_LENGTH, n_mels=N_MELS):\n    y, sr = librosa.load(audio_path, sr=sr)\n    if len(y) == 0:\n        y = np.zeros(sr)\n    mel = librosa.feature.melspectrogram(\n        y=y, sr=sr, n_fft=n_fft, hop_length=hop_length, n_mels=n_mels)\n    mel_db = librosa.power_to_db(mel, ref=np.max)\n\n    # figsize*(dpi) = pixels. (3.84, 2.56) * 100dpi = 384x256 px\n    fig = plt.figure(figsize=(3.84, 2.56), dpi=100)\n    ax = plt.axes([0, 0, 1, 1])\n    ax.set_axis_off()\n    ax.imshow(mel_db, origin='lower', aspect='auto', cmap='magma')\n    fig.savefig(out_path, pad_inches=0)   # note: no bbox_inches='tight' — keeps exact size\n    plt.close(fig)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T23:19:06.602195Z","iopub.execute_input":"2026-08-09T23:19:06.602514Z","iopub.status.idle":"2026-08-09T23:19:06.609735Z","shell.execute_reply.started":"2026-08-09T23:19:06.602484Z","shell.execute_reply":"2026-08-09T23:19:06.608879Z"}},"outputs":[],"execution_count":null},{"cell_type":"code","source":"train.head(5)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T23:19:08.486137Z","iopub.execute_input":"2026-08-09T23:19:08.486899Z","iopub.status.idle":"2026-08-09T23:19:08.497116Z","shell.execute_reply.started":"2026-08-09T23:19:08.486860Z","shell.execute_reply":"2026-08-09T23:19:08.496188Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Generate Train","metadata":{}},{"cell_type":"code","source":"BASE_AUDIO = '/kaggle/input/competitions/freesound-audio-tagging/audio_train'\nfor row in tqdm(train.itertuples(), total=len(train)):\n    out = f'{OUT_TRAIN}/{row.label}/{row.fname.replace(\".wav\", \".png\")}'\n    save_melspec_png(f'{BASE_AUDIO}/{row.fname}', out)","metadata":{"trusted":true,"execution":{"iopub.status.busy":"2026-08-09T23:20:00.098209Z","iopub.execute_input":"2026-08-09T23:20:00.099180Z","iopub.status.idle":"2026-08-09T23:20:12.596355Z","shell.execute_reply.started":"2026-08-09T23:20:00.099140Z","shell.execute_reply":"2026-08-09T23:20:12.595516Z"}},"outputs":[],"execution_count":null},{"cell_type":"markdown","source":"# Generate Test","metadata":{}},{"cell_type":"code","source":"sub = pd.read_csv('/kaggle/input/competitions/freesound-audio-tagging/sample_submission.csv')\nBASE_AUDIO = '/kaggle/input/competitions/freesound-audio-tagging/audio_test'\nfor row in tqdm(sub.itertuples(), total=len(sub)):\n    out = f'{OUT_TEST}/{row.fname.replace(\".wav\", \".png\")}'\n    save_melspec_png(f'{BASE_AUDIO}/{row.fname}', out)","metadata":{"trusted":true},"outputs":[],"execution_count":null}]}