{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":73154,"databundleVersionId":8047843,"sourceType":"competition"}],"dockerImageVersionId":30673,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport torch\nimport torchaudio\nimport matplotlib.pyplot as plt\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\nimport pandas as pd\n\ndf = pd.read_csv('/kaggle/input/itmo-acoustic-event-detection-2024/train.csv');\ndf","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-03-28T07:13:39.689879Z","iopub.execute_input":"2024-03-28T07:13:39.690337Z","iopub.status.idle":"2024-03-28T07:13:46.422080Z","shell.execute_reply.started":"2024-03-28T07:13:39.690303Z","shell.execute_reply":"2024-03-28T07:13:46.420210Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def print_stats(waveform, sample_rate=None, src=None):\n  if src:\n    print(\"-\" * 10)\n    print(\"Source:\", src)\n    print(\"-\" * 10)\n  if sample_rate:\n    print(\"Sample Rate:\", sample_rate)\n  print(\"Shape:\", tuple(waveform.shape))\n  print(\"Dtype:\", waveform.dtype)\n  print(f\" - Max:     {waveform.max().item():6.3f}\")\n  print(f\" - Min:     {waveform.min().item():6.3f}\")\n  print(f\" - Mean:    {waveform.mean().item():6.3f}\")\n  print(f\" - Std Dev: {waveform.std().item():6.3f}\")\n  print()\n  print(waveform)\n  print(waveform.shape)\n  print()\n\ndef plot_waveform(waveform, sample_rate, title=\"Waveform\", xlim=None, ylim=None):\n  waveform = waveform.numpy()\n\n  num_channels, num_frames = waveform.shape\n  time_axis = torch.arange(0, num_frames) / sample_rate\n\n  figure, axes = plt.subplots(num_channels, 1)\n  if num_channels == 1:\n    axes = [axes]\n  for c in range(num_channels):\n    axes[c].plot(time_axis, waveform[c], linewidth=1)\n    axes[c].grid(True)\n    if num_channels > 1:\n      axes[c].set_ylabel(f'Channel {c+1}')\n    if xlim:\n      axes[c].set_xlim(xlim)\n    if ylim:\n      axes[c].set_ylim(ylim)\n  figure.suptitle(title)\n  plt.show(block=False)\n\ndef plot_specgram(waveform, sample_rate, title=\"Spectrogram\", xlim=None):\n  waveform = waveform.numpy()\n\n  num_channels, num_frames = waveform.shape\n  time_axis = torch.arange(0, num_frames) / sample_rate\n\n  figure, axes = plt.subplots(num_channels, 1)\n  if num_channels == 1:\n    axes = [axes]\n  for c in range(num_channels):\n    axes[c].specgram(waveform[c], Fs=sample_rate)\n    if num_channels > 1:\n      axes[c].set_ylabel(f'Channel {c+1}')\n    if xlim:\n      axes[c].set_xlim(xlim)\n  figure.suptitle(title)\n  plt.show(block=False)\n","metadata":{"execution":{"iopub.status.busy":"2024-03-28T07:22:25.517379Z","iopub.execute_input":"2024-03-28T07:22:25.517950Z","iopub.status.idle":"2024-03-28T07:22:25.538445Z","shell.execute_reply.started":"2024-03-28T07:22:25.517908Z","shell.execute_reply":"2024-03-28T07:22:25.536984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T07:22:27.107597Z","iopub.execute_input":"2024-03-28T07:22:27.108130Z","iopub.status.idle":"2024-03-28T07:22:27.123303Z","shell.execute_reply.started":"2024-03-28T07:22:27.108093Z","shell.execute_reply":"2024-03-28T07:22:27.121844Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df['label'].value_counts().shape","metadata":{"execution":{"iopub.status.busy":"2024-03-28T07:22:27.357244Z","iopub.execute_input":"2024-03-28T07:22:27.357800Z","iopub.status.idle":"2024-03-28T07:22:27.368802Z","shell.execute_reply.started":"2024-03-28T07:22:27.357729Z","shell.execute_reply":"2024-03-28T07:22:27.367209Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"! ls '/kaggle/input/itmo-acoustic-event-detection-2024'","metadata":{"execution":{"iopub.status.busy":"2024-03-28T07:22:27.716158Z","iopub.execute_input":"2024-03-28T07:22:27.716629Z","iopub.status.idle":"2024-03-28T07:22:28.882478Z","shell.execute_reply.started":"2024-03-28T07:22:27.716594Z","shell.execute_reply":"2024-03-28T07:22:28.880673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sample_file = f'/kaggle/input/itmo-acoustic-event-detection-2024/audio_train/train/{df[\"fname\"][2]}'\nsample_file","metadata":{"execution":{"iopub.status.busy":"2024-03-28T07:22:28.885264Z","iopub.execute_input":"2024-03-28T07:22:28.885677Z","iopub.status.idle":"2024-03-28T07:22:28.895696Z","shell.execute_reply.started":"2024-03-28T07:22:28.885643Z","shell.execute_reply":"2024-03-28T07:22:28.894301Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"waveform, sample_rate = torchaudio.load(sample_file)\n\nprint_stats(waveform, sample_rate=sample_rate)","metadata":{"execution":{"iopub.status.busy":"2024-03-28T07:22:29.074778Z","iopub.execute_input":"2024-03-28T07:22:29.075230Z","iopub.status.idle":"2024-03-28T07:22:29.111942Z","shell.execute_reply.started":"2024-03-28T07:22:29.075198Z","shell.execute_reply":"2024-03-28T07:22:29.109054Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_waveform(waveform, sample_rate)\nplot_specgram(waveform, sample_rate)","metadata":{"execution":{"iopub.status.busy":"2024-03-28T07:22:30.059321Z","iopub.execute_input":"2024-03-28T07:22:30.059787Z","iopub.status.idle":"2024-03-28T07:22:41.031098Z","shell.execute_reply.started":"2024-03-28T07:22:30.059737Z","shell.execute_reply":"2024-03-28T07:22:41.029585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rand_label_list = list()\nrand_wav_list = list()\n\nfor label,group in df.groupby(['label']):\n    \n    print('label : ',label,group.shape[0])\n    \n    rand_label_list.append( label[0] )\n    rand_wav_list.append( group.sample(1)['fname'].values[0] )\n    \nrand_df = pd.DataFrame({'fname':rand_wav_list, 'label':rand_label_list})\nrand_df","metadata":{"execution":{"iopub.status.busy":"2024-03-28T08:57:43.194286Z","iopub.execute_input":"2024-03-28T08:57:43.194667Z","iopub.status.idle":"2024-03-28T08:57:43.229361Z","shell.execute_reply.started":"2024-03-28T08:57:43.194638Z","shell.execute_reply":"2024-03-28T08:57:43.228528Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualization of spectrogram","metadata":{}},{"cell_type":"code","source":"# settings\nh, w = 10, 10        # for raster image\nnrows, ncols = 5, 4  # array of sub-plots\nfigsize = [8, 8]     # figure size, inches\n\n# create figure (fig), and array of axes (ax)\nfig, ax = plt.subplots(nrows=nrows, ncols=ncols, figsize=figsize)\n\n# plot simple raster image on each sub-plot\nfor i, axi in enumerate(ax.flat):\n    # get indices of row/column\n    rowid = i // ncols\n    colid = i % ncols\n    \n    wav_file, label =rand_df.iloc[i]['fname'], rand_df.iloc[i]['label']\n    wav_file = f'/kaggle/input/itmo-acoustic-event-detection-2024/audio_train/train/{wav_file}'\n    waveform, sample_rate = torchaudio.load(wav_file)\n    \n    axi.specgram(waveform.numpy()[0], Fs=sample_rate)\n    # write row/col indices as axes' title for identification\n    axi.set_title(label)\n    axi.axes.get_xaxis().set_visible(False)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T09:03:48.200405Z","iopub.execute_input":"2024-03-28T09:03:48.200911Z","iopub.status.idle":"2024-03-28T09:03:51.566489Z","shell.execute_reply.started":"2024-03-28T09:03:48.200874Z","shell.execute_reply":"2024-03-28T09:03:51.562079Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# settings\nh, w = 10, 10        # for raster image\nnrows, ncols = 5, 4  # array of sub-plots\nfigsize = [8, 8]     # figure size, inches\n\n# create figure (fig), and array of axes (ax)\nfig, ax = plt.subplots(nrows=nrows, ncols=ncols, figsize=figsize)\n\n# plot simple raster image on each sub-plot\nfor i, axi in enumerate(ax.flat):\n    # get indices of row/column\n    rowid = i // ncols\n    colid = i % ncols\n    \n    wav_file, label =rand_df[20:].iloc[i]['fname'], rand_df[20:].iloc[i]['label']\n    wav_file = f'/kaggle/input/itmo-acoustic-event-detection-2024/audio_train/train/{wav_file}'\n    waveform, sample_rate = torchaudio.load(wav_file)\n    \n    axi.specgram(waveform.numpy()[0], Fs=sample_rate)\n    # write row/col indices as axes' title for identification\n    axi.set_title(label)\n    axi.axes.get_xaxis().set_visible(False)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T09:04:59.381447Z","iopub.execute_input":"2024-03-28T09:04:59.382772Z","iopub.status.idle":"2024-03-28T09:05:02.392253Z","shell.execute_reply.started":"2024-03-28T09:04:59.382705Z","shell.execute_reply":"2024-03-28T09:05:02.390943Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wav_file, label =rand_df[40:].iloc[0]['fname'], rand_df[40:].iloc[0]['label']\nwav_file = f'/kaggle/input/itmo-acoustic-event-detection-2024/audio_train/train/{wav_file}'\nwaveform, sample_rate = torchaudio.load(wav_file)\n\nprint('label : ',label)\nplot_specgram(waveform, sample_rate)","metadata":{"execution":{"iopub.status.busy":"2024-03-28T09:58:42.221406Z","iopub.execute_input":"2024-03-28T09:58:42.221940Z","iopub.status.idle":"2024-03-28T09:58:42.706463Z","shell.execute_reply.started":"2024-03-28T09:58:42.221904Z","shell.execute_reply":"2024-03-28T09:58:42.705274Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Visualization of waveform","metadata":{}},{"cell_type":"code","source":"# settings\nh, w = 10, 10        # for raster image\nnrows, ncols = 5, 4  # array of sub-plots\nfigsize = [8, 8]     # figure size, inches\n\n# create figure (fig), and array of axes (ax)\nfig, ax = plt.subplots(nrows=nrows, ncols=ncols, figsize=figsize)\n\n# plot simple raster image on each sub-plot\nfor i, axi in enumerate(ax.flat):\n    # get indices of row/column\n    rowid = i // ncols\n    colid = i % ncols\n    \n    wav_file, label = rand_df.iloc[i]['fname'], rand_df.iloc[i]['label']\n    wav_file = f'/kaggle/input/itmo-acoustic-event-detection-2024/audio_train/train/{wav_file}'\n    waveform, sample_rate = torchaudio.load(wav_file)\n    num_channels, num_frames = waveform.shape\n    time_axis = torch.arange(0, num_frames) / sample_rate\n    axi.plot(time_axis, waveform.numpy()[0], linewidth=1)\n    axi.grid(True)\n    # write row/col indices as axes' title for identification\n    axi.set_title(label)\n    axi.axes.get_xaxis().set_visible(False)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T09:54:19.109186Z","iopub.execute_input":"2024-03-28T09:54:19.109668Z","iopub.status.idle":"2024-03-28T09:55:42.432578Z","shell.execute_reply.started":"2024-03-28T09:54:19.109635Z","shell.execute_reply":"2024-03-28T09:55:42.431193Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# settings\nh, w = 10, 10        # for raster image\nnrows, ncols = 5, 4  # array of sub-plots\nfigsize = [8, 8]     # figure size, inches\n\n# create figure (fig), and array of axes (ax)\nfig, ax = plt.subplots(nrows=nrows, ncols=ncols, figsize=figsize)\n\n# plot simple raster image on each sub-plot\nfor i, axi in enumerate(ax.flat):\n    # get indices of row/column\n    rowid = i // ncols\n    colid = i % ncols\n    \n    wav_file, label = rand_df[20:].iloc[i]['fname'], rand_df[20:].iloc[i]['label']\n    wav_file = f'/kaggle/input/itmo-acoustic-event-detection-2024/audio_train/train/{wav_file}'\n    waveform, sample_rate = torchaudio.load(wav_file)\n    num_channels, num_frames = waveform.shape\n    time_axis = torch.arange(0, num_frames) / sample_rate\n    axi.plot(time_axis, waveform.numpy()[0], linewidth=1)\n    axi.grid(True)\n    # write row/col indices as axes' title for identification\n    axi.set_title(label)\n    axi.axes.get_xaxis().set_visible(False)\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-03-28T09:56:53.108293Z","iopub.execute_input":"2024-03-28T09:56:53.109070Z","iopub.status.idle":"2024-03-28T09:57:56.079366Z","shell.execute_reply.started":"2024-03-28T09:56:53.109024Z","shell.execute_reply":"2024-03-28T09:57:56.077813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"wav_file, label =rand_df[40:].iloc[0]['fname'], rand_df[40:].iloc[0]['label']\nwav_file = f'/kaggle/input/itmo-acoustic-event-detection-2024/audio_train/train/{wav_file}'\nwaveform, sample_rate = torchaudio.load(wav_file)\n\nprint('label : ',label)\n\nplot_waveform(waveform, sample_rate)","metadata":{"execution":{"iopub.status.busy":"2024-03-28T09:58:58.908134Z","iopub.execute_input":"2024-03-28T09:58:58.909550Z","iopub.status.idle":"2024-03-28T09:59:08.162145Z","shell.execute_reply.started":"2024-03-28T09:58:58.909493Z","shell.execute_reply":"2024-03-28T09:59:08.160555Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}