{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":73047,"databundleVersionId":8149390,"sourceType":"competition"},{"sourceId":8217967,"sourceType":"datasetVersion","datasetId":4871244}],"dockerImageVersionId":30698,"isInternetEnabled":false,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport pandas as pd\nimport librosa\nfrom tqdm import tqdm\nimport numpy as np\nimport random\nfrom pydub import AudioSegment\nimport librosa\nimport matplotlib.pyplot as plt\nimport os\nimport librosa\nfrom multiprocessing import Pool\nimport time\nimport datasets\nfrom datasets import Dataset\nimport matplotlib.pyplot as plt\nimport seaborn as sns\nfrom wordcloud import WordCloud, STOPWORDS","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:56:21.234449Z","iopub.execute_input":"2024-04-24T16:56:21.235924Z","iopub.status.idle":"2024-04-24T16:56:21.243103Z","shell.execute_reply.started":"2024-04-24T16:56:21.235853Z","shell.execute_reply":"2024-04-24T16:56:21.241728Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BASE_DIR = '/kaggle/input/ben10/ben10'\ntrain_data_dir = f\"{BASE_DIR}/16_kHz_train_audio/\"\ntest_data_dir = f\"{BASE_DIR}/16_kHz_valid_audio/\"\ndata_path = f\"{BASE_DIR}/train.csv\"","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:56:21.533342Z","iopub.execute_input":"2024-04-24T16:56:21.533802Z","iopub.status.idle":"2024-04-24T16:56:21.540118Z","shell.execute_reply.started":"2024-04-24T16:56:21.533767Z","shell.execute_reply":"2024-04-24T16:56:21.538617Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"split2path = {\n    \"train\": train_data_dir,\n    \"test\": test_data_dir,\n}","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:56:21.845953Z","iopub.execute_input":"2024-04-24T16:56:21.846409Z","iopub.status.idle":"2024-04-24T16:56:21.852359Z","shell.execute_reply.started":"2024-04-24T16:56:21.846377Z","shell.execute_reply":"2024-04-24T16:56:21.851084Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = pd.read_csv(data_path)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:56:22.329896Z","iopub.execute_input":"2024-04-24T16:56:22.330436Z","iopub.status.idle":"2024-04-24T16:56:22.499006Z","shell.execute_reply.started":"2024-04-24T16:56:22.330394Z","shell.execute_reply":"2024-04-24T16:56:22.497795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_split(filename):\n    filename_ = filename.split(\"_\")\n    split = filename_[0]\n    return split\n\ndef extract_district(filename):\n    filename_ = filename.split(\" \")[0]\n    district = filename_.split(\"_\")[1]\n    return district\n\ndef beautify_dataset(data):\n    splits = []\n    districts = []\n    newpaths = []\n    transcripts = []\n    \n    for i in range(len(data)):\n        filename, transcript = data.iloc[i]\n        split = extract_split(filename)\n        district = extract_district(filename)\n        dir_path = split2path[split]\n        composed_path = f\"{dir_path}{filename}\"\n        \n        if os.path.exists(composed_path) == False:\n            print(f\"{composed_path} does not exist.\")\n            continue\n        \n        # replace any newline characters\n        transcript = transcript.replace(\"\\n\", \" \")\n        transcript = \" \".join(transcript.split())\n        \n        splits.append(split)\n        districts.append(district)\n        newpaths.append(composed_path)\n        transcripts.append(transcript)\n    \n    data['file_path'] = newpaths\n    data['district'] = districts\n    data['split'] = splits\n    data['transcripts'] = transcripts\n    \n#     data.drop(columns=['file_name'], inplace=True)\n    \n    return data","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:56:22.578885Z","iopub.execute_input":"2024-04-24T16:56:22.579271Z","iopub.status.idle":"2024-04-24T16:56:22.588220Z","shell.execute_reply.started":"2024-04-24T16:56:22.579242Z","shell.execute_reply":"2024-04-24T16:56:22.586947Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"data = beautify_dataset(data)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:56:22.826386Z","iopub.execute_input":"2024-04-24T16:56:22.826838Z","iopub.status.idle":"2024-04-24T16:56:42.510258Z","shell.execute_reply.started":"2024-04-24T16:56:22.826802Z","shell.execute_reply":"2024-04-24T16:56:42.509012Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/ben10/ben10/train.csv\")\ntest = pd.read_csv(\"/kaggle/input/ben10/sample_submission.csv\")\n\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:56:42.512404Z","iopub.execute_input":"2024-04-24T16:56:42.512882Z","iopub.status.idle":"2024-04-24T16:56:42.675166Z","shell.execute_reply.started":"2024-04-24T16:56:42.512838Z","shell.execute_reply":"2024-04-24T16:56:42.673956Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Total samples in Train set :\",df.shape[0])\nprint(\"Total samples in test set\",len(os.listdir(\"/kaggle/input/ben10/ben10/16_kHz_valid_audio\")))","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:56:42.676423Z","iopub.execute_input":"2024-04-24T16:56:42.676794Z","iopub.status.idle":"2024-04-24T16:56:42.683859Z","shell.execute_reply.started":"2024-04-24T16:56:42.676766Z","shell.execute_reply":"2024-04-24T16:56:42.682280Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def extract_regions(path):\n    unwanted_strs = [\"train_\",\"valid_\",\".wav\",\"1\",\"2\",\"3\",\"4\",\"5\",\"6\",\"7\",\"8\",\"9\",\"0\",\"(\",\")\",\" \"]\n    for i in unwanted_strs:\n        path = path.replace(i,\"\")\n    \n    return path\ndf[\"region\"] = df[\"file_name\"].apply(lambda x:extract_regions(x))\ntest[\"region\"] = test[\"id\"].apply(lambda x:extract_regions(x))\nlist(df[\"region\"].unique())","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:56:42.687121Z","iopub.execute_input":"2024-04-24T16:56:42.687984Z","iopub.status.idle":"2024-04-24T16:56:42.743822Z","shell.execute_reply.started":"2024-04-24T16:56:42.687941Z","shell.execute_reply":"2024-04-24T16:56:42.742588Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# EDA on transcript","metadata":{}},{"cell_type":"code","source":"custom_palette = sns.color_palette(\"pastel\")\n\n# Set the background style\nsns.set_style(\"whitegrid\")\n\n# Set font scale\nsns.set(font_scale=1.4)\n\n# Create the subplots\nfig, axes = plt.subplots(1, 2, figsize=(12, 6))\n\n# Plot the distribution of regions in the training set\ndf['region'].value_counts().sort_values().plot(kind='barh', ax=axes[0], color=custom_palette)\naxes[0].set_title('Distribution of Regions in training set')\n\n# Plot the distribution of regions in the test set\ntest['region'].value_counts().sort_values().plot(kind='barh', ax=axes[1], color=custom_palette)\naxes[1].set_title('Distribution of Regions in test')\n\n# Add labels and title with increased font size\nfor ax in axes:\n    ax.set_xlabel(\"Number of Samples\", labelpad=12, fontsize=14)\n    ax.set_ylabel(\"Region\", labelpad=12, fontsize=14)\n    ax.xaxis.set_tick_params(labelsize=12)\n    ax.yaxis.set_tick_params(labelsize=12)\n\n# Tight layout\nplt.tight_layout()\n\n# Show the plot\nplt.show()\n","metadata":{"execution":{"iopub.status.busy":"2024-04-24T17:03:28.337851Z","iopub.execute_input":"2024-04-24T17:03:28.339117Z","iopub.status.idle":"2024-04-24T17:03:29.160178Z","shell.execute_reply.started":"2024-04-24T17:03:28.339077Z","shell.execute_reply":"2024-04-24T17:03:29.158830Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_dir = \"/kaggle/input/ben10/ben10/16_kHz_train_audio/\"\npaths = [train_dir+path for path in os.listdir(train_dir)]\n#print(\"First two path : \",paths[:2])\n\ndef get_duration(file):\n    try:\n        # Load audio file\n        y, sr = librosa.load(file, sr=None)\n        # Calculate duration\n        duration = librosa.get_duration(y=y, sr=sr)\n        return duration\n    except Exception as e:\n        return file, None\n\ndef get_durations_parallel(files):\n    with Pool() as pool:\n        results = pool.map(get_duration, files)\n    return results\n\n\n#Getting durations here. ALso checking how much time does it take to laod all the audios\nstart = time.time()\ndurations_train = get_durations_parallel(paths)\ntest_dir = \"/kaggle/input/ben10/ben10/16_kHz_valid_audio/\"\ntest_paths = [test_dir+path for path in os.listdir(test_dir)]\ndurations_test = get_durations_parallel(test_paths)\n","metadata":{"execution":{"iopub.status.busy":"2024-04-24T17:02:42.101474Z","iopub.execute_input":"2024-04-24T17:02:42.102007Z","iopub.status.idle":"2024-04-24T17:03:28.335294Z","shell.execute_reply.started":"2024-04-24T17:02:42.101971Z","shell.execute_reply":"2024-04-24T17:03:28.333732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_style(\"whitegrid\")\n\n# Set font scale\nsns.set(font_scale=1.4)\n\n# Create the subplots\nfig, axes = plt.subplots(1, 2, figsize=(12, 6))\n\n# Plotting the training data histogram\nsns.histplot(durations_train, bins=[i for i in range(0, 31, 3)], ax=axes[0])\naxes[0].set_title('Training audio file durations')\naxes[0].set_xlabel('Duration (s)')\naxes[0].set_ylabel('Frequency')\n\n# Plotting the test data histogram\nsns.histplot(durations_test, bins=[i for i in range(0, 31, 3)], ax=axes[1])\naxes[1].set_title('Test audio file durations')\naxes[1].set_xlabel('Duration (s)')\naxes[1].set_ylabel('Frequency')\n\n# Tight layout\nplt.tight_layout()\n\n# Show the plot\nplt.show()\n\n","metadata":{"execution":{"iopub.status.busy":"2024-04-24T17:05:48.357992Z","iopub.execute_input":"2024-04-24T17:05:48.358924Z","iopub.status.idle":"2024-04-24T17:05:49.185680Z","shell.execute_reply.started":"2024-04-24T17:05:48.358886Z","shell.execute_reply":"2024-04-24T17:05:49.184194Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.set_style(\"whitegrid\")\n\nsns.set(font_scale=1.4)\n\n\ndf['len'] = df['transcripts'].apply(lambda x: len(x.split()))\n\nplt.figure(figsize=(8, 6))\nsns.histplot(df['len'], bins=[i for i in range(0, 151, 10)], kde=False)\nplt.xticks(np.arange(0, 150, step=10))\nplt.xlabel('Sentence Length')\nplt.title('Train Sentence Length Distributions')\n\n# Show the plot\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T17:06:17.357842Z","iopub.execute_input":"2024-04-24T17:06:17.358282Z","iopub.status.idle":"2024-04-24T17:06:17.882434Z","shell.execute_reply.started":"2024-04-24T17:06:17.358252Z","shell.execute_reply":"2024-04-24T17:06:17.880983Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['len']==0]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:57:18.351868Z","iopub.status.idle":"2024-04-24T16:57:18.352536Z","shell.execute_reply.started":"2024-04-24T16:57:18.352263Z","shell.execute_reply":"2024-04-24T16:57:18.352287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['len']<10]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:57:18.354011Z","iopub.status.idle":"2024-04-24T16:57:18.354463Z","shell.execute_reply.started":"2024-04-24T16:57:18.354246Z","shell.execute_reply":"2024-04-24T16:57:18.354262Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df[df['len']>100]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:57:18.355667Z","iopub.status.idle":"2024-04-24T16:57:18.356156Z","shell.execute_reply.started":"2024-04-24T16:57:18.355928Z","shell.execute_reply":"2024-04-24T16:57:18.355946Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- there are some very short and very large transcripts that need to be handled","metadata":{}},{"cell_type":"code","source":"vocab = {}\nfor sen in df.transcripts:\n    for j in sen.split(\" \"):\n        try:\n            vocab[j]+=1\n        except:\n            vocab[j]=1\nprint(\"Total words in vocabulary : \",len(vocab))\n\nsorted_vocab = sorted(vocab.items(),key = lambda kv:kv[1],reverse=True)\nsorted_vocab[:10]","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:57:18.357452Z","iopub.status.idle":"2024-04-24T16:57:18.358115Z","shell.execute_reply.started":"2024-04-24T16:57:18.357828Z","shell.execute_reply":"2024-04-24T16:57:18.357851Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\ndef plot_world(text):\n\n    wordcloud = WordCloud(width = 500, height = 500, \n                    background_color ='black', \n                    font_path=\"/kaggle/input/notosansfont/NotoSansBengali_Condensed-Regular.ttf\",\n                    min_font_size = 10).generate(text) \n\n    # plot the WordCloud image                        \n    plt.figure(figsize = (5, 5), facecolor = 'k', edgecolor = 'k' ) \n    plt.imshow(wordcloud) \n    plt.axis(\"off\") \n    plt.tight_layout(pad = 0) \n\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:57:18.360899Z","iopub.status.idle":"2024-04-24T16:57:18.362299Z","shell.execute_reply.started":"2024-04-24T16:57:18.361952Z","shell.execute_reply":"2024-04-24T16:57:18.361974Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"s= \" \".join(df.transcripts[:100])\nplot_world(s)","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:25:08.218915Z","iopub.execute_input":"2024-04-24T16:25:08.219330Z","iopub.status.idle":"2024-04-24T16:25:10.508017Z","shell.execute_reply.started":"2024-04-24T16:25:08.219303Z","shell.execute_reply":"2024-04-24T16:25:10.506732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chars = {}\nfor sen in df.transcripts:\n    for j in sen:\n        try:\n            chars[j]+=1\n        except:\n            chars[j]=1\nchars","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:08:42.629622Z","iopub.execute_input":"2024-04-24T16:08:42.630053Z","iopub.status.idle":"2024-04-24T16:08:43.216706Z","shell.execute_reply.started":"2024-04-24T16:08:42.630020Z","shell.execute_reply":"2024-04-24T16:08:43.215542Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"- english characters and numbers need to be removed","metadata":{}},{"cell_type":"markdown","source":"# EDA on Audio","metadata":{}},{"cell_type":"code","source":"import random\n\ndef display_random_audio(region):\n    sample = df[df[\"region\"] == region]\n    idx = random.randint(0, len(sample) - 1)\n\n    file = sample['file_name'].iloc[idx]\n    path = train_dir + file\n    \n    print(\"Region:\", region)\n    display(AudioSegment.from_file(path))\n    print(\"Transcript:\", sample['transcripts'].iloc[idx])\n","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:29:52.265766Z","iopub.execute_input":"2024-04-24T16:29:52.266213Z","iopub.status.idle":"2024-04-24T16:29:52.273479Z","shell.execute_reply.started":"2024-04-24T16:29:52.266181Z","shell.execute_reply":"2024-04-24T16:29:52.272283Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"regions = ['sandwip', 'barishal', 'chittagong', 'habiganj', 'kishoreganj', \n           'narail', 'narsingdi', 'rangpur', 'sylhet', 'tangail']\n\nfor region in regions:\n    display_random_audio(region)\n    print(\"\\n\")\n","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:29:53.265757Z","iopub.execute_input":"2024-04-24T16:29:53.266143Z","iopub.status.idle":"2024-04-24T16:29:55.043042Z","shell.execute_reply.started":"2024-04-24T16:29:53.266113Z","shell.execute_reply":"2024-04-24T16:29:55.041829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = data['file_path'].iloc[0]\n\n# Load the audio file\ny, sr = librosa.load(file_path, sr=None)\n\n# Compute the mel spectrogram\nS = librosa.feature.melspectrogram(y=y, sr=sr)\n\n# Convert to decibels\nS_dB = librosa.power_to_db(S, ref=np.max)\n\n# Plot the mel spectrogram\nplt.figure(figsize=(10, 6))\nlibrosa.display.specshow(S_dB, sr=sr, x_axis='time', y_axis='mel',cmap='viridis')\nplt.colorbar(format='%+2.0f dB')\nplt.title('Mel spectrogram')\nplt.xlabel('Time (s)')\nplt.ylabel('Mel frequency')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:45:02.101206Z","iopub.execute_input":"2024-04-24T16:45:02.102851Z","iopub.status.idle":"2024-04-24T16:45:02.605141Z","shell.execute_reply.started":"2024-04-24T16:45:02.102806Z","shell.execute_reply":"2024-04-24T16:45:02.603902Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"file_path = data['file_path'].iloc[0]\n\n# Load the audio file\ny, sr = librosa.load(file_path, sr=None)\n\n# Compute the log power spectrogram\nS = librosa.stft(y)\nS_db = librosa.amplitude_to_db(abs(S))\n\n# Compute the spectral centroid\ncent = librosa.feature.spectral_centroid(y=y, sr=sr)\n\n# Plot the log power spectrogram with the spectral centroid overlay\ntimes = librosa.times_like(cent)\nfig, ax = plt.subplots()\nimg = librosa.display.specshow(S_db, y_axis='log', x_axis='time', ax=ax, cmap='viridis')\n\n# Plot the spectral centroid\nax.plot(times, cent.T, label='Spectral centroid', color='w')\n\n# Add colorbar and legend\nfig.colorbar(img, ax=ax, format='%+2.0f dB')\nax.legend(loc='upper right')\n\n# Set title and labels\nax.set(title='Log Power Spectrogram with Spectral Centroid Overlay')\nplt.xlabel('Time (s)')\nplt.ylabel('Frequency (Hz)')\n\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2024-04-24T16:49:33.599245Z","iopub.execute_input":"2024-04-24T16:49:33.599684Z","iopub.status.idle":"2024-04-24T16:49:34.409787Z","shell.execute_reply.started":"2024-04-24T16:49:33.599649Z","shell.execute_reply":"2024-04-24T16:49:34.408959Z"},"trusted":true},"execution_count":null,"outputs":[]}]}