{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.12","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# import pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:49:50.511531Z","iopub.execute_input":"2023-09-01T09:49:50.511930Z","iopub.status.idle":"2023-09-01T09:49:50.516529Z","shell.execute_reply.started":"2023-09-01T09:49:50.511901Z","shell.execute_reply":"2023-09-01T09:49:50.515495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import numpy as np","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:49:51.440805Z","iopub.execute_input":"2023-09-01T09:49:51.441268Z","iopub.status.idle":"2023-09-01T09:49:51.446333Z","shell.execute_reply.started":"2023-09-01T09:49:51.441230Z","shell.execute_reply":"2023-09-01T09:49:51.445240Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# !pip install keras_preprocessing","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:49:51.635491Z","iopub.execute_input":"2023-09-01T09:49:51.636474Z","iopub.status.idle":"2023-09-01T09:49:51.640796Z","shell.execute_reply.started":"2023-09-01T09:49:51.636433Z","shell.execute_reply":"2023-09-01T09:49:51.639738Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import keras_preprocessing","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:49:51.825755Z","iopub.execute_input":"2023-09-01T09:49:51.826149Z","iopub.status.idle":"2023-09-01T09:49:51.831671Z","shell.execute_reply.started":"2023-09-01T09:49:51.826103Z","shell.execute_reply":"2023-09-01T09:49:51.830511Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import librosa\n# import librosa.display\n# import matplotlib.pyplot as plt\n# from keras.preprocessing.text import Tokenizer\n# from keras_preprocessing.sequence import pad_sequences\n# from keras.utils import to_categorical\n# import os\n# from sklearn.model_selection import train_test_split\n# from keras.models import Sequential\n# from keras.layers import Dense, LSTM, Embedding, TimeDistributed, Dropout\n# from keras.optimizers import Adam\n# from keras.callbacks import EarlyStopping","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:49:53.930926Z","iopub.execute_input":"2023-09-01T09:49:53.931655Z","iopub.status.idle":"2023-09-01T09:49:53.937483Z","shell.execute_reply.started":"2023-09-01T09:49:53.931615Z","shell.execute_reply":"2023-09-01T09:49:53.935458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# dir(keras_preprocessing)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:49:54.065923Z","iopub.execute_input":"2023-09-01T09:49:54.066327Z","iopub.status.idle":"2023-09-01T09:49:54.071437Z","shell.execute_reply.started":"2023-09-01T09:49:54.066297Z","shell.execute_reply":"2023-09-01T09:49:54.070273Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Path to the train_mp3s folder\n# train_mp3s_folder = '/kaggle/input/bengaliai-speech/train_mp3s/'\n# test_mp3s_folder = '/kaggle/input/bengaliai-speech/test_mp3s/'","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:49:54.280452Z","iopub.execute_input":"2023-09-01T09:49:54.280763Z","iopub.status.idle":"2023-09-01T09:49:54.285741Z","shell.execute_reply.started":"2023-09-01T09:49:54.280737Z","shell.execute_reply":"2023-09-01T09:49:54.284712Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # List of audio file names in the train_mp3s folder\n# audio_files = [f for f in os.listdir(train_mp3s_folder) if f.endswith('.mp3')]\n\n# # Loop through each audio file\n# for audio_file in audio_files[:5]:\n#     # Load the audio file using librosa\n#     audio_path = os.path.join(train_mp3s_folder, audio_file)\n#     y, sr = librosa.load(audio_path, sr=None)  # sr=None retains the original sample rate\n\n#     # Visualize the waveform\n#     plt.figure(figsize=(10, 4))\n#     librosa.display.waveshow(y, sr=sr)\n#     plt.title('Waveform of ' + audio_file)\n#     plt.xlabel('Time (s)')\n#     plt.ylabel('Amplitude')\n#     plt.show()\n\n#     # Convert to mono\n#     y_mono = librosa.to_mono(y)\n\n#     # Visualize the spectrogram\n#     plt.figure(figsize=(10, 4))\n#     D = librosa.amplitude_to_db(librosa.stft(y_mono), ref=np.max)\n#     librosa.display.specshow(D, sr=sr, x_axis='time', y_axis='log')\n#     plt.colorbar(format='%+2.0f dB')\n#     plt.title('Spectrogram of ' + audio_file)\n#     plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:49:54.385535Z","iopub.execute_input":"2023-09-01T09:49:54.386216Z","iopub.status.idle":"2023-09-01T09:49:54.391888Z","shell.execute_reply.started":"2023-09-01T09:49:54.386182Z","shell.execute_reply":"2023-09-01T09:49:54.390795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_csv_path = '/kaggle/input/bengaliai-speech/train.csv'","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:49:57.630923Z","iopub.execute_input":"2023-09-01T09:49:57.631632Z","iopub.status.idle":"2023-09-01T09:49:57.636566Z","shell.execute_reply.started":"2023-09-01T09:49:57.631598Z","shell.execute_reply":"2023-09-01T09:49:57.635363Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# train_df = pd.read_csv(train_csv_path)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:49:57.775430Z","iopub.execute_input":"2023-09-01T09:49:57.775732Z","iopub.status.idle":"2023-09-01T09:49:57.779943Z","shell.execute_reply.started":"2023-09-01T09:49:57.775706Z","shell.execute_reply":"2023-09-01T09:49:57.778944Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:49:57.925900Z","iopub.execute_input":"2023-09-01T09:49:57.926597Z","iopub.status.idle":"2023-09-01T09:49:57.931420Z","shell.execute_reply.started":"2023-09-01T09:49:57.926565Z","shell.execute_reply":"2023-09-01T09:49:57.930454Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df['sentence'].nunique() ","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:49:59.006852Z","iopub.execute_input":"2023-09-01T09:49:59.007279Z","iopub.status.idle":"2023-09-01T09:49:59.011774Z","shell.execute_reply.started":"2023-09-01T09:49:59.007250Z","shell.execute_reply":"2023-09-01T09:49:59.010711Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# tokenizer = Tokenizer()\n# tokenizer.fit_on_texts(train_df['sentence'])\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:49:59.155953Z","iopub.execute_input":"2023-09-01T09:49:59.156331Z","iopub.status.idle":"2023-09-01T09:49:59.160471Z","shell.execute_reply.started":"2023-09-01T09:49:59.156303Z","shell.execute_reply":"2023-09-01T09:49:59.159474Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# audio_files = train_df['id'].apply(lambda x: f'{x}.mp3')\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:49:59.280920Z","iopub.execute_input":"2023-09-01T09:49:59.281635Z","iopub.status.idle":"2023-09-01T09:49:59.285800Z","shell.execute_reply.started":"2023-09-01T09:49:59.281600Z","shell.execute_reply":"2023-09-01T09:49:59.284793Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# audio_files","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:00.290788Z","iopub.execute_input":"2023-09-01T09:50:00.291512Z","iopub.status.idle":"2023-09-01T09:50:00.295944Z","shell.execute_reply.started":"2023-09-01T09:50:00.291475Z","shell.execute_reply":"2023-09-01T09:50:00.294921Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# X_features = []\n# Y_transcriptions = []\n# print(audio_files)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:00.398204Z","iopub.execute_input":"2023-09-01T09:50:00.398569Z","iopub.status.idle":"2023-09-01T09:50:00.403709Z","shell.execute_reply.started":"2023-09-01T09:50:00.398542Z","shell.execute_reply":"2023-09-01T09:50:00.402679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# x = 0\n\n# for audio_file in audio_files[:500]:\n#     x = x+1\n#     if x%100 == 0:\n#         print(x)\n#     audio_path = os.path.join(train_mp3s_folder, audio_file)\n#     y, sr = librosa.load(audio_path, sr=None)\n#     y_mono = librosa.to_mono(y)\n    \n#     # Extracting MFCC features\n#     mfccs = librosa.feature.mfcc(y=y_mono, sr=sr, n_mfcc=13)\n#     mfccs_processed = np.mean(mfccs.T, axis=0)\n#     X_features.append(mfccs_processed)\n    \n#     # Extracting transcription for the audio file --\n#     transcription = train_df[train_df['id'] == audio_file[:-4]]['sentence'].values[0]\n#     Y_transcriptions.append(transcription)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:00.535498Z","iopub.execute_input":"2023-09-01T09:50:00.536136Z","iopub.status.idle":"2023-09-01T09:50:00.542084Z","shell.execute_reply.started":"2023-09-01T09:50:00.536105Z","shell.execute_reply":"2023-09-01T09:50:00.539964Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# print( X_features[:1])","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:01.526316Z","iopub.execute_input":"2023-09-01T09:50:01.527135Z","iopub.status.idle":"2023-09-01T09:50:01.531780Z","shell.execute_reply.started":"2023-09-01T09:50:01.527093Z","shell.execute_reply":"2023-09-01T09:50:01.530632Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Y_sequences = tokenizer.texts_to_sequences(Y_transcriptions)\n# Y_padded = pad_sequences(Y_sequences, padding='post')\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:01.640877Z","iopub.execute_input":"2023-09-01T09:50:01.641662Z","iopub.status.idle":"2023-09-01T09:50:01.646450Z","shell.execute_reply.started":"2023-09-01T09:50:01.641608Z","shell.execute_reply":"2023-09-01T09:50:01.645487Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# # Converting padded sequences to categorical format\n# Y_categorical = [to_categorical(sequence, num_classes=len(tokenizer.word_index) + 1) for sequence in Y_padded]\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:01.760302Z","iopub.execute_input":"2023-09-01T09:50:01.761101Z","iopub.status.idle":"2023-09-01T09:50:01.765325Z","shell.execute_reply.started":"2023-09-01T09:50:01.761069Z","shell.execute_reply":"2023-09-01T09:50:01.764297Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# X_features = np.array(X_features)\n# Y_categorical = np.array(Y_categorical)\n\n# # Print the shapes of processed data\n# print(\"Features shape:\", X_features.shape)\n# print(\"Transcriptions shape:\", Y_categorical.shape)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:02.730865Z","iopub.execute_input":"2023-09-01T09:50:02.731236Z","iopub.status.idle":"2023-09-01T09:50:02.737397Z","shell.execute_reply.started":"2023-09-01T09:50:02.731208Z","shell.execute_reply":"2023-09-01T09:50:02.735071Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# X_train, X_val, Y_train, Y_val = train_test_split(X_features, Y_categorical, test_size=0.2, random_state=42)\n# X_train.shape\n\n# print(X_val.shape)\n# print(Y_val.shape)\n\n# X_train_reshaped = X_train.reshape(X_train.shape[0], X_train.shape[1], 1)\n# num_output_units = len(tokenizer.word_index) + 1\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:02.855775Z","iopub.execute_input":"2023-09-01T09:50:02.856127Z","iopub.status.idle":"2023-09-01T09:50:02.860824Z","shell.execute_reply.started":"2023-09-01T09:50:02.856100Z","shell.execute_reply":"2023-09-01T09:50:02.859587Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Padding or truncating target sequences to match the model's output length\n# Y_train_padded = pad_sequences(Y_train, maxlen=X_train_reshaped.shape[1], padding='post', truncating='post')\n# Y_val_padded = pad_sequences(Y_val, maxlen=X_train_reshaped.shape[1], padding='post', truncating='post')","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:03.010431Z","iopub.execute_input":"2023-09-01T09:50:03.010772Z","iopub.status.idle":"2023-09-01T09:50:03.015051Z","shell.execute_reply.started":"2023-09-01T09:50:03.010745Z","shell.execute_reply":"2023-09-01T09:50:03.014019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# print(X_train_reshaped.shape)\n# print(Y_train_padded.shape)\n# print(X_val.shape)\n# print(Y_val_padded.shape)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:03.170621Z","iopub.execute_input":"2023-09-01T09:50:03.170955Z","iopub.status.idle":"2023-09-01T09:50:03.175067Z","shell.execute_reply.started":"2023-09-01T09:50:03.170930Z","shell.execute_reply":"2023-09-01T09:50:03.174076Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# # Define early stopping callback\n# early_stopping = EarlyStopping(monitor='val_loss', patience=3)\n\n# # Define the speech recognition model\n# model = Sequential()\n# model.add(LSTM(units=128, input_shape=(X_train_reshaped.shape[1], X_train_reshaped.shape[2]), return_sequences=True))\n# model.add(Dropout(0.2))\n# model.add(LSTM(units=128, return_sequences=True))\n# model.add(TimeDistributed(Dense(num_output_units, activation='softmax')))\n\n# # Compile the model\n# model.compile(loss='categorical_crossentropy', optimizer=Adam(learning_rate=0.001), metrics=['accuracy'])\n\n# # Train the model\n# history = model.fit(X_train_reshaped, Y_train_padded, validation_data=(X_val, Y_val_padded), batch_size=32, epochs=100, callbacks=[early_stopping])\n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:04.085816Z","iopub.execute_input":"2023-09-01T09:50:04.086217Z","iopub.status.idle":"2023-09-01T09:50:04.091652Z","shell.execute_reply.started":"2023-09-01T09:50:04.086184Z","shell.execute_reply":"2023-09-01T09:50:04.090714Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# # Evaluating  our model on validation data\n# val_loss, val_accuracy = model.evaluate(X_val, Y_val_padded)\n\n# print(\"Validation Loss:\", val_loss)\n# print(\"Validation Accuracy:\", val_accuracy)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:04.375920Z","iopub.execute_input":"2023-09-01T09:50:04.376309Z","iopub.status.idle":"2023-09-01T09:50:04.380734Z","shell.execute_reply.started":"2023-09-01T09:50:04.376281Z","shell.execute_reply":"2023-09-01T09:50:04.379765Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n\n# # Process test audio files\n# test_audio_files = os.listdir(test_mp3s_folder)\n# X_test_features = []\n\n# for audio_file in test_audio_files:\n#     audio_path = os.path.join(test_mp3s_folder, audio_file)\n#     y_test, sr_test = librosa.load(audio_path, sr=None)\n#     y_test_mono = librosa.to_mono(y_test)\n    \n#     mfccs_test = librosa.feature.mfcc(y=y_test_mono, sr=sr_test, n_mfcc=13)\n#     mfccs_test_processed = np.mean(mfccs_test.T, axis=0)\n#     X_test_features.append(mfccs_test_processed)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:04.540807Z","iopub.execute_input":"2023-09-01T09:50:04.541193Z","iopub.status.idle":"2023-09-01T09:50:04.546094Z","shell.execute_reply.started":"2023-09-01T09:50:04.541162Z","shell.execute_reply":"2023-09-01T09:50:04.545078Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# # Convert test features to a NumPy array\n# X_test_features = np.array(X_test_features)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:04.650670Z","iopub.execute_input":"2023-09-01T09:50:04.651046Z","iopub.status.idle":"2023-09-01T09:50:04.655444Z","shell.execute_reply.started":"2023-09-01T09:50:04.650993Z","shell.execute_reply":"2023-09-01T09:50:04.654402Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# # Perform inference on test data\n# predictions_test = model.predict(X_test_features)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:04.930954Z","iopub.execute_input":"2023-09-01T09:50:04.932300Z","iopub.status.idle":"2023-09-01T09:50:04.936962Z","shell.execute_reply.started":"2023-09-01T09:50:04.932256Z","shell.execute_reply":"2023-09-01T09:50:04.935842Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"\n# # Convert predictions to transcriptions using the tokenizer\n# predicted_sequences_test = np.argmax(predictions_test, axis=2)\n# predicted_transcriptions_test = tokenizer.sequences_to_texts(predicted_sequences_test)\n\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:05.071803Z","iopub.execute_input":"2023-09-01T09:50:05.072186Z","iopub.status.idle":"2023-09-01T09:50:05.076518Z","shell.execute_reply.started":"2023-09-01T09:50:05.072158Z","shell.execute_reply":"2023-09-01T09:50:05.075505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# # Assuming you have a DataFrame named submission_df\n# # Filter out rows with NaN predictions\n# submission_df = pd.DataFrame({'id': test_audio_files, 'sentence': predicted_transcriptions_test})\n\n\n# valid_predictions_df = submission_df.dropna(subset=['sentence'])\n\n# # Remove the \".mp3\" extension from the 'id' column\n# valid_predictions_df['id'] = valid_predictions_df['id'].str.replace('.mp3', '')\n\n# # Save the filtered DataFrame to a CSV file\n# valid_predictions_df.to_csv('submission.csv', index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:05.235424Z","iopub.execute_input":"2023-09-01T09:50:05.235980Z","iopub.status.idle":"2023-09-01T09:50:05.240894Z","shell.execute_reply.started":"2023-09-01T09:50:05.235946Z","shell.execute_reply":"2023-09-01T09:50:05.239898Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# import pandas as pd","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:05.385830Z","iopub.execute_input":"2023-09-01T09:50:05.386246Z","iopub.status.idle":"2023-09-01T09:50:05.390767Z","shell.execute_reply.started":"2023-09-01T09:50:05.386214Z","shell.execute_reply":"2023-09-01T09:50:05.389677Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# submission  = pd.read_csv('/kaggle/input/bengaliai-speech/sample_submission.csv')\n# submission.to_csv('submission.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:05.570726Z","iopub.execute_input":"2023-09-01T09:50:05.571109Z","iopub.status.idle":"2023-09-01T09:50:05.575612Z","shell.execute_reply.started":"2023-09-01T09:50:05.571079Z","shell.execute_reply":"2023-09-01T09:50:05.574481Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# final_1 = pd.read_csv(\"/kaggle/working/submission.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:05.745460Z","iopub.execute_input":"2023-09-01T09:50:05.745762Z","iopub.status.idle":"2023-09-01T09:50:05.750256Z","shell.execute_reply.started":"2023-09-01T09:50:05.745737Z","shell.execute_reply":"2023-09-01T09:50:05.749265Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# final_1","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:06.110890Z","iopub.execute_input":"2023-09-01T09:50:06.111270Z","iopub.status.idle":"2023-09-01T09:50:06.118235Z","shell.execute_reply.started":"2023-09-01T09:50:06.111241Z","shell.execute_reply":"2023-09-01T09:50:06.117333Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport librosa\nimport torch\nimport torchaudio\nimport numpy as np\n\nfrom transformers import WhisperTokenizer\nfrom transformers import WhisperProcessor\nfrom transformers import WhisperFeatureExtractor\nfrom transformers import WhisperForConditionalGeneration\nfrom tqdm import tqdm\nimport pandas as pd\nimport soundfile as sf\nfrom pydub import AudioSegment","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:06.796189Z","iopub.execute_input":"2023-09-01T09:50:06.796592Z","iopub.status.idle":"2023-09-01T09:50:06.803360Z","shell.execute_reply.started":"2023-09-01T09:50:06.796562Z","shell.execute_reply":"2023-09-01T09:50:06.802245Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"device = torch.device('cuda' if torch.cuda.is_available() else 'cpu')\n\nmodel_path = \"bangla-speech-processing/BanglaASR\"\nfeature_extractor = WhisperFeatureExtractor.from_pretrained(model_path)\ntokenizer = WhisperTokenizer.from_pretrained(model_path)\nprocessor = WhisperProcessor.from_pretrained(model_path)\nmodel = WhisperForConditionalGeneration.from_pretrained(model_path).to(device)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:07.056020Z","iopub.execute_input":"2023-09-01T09:50:07.056402Z","iopub.status.idle":"2023-09-01T09:50:36.913155Z","shell.execute_reply.started":"2023-09-01T09:50:07.056372Z","shell.execute_reply":"2023-09-01T09:50:36.911595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def inference_fn(path):\n    speech_array, sampling_rate = sf.read(mp3_path)\n    speech_array = librosa.resample(np.asarray(speech_array), orig_sr=sampling_rate, target_sr=16000)\n    input_features = feature_extractor(speech_array, sampling_rate=16000, return_tensors=\"pt\").input_features\n    predicted_ids = model.generate(inputs=input_features.to(device))[0]\n    transcription = processor.decode(predicted_ids, skip_special_tokens=True)\n    return transcription","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:43.911158Z","iopub.execute_input":"2023-09-01T09:50:43.911523Z","iopub.status.idle":"2023-09-01T09:50:43.919039Z","shell.execute_reply.started":"2023-09-01T09:50:43.911494Z","shell.execute_reply":"2023-09-01T09:50:43.918046Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/bengaliai-speech/train.csv\")\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:45.156252Z","iopub.execute_input":"2023-09-01T09:50:45.159146Z","iopub.status.idle":"2023-09-01T09:50:48.527661Z","shell.execute_reply.started":"2023-09-01T09:50:45.159107Z","shell.execute_reply":"2023-09-01T09:50:48.526732Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"root_path = \"/kaggle/input/bengaliai-speech/train_mp3s/\"\nidx = 1\nmp3_path = root_path + df['id'].iloc[idx]+\".mp3\"\n\nprint(f\"File name :\",mp3_path)\nAudioSegment.from_file(mp3_path)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:50.361055Z","iopub.execute_input":"2023-09-01T09:50:50.361427Z","iopub.status.idle":"2023-09-01T09:50:50.694211Z","shell.execute_reply.started":"2023-09-01T09:50:50.361398Z","shell.execute_reply":"2023-09-01T09:50:50.693270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Original Transcription : {df['sentence'].iloc[idx]}\")\nprint(f\"Predicted Text : {inference_fn(mp3_path)}\")","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:50.696139Z","iopub.execute_input":"2023-09-01T09:50:50.696509Z","iopub.status.idle":"2023-09-01T09:50:51.516583Z","shell.execute_reply.started":"2023-09-01T09:50:50.696473Z","shell.execute_reply":"2023-09-01T09:50:51.515630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import time\ntotal = 0\nfor idx in range(5,10):\n    mp3_path = root_path + df['id'].iloc[idx]+\".mp3\"\n    print(f\"File name :\",mp3_path)\n    print(f\"Original Transcription : {df['sentence'].iloc[idx]}\")\n    start = time.time()\n    print(f\"Predicted Text : {inference_fn(mp3_path)}\")\n    end = time.time()\n    total+=end-start\n    print(\"\\n\")\nprint(f\"Total Inference time : {total} seconds\")","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:50:51.518802Z","iopub.execute_input":"2023-09-01T09:50:51.519598Z","iopub.status.idle":"2023-09-01T09:51:01.057563Z","shell.execute_reply.started":"2023-09-01T09:50:51.519564Z","shell.execute_reply":"2023-09-01T09:51:01.055739Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"mp3_path = \"/kaggle/input/bengaliai-speech/test_mp3s/0f3dac00655e.mp3\"\nprint(f\"File name :\",mp3_path)\nAudioSegment.from_file(mp3_path)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:51:01.060023Z","iopub.execute_input":"2023-09-01T09:51:01.060372Z","iopub.status.idle":"2023-09-01T09:51:01.409224Z","shell.execute_reply.started":"2023-09-01T09:51:01.060339Z","shell.execute_reply":"2023-09-01T09:51:01.408032Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(f\"Prediction : {inference_fn(mp3_path)}\")\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:51:01.410749Z","iopub.execute_input":"2023-09-01T09:51:01.411138Z","iopub.status.idle":"2023-09-01T09:51:02.220329Z","shell.execute_reply.started":"2023-09-01T09:51:01.411103Z","shell.execute_reply":"2023-09-01T09:51:02.218453Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sub = pd.read_csv(\"/kaggle/input/bengaliai-speech/sample_submission.csv\")\nsub.head()","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:51:02.221852Z","iopub.execute_input":"2023-09-01T09:51:02.222231Z","iopub.status.idle":"2023-09-01T09:51:02.236231Z","shell.execute_reply.started":"2023-09-01T09:51:02.222198Z","shell.execute_reply":"2023-09-01T09:51:02.235062Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\ntest_path = \"/kaggle/input/bengaliai-speech/test_mp3s/\"\nfiles = os.listdir(test_path)\nids = []\nsentences = []\nfor file in tqdm(files):\n    ids.append(file.split(\".\")[0])\n    mp3_path = os.path.join(test_path,file)\n    prediction = inference_fn(mp3_path)\n    \n    #sanity check\n    if len(prediction)==0:\n        prediction = \"\\n\"\n    \n    sentences.append(prediction)","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:51:02.238056Z","iopub.execute_input":"2023-09-01T09:51:02.238445Z","iopub.status.idle":"2023-09-01T09:51:06.284294Z","shell.execute_reply.started":"2023-09-01T09:51:02.238410Z","shell.execute_reply":"2023-09-01T09:51:06.283323Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.DataFrame({\"id\":ids,\"sentence\":sentences})\ndf.head()\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:51:06.287463Z","iopub.execute_input":"2023-09-01T09:51:06.287809Z","iopub.status.idle":"2023-09-01T09:51:06.297642Z","shell.execute_reply.started":"2023-09-01T09:51:06.287782Z","shell.execute_reply":"2023-09-01T09:51:06.296518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv(\"submission.csv\",index=False)\n","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:51:06.299232Z","iopub.execute_input":"2023-09-01T09:51:06.299899Z","iopub.status.idle":"2023-09-01T09:51:06.308937Z","shell.execute_reply.started":"2023-09-01T09:51:06.299866Z","shell.execute_reply":"2023-09-01T09:51:06.307977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dx = pd.read_csv('/kaggle/working/submission.csv')","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:54:57.960634Z","iopub.execute_input":"2023-09-01T09:54:57.961016Z","iopub.status.idle":"2023-09-01T09:54:57.969372Z","shell.execute_reply.started":"2023-09-01T09:54:57.960971Z","shell.execute_reply":"2023-09-01T09:54:57.968287Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dx","metadata":{"execution":{"iopub.status.busy":"2023-09-01T09:55:06.052187Z","iopub.execute_input":"2023-09-01T09:55:06.052743Z","iopub.status.idle":"2023-09-01T09:55:06.063353Z","shell.execute_reply.started":"2023-09-01T09:55:06.052702Z","shell.execute_reply":"2023-09-01T09:55:06.062435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}],"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}}