{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":73047,"databundleVersionId":8149390,"sourceType":"competition"},{"sourceId":8138802,"sourceType":"datasetVersion","datasetId":4811651},{"sourceId":8138812,"sourceType":"datasetVersion","datasetId":4811660}],"dockerImageVersionId":30684,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"!pip install jiwer","metadata":{"execution":{"iopub.status.busy":"2024-04-16T16:06:45.223674Z","iopub.execute_input":"2024-04-16T16:06:45.224879Z","iopub.status.idle":"2024-04-16T16:07:04.770677Z","shell.execute_reply.started":"2024-04-16T16:06:45.224831Z","shell.execute_reply":"2024-04-16T16:07:04.769102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from jiwer import wer\nimport pandas as pd\nimport os","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:24:53.325983Z","iopub.execute_input":"2024-04-16T18:24:53.326511Z","iopub.status.idle":"2024-04-16T18:24:53.331844Z","shell.execute_reply.started":"2024-04-16T18:24:53.326474Z","shell.execute_reply":"2024-04-16T18:24:53.330968Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dic = {}","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:24:55.805189Z","iopub.execute_input":"2024-04-16T18:24:55.805695Z","iopub.status.idle":"2024-04-16T18:24:55.810851Z","shell.execute_reply.started":"2024-04-16T18:24:55.805658Z","shell.execute_reply":"2024-04-16T18:24:55.809630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folder = \"/kaggle/input/location-wise-sep/locationWiseSeparationOfTrainData\"\nfor filename in os.listdir(folder):\n    path = folder + \"/\" + filename\n    df = pd.read_csv(path)\n    for index, row in df.iterrows():\n        file_name = row['file_name']\n        transcripts = row['transcripts']\n        if file_name not in dic.keys():\n            dic[file_name] = [transcripts]\n        else:\n            dic[file_name].append(transcripts)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:24:58.872944Z","iopub.execute_input":"2024-04-16T18:24:58.873685Z","iopub.status.idle":"2024-04-16T18:25:00.028068Z","shell.execute_reply.started":"2024-04-16T18:24:58.873650Z","shell.execute_reply":"2024-04-16T18:25:00.026756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"folder = \"/kaggle/input/all-pred-whisper-1st/all_predictions_whisper_1st\"\nfor filename in os.listdir(folder):\n    path = folder + \"/\" + filename\n    df = pd.read_csv(path)\n    for index, row in df.iterrows():\n        file_name = row['id'].split('/')[-1]\n        transcripts = row['sentence']\n        if file_name not in dic.keys():\n            dic[file_name] = [transcripts]\n        else:\n            dic[file_name].append(transcripts)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:25:03.022793Z","iopub.execute_input":"2024-04-16T18:25:03.023224Z","iopub.status.idle":"2024-04-16T18:25:04.179614Z","shell.execute_reply.started":"2024-04-16T18:25:03.023194Z","shell.execute_reply":"2024-04-16T18:25:04.177847Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = \"https://raw.githubusercontent.com/Jak57/datasets/main/datathon_asr_2024/data_with_duration_asr/train_with_duration_13481_all.csv\"\ndf = pd.read_csv(path)\npath_dic = {}\nfor idx, row in df.iterrows():\n    file_name = row['file_name']\n    audio_length = row['audio_length']\n    path_dic[file_name] = audio_length\nprint(len(path_dic))","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:25:08.524677Z","iopub.execute_input":"2024-04-16T18:25:08.525068Z","iopub.status.idle":"2024-04-16T18:25:09.861747Z","shell.execute_reply.started":"2024-04-16T18:25:08.525039Z","shell.execute_reply":"2024-04-16T18:25:09.860321Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bucket = []","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:25:13.716166Z","iopub.execute_input":"2024-04-16T18:25:13.716601Z","iopub.status.idle":"2024-04-16T18:25:13.722988Z","shell.execute_reply.started":"2024-04-16T18:25:13.716572Z","shell.execute_reply":"2024-04-16T18:25:13.721604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for k in dic:\n    label = dic[k][0]\n    pred = dic[k][1]\n    error = wer(label, pred)\n    bucket.append((error, path_dic[k], k, label, pred))\nprint(len(bucket))","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:25:16.871955Z","iopub.execute_input":"2024-04-16T18:25:16.872377Z","iopub.status.idle":"2024-04-16T18:25:19.415009Z","shell.execute_reply.started":"2024-04-16T18:25:16.872348Z","shell.execute_reply":"2024-04-16T18:25:19.413440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bucket[:2]","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:25:22.349535Z","iopub.execute_input":"2024-04-16T18:25:22.349961Z","iopub.status.idle":"2024-04-16T18:25:22.359200Z","shell.execute_reply.started":"2024-04-16T18:25:22.349929Z","shell.execute_reply":"2024-04-16T18:25:22.358074Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bucket = sorted(bucket)\ngood = []\nbad = []\nfor t in bucket:\n    if t[1] >= 15.0 and t[0] <= 0.75:\n        good.append(t)\n    else:\n        bad.append(t)\nprint(len(good), len(bad), len(good)/len(bucket), len(bad)/len(bucket))","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:27:03.413499Z","iopub.execute_input":"2024-04-16T18:27:03.413949Z","iopub.status.idle":"2024-04-16T18:27:03.434307Z","shell.execute_reply.started":"2024-04-16T18:27:03.413918Z","shell.execute_reply":"2024-04-16T18:27:03.432899Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"good = sorted(good)\ngood[:5]","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:27:38.850001Z","iopub.execute_input":"2024-04-16T18:27:38.850451Z","iopub.status.idle":"2024-04-16T18:27:38.859870Z","shell.execute_reply.started":"2024-04-16T18:27:38.850397Z","shell.execute_reply":"2024-04-16T18:27:38.858585Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"bad = sorted(bad)\nbad[:5]","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:27:51.262514Z","iopub.execute_input":"2024-04-16T18:27:51.262954Z","iopub.status.idle":"2024-04-16T18:27:51.274612Z","shell.execute_reply.started":"2024-04-16T18:27:51.262923Z","shell.execute_reply":"2024-04-16T18:27:51.273056Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"path = []\nsentence = []\naudio_length = []\nprediction_whisper = []\nwer_whisper = []\nfor t in bucket:\n    print(t)\n    break","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:27:04.749851Z","iopub.execute_input":"2024-04-16T18:27:04.750384Z","iopub.status.idle":"2024-04-16T18:27:04.758309Z","shell.execute_reply.started":"2024-04-16T18:27:04.750347Z","shell.execute_reply":"2024-04-16T18:27:04.756889Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for t in bucket:\n    path.append(t[2])\n    sentence.append(t[3])\n    audio_length.append(t[1])\n    prediction_whisper.append(t[4])\n    wer_whisper.append(t[0])\ndata = {\n    'path': path,\n    'sentence': sentence,\n    'audio_length': audio_length,\n    'prediction_whisper': prediction_whisper,\n    'wer_whisper': wer_whisper\n}\ndf = pd.DataFrame(data, columns=data.keys())\nprint(len(df))\ndf.head(5)","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:27:06.911362Z","iopub.execute_input":"2024-04-16T18:27:06.911776Z","iopub.status.idle":"2024-04-16T18:27:06.974300Z","shell.execute_reply.started":"2024-04-16T18:27:06.911748Z","shell.execute_reply":"2024-04-16T18:27:06.973081Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.to_csv('all_information_asr_dataset_train_13481_v1.csv')","metadata":{"execution":{"iopub.status.busy":"2024-04-16T18:20:59.807574Z","iopub.execute_input":"2024-04-16T18:20:59.808857Z","iopub.status.idle":"2024-04-16T18:21:00.199620Z","shell.execute_reply.started":"2024-04-16T18:20:59.808803Z","shell.execute_reply":"2024-04-16T18:21:00.198376Z"},"trusted":true},"execution_count":null,"outputs":[]}]}