{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"name":"python","version":"3.10.13","mimetype":"text/x-python","codemirror_mode":{"name":"ipython","version":3},"pygments_lexer":"ipython3","nbconvert_exporter":"python","file_extension":".py"},"kaggle":{"accelerator":"none","dataSources":[{"sourceId":70203,"databundleVersionId":8068726,"sourceType":"competition"},{"sourceId":172627469,"sourceType":"kernelVersion"},{"sourceId":181198074,"sourceType":"kernelVersion"}],"dockerImageVersionId":30698,"isInternetEnabled":true,"language":"python","sourceType":"notebook","isGpuEnabled":false}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\n\nfrom sklearn.model_selection import StratifiedKFold","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2024-05-11T13:52:44.093078Z","iopub.execute_input":"2024-05-11T13:52:44.093639Z","iopub.status.idle":"2024-05-11T13:52:44.101226Z","shell.execute_reply.started":"2024-05-11T13:52:44.093602Z","shell.execute_reply":"2024-05-11T13:52:44.099413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dup_df = pd.read_csv(\"/kaggle/input/bc24-duplicate-audio-files/dupes.csv\")\ndup_df.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:52:44.105235Z","iopub.execute_input":"2024-05-11T13:52:44.106806Z","iopub.status.idle":"2024-05-11T13:52:44.125283Z","shell.execute_reply.started":"2024-05-11T13:52:44.106743Z","shell.execute_reply":"2024-05-11T13:52:44.123906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df = pd.read_csv(\"/kaggle/input/pretrained-data/train.csv\")\n\nRANDOM_SEED = 42\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:52:44.126990Z","iopub.execute_input":"2024-05-11T13:52:44.127375Z","iopub.status.idle":"2024-05-11T13:52:44.328927Z","shell.execute_reply.started":"2024-05-11T13:52:44.127342Z","shell.execute_reply":"2024-05-11T13:52:44.327611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:52:44.330897Z","iopub.execute_input":"2024-05-11T13:52:44.331897Z","iopub.status.idle":"2024-05-11T13:52:44.340475Z","shell.execute_reply.started":"2024-05-11T13:52:44.331851Z","shell.execute_reply":"2024-05-11T13:52:44.339357Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# df = df[df.filename != dup_df['filename_b']]\ndf.drop(df[df['filename'].isin(dup_df['filename_b'])].index, inplace=True)\ndf = df.reset_index(drop=True)\ndf.shape","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:52:44.343371Z","iopub.execute_input":"2024-05-11T13:52:44.344860Z","iopub.status.idle":"2024-05-11T13:52:44.375037Z","shell.execute_reply.started":"2024-05-11T13:52:44.344768Z","shell.execute_reply":"2024-05-11T13:52:44.373098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.groupby(['primary_label']).size().sort_values(ascending=False) ","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:52:44.376983Z","iopub.execute_input":"2024-05-11T13:52:44.378246Z","iopub.status.idle":"2024-05-11T13:52:44.396953Z","shell.execute_reply.started":"2024-05-11T13:52:44.378193Z","shell.execute_reply":"2024-05-11T13:52:44.395611Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":" df['fold'] = -1","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:52:44.398718Z","iopub.execute_input":"2024-05-11T13:52:44.400014Z","iopub.status.idle":"2024-05-11T13:52:44.407470Z","shell.execute_reply.started":"2024-05-11T13:52:44.399969Z","shell.execute_reply":"2024-05-11T13:52:44.405789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"skf = StratifiedKFold(n_splits=5, shuffle=True, random_state=RANDOM_SEED)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:52:44.409577Z","iopub.execute_input":"2024-05-11T13:52:44.410713Z","iopub.status.idle":"2024-05-11T13:52:44.420846Z","shell.execute_reply.started":"2024-05-11T13:52:44.410642Z","shell.execute_reply":"2024-05-11T13:52:44.419263Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for fold, (train_index, test_index) in enumerate(skf.split(df, df[\"primary_label\"])):\n    df.loc[test_index, 'fold'] = fold\nprint(df.shape)\nerror_paths = [\n\"aspfly1/XC775312.mp3\",\n\"comior1/XC881009.mp3\",\n\"hoopoe/XC891005.mp3\",\n\"hoopoe/XC891004.mp3\",\n\"hoopoe/XC798809.mp3\",\n\"hoopoe/XC798808.mp3\",\n\"hoopoe/XC798807.mp3\",\n\"hoopoe/XC798806.mp3\",\n\"hoopoe/XC798805.mp3\",\n\"eaywag1/XC835367.mp3\",\n\"orihob2/XC762524.mp3\"]\ndf = df[~df['filename'].isin(error_paths)]\nprint(df.shape)\ndf.to_csv('add_train.csv', index=False)","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:52:44.423272Z","iopub.execute_input":"2024-05-11T13:52:44.423850Z","iopub.status.idle":"2024-05-11T13:52:44.924050Z","shell.execute_reply.started":"2024-05-11T13:52:44.423800Z","shell.execute_reply":"2024-05-11T13:52:44.922523Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2024-05-11T13:52:44.925621Z","iopub.execute_input":"2024-05-11T13:52:44.926064Z","iopub.status.idle":"2024-05-11T13:52:44.943976Z","shell.execute_reply.started":"2024-05-11T13:52:44.926027Z","shell.execute_reply":"2024-05-11T13:52:44.942255Z"},"trusted":true},"execution_count":null,"outputs":[]}]}