{"cells":[{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"df= pd.read_csv(\"../input/rfcx-species-audio-detection/train_tp.csv\")\ndf.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"print(df.species_id.unique())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":" len(df) - len(df.recording_id.unique()) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fp = pd.read_csv(\"../input/rfcx-species-audio-detection/train_fp.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":" len(fp) - len(fp.recording_id.unique()) ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"duplicate = [i  for i in fp.recording_id.unique() if i in df.recording_id.unique()]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df[df['recording_id']=='015113cad']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fp[fp['recording_id']=='015113cad']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.describe()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.t_min.hist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.t_max.hist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df['duration'] = df['t_max'] - df['t_min']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df.duration.hist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df[df['recording_id']=='c12e0a62b']","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## Remove duplicate\n"},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = df.copy()\ndf_train.drop_duplicates(subset=['recording_id'], keep=False, inplace=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train['type'] = df_train['species_id']\ndf_train['species_id'] = df_train['species_id'].apply(lambda x: [x])\ndf_train['t_min'] = df_train['t_min'].apply(lambda x: [x])\ndf_train['t_max'] = df_train['t_max'].apply(lambda x: [x])\n","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dt = df.recording_id.value_counts()\nindex = dt[dt>1].index","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"recording_id_all = []\nspecies_id_all = []\nt_min_all = []\nt_max_all = []\nfor idx in index:\n    tmp = df[df['recording_id']==idx].values\n    species_id_tmp = []\n    t_min_tmp = []\n    t_max_tmp = []\n    for item in tmp:\n        species_id_tmp.append(item[1])\n        t_min_tmp.append(item[3])\n        t_max_tmp.append(item[5])\n    recording_id_all.append(idx)\n    species_id_all.append(species_id_tmp)\n    t_min_all.append(t_min_tmp)\n    t_max_all.append(t_max_tmp)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_dup = pd.DataFrame({\"recording_id\": recording_id_all,\n                     \"species_id\":species_id_all,\n                      \"t_min\":t_min_all,\n                      \"t_max\":t_max_all,\n                      \"type\":25\n                      })","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_all = pd.concat((df_train, df_dup))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_all.sample(20)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"from sklearn.model_selection import StratifiedKFold\n\ndef sampling_k_elements(group, k=50):\n    if len(group) < k:\n        return group\n    return group.sample(k)\n    \ntrain_full_df = (\n            df_all.groupby(\"type\")\n            .apply(sampling_k_elements)\n            .reset_index(drop=True)\n        )\n\nskf = StratifiedKFold(n_splits=5, shuffle=True, random_state=2020)\n\ntrain_full_df[\"fold\"] = -1\nfor fold_id, (train_index, val_index) in enumerate(skf.split(train_full_df, train_full_df[\"type\"])):\n    train_full_df.iloc[val_index, -1] = fold_id\ntrain_full_df.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for use_fold in range(5):\n#     train_full_df = pd.read_csv(\"/content/drive/Shareddrives/Working/Kaggle/RainForest/train_tp.csv\")\n    train_file_list = train_full_df.query(\"fold != @use_fold\")\n    val_file_list = train_full_df.query(\"fold == @use_fold\")\n    print(\"[fold {}] train: {}, val: {}\".format(use_fold, len(train_file_list), len(val_file_list)))\n    print(train_file_list.type.value_counts())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_full_df.duration.hist()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_full_df.to_csv(\"train_preprocessing.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"len(train_full_df)- len(train_full_df.recording_id.value_counts())","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in train_full_df.species_id.values:\n    for j in i:\n        pass","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dfx= pd.read_csv(\"train_preprocessing.csv\")","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in dfx.species_id.values:\n    for j in i:\n        pass","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_full_df.head()","execution_count":null,"outputs":[]},{"metadata":{},"cell_type":"markdown","source":"## False positive"},{"metadata":{"trusted":true},"cell_type":"code","source":"fp = pd.read_csv(\"../input/rfcx-species-audio-detection/train_fp.csv\")\nfp.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"fp = fp[fp['recording_id'].isin(train_full_df.recording_id.tolist())]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_train = fp.copy()\ndf_train.drop_duplicates(subset=['recording_id'], keep=False, inplace=True)\ndf_train['type'] = df_train['species_id']\ndf_train['species_id'] = df_train['species_id'].apply(lambda x: [x])\ndf_train['t_min'] = df_train['t_min'].apply(lambda x: [x])\ndf_train['t_max'] = df_train['t_max'].apply(lambda x: [x])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"dt = fp.recording_id.value_counts()\nindex = dt[dt>1].index\nrecording_id_all = []\nspecies_id_all = []\nt_min_all = []\nt_max_all = []\nfor idx in index:\n    tmp = fp[fp['recording_id']==idx].values\n    species_id_tmp = []\n    t_min_tmp = []\n    t_max_tmp = []\n    for item in tmp:\n        species_id_tmp.append(item[1])\n        t_min_tmp.append(item[3])\n        t_max_tmp.append(item[5])\n    recording_id_all.append(idx)\n    species_id_all.append(species_id_tmp)\n    t_min_all.append(t_min_tmp)\n    t_max_all.append(t_max_tmp)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_dup = pd.DataFrame({\"recording_id\": recording_id_all,\n                     \"species_id\":species_id_all,\n                      \"t_min\":t_min_all,\n                      \"t_max\":t_max_all,\n                      \"type\":25\n                      })\n\ndf_all = pd.concat((df_train, df_dup))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_all","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_all = df_all[['recording_id', 'species_id', 't_min', 't_max']]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_all.columns = ['recording_id', 'FPspecies_id', 'FPt_min', 'FPt_max']","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_all.head()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_x = train_full_df.join(df_all.set_index('recording_id'), on='recording_id') ","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_x","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"df_x.to_csv(\"train_tp_fp.csv\",index=False)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}