{"cells":[{"metadata":{},"cell_type":"markdown","source":"This is a modified version of Bojan's Adverserial Rainforest notebook. https://www.kaggle.com/tunguz/adversarial-rainforest\n- fft features are extracted from both train and test at the file level instead of using tp and fp labels for train.\n- The classier trained on this data scores **0.6372492726994334** roc_auc vs **0.8668358325958252** when using tp and fp slices to extract fft features for train.\n- The lower score of ~0.6 suggests the train and test distributions do not differ greatly which seems to be more in line with CV vs LB scores. "},{"metadata":{"trusted":true},"cell_type":"code","source":"!pip install --use-feature=2020-resolver https://s3-us-west-2.amazonaws.com/xgboost-nightly-builds/xgboost-1.3.0_SNAPSHOT%2Bdda9e1e4879118738d9f9d5094246692c0f6123c-py3-none-manylinux2010_x86_64.whl","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport random\nimport glob\nimport os\nfrom scipy.interpolate import interp1d\nfrom scipy import signal\nimport gc\nfrom sklearn.linear_model import LogisticRegression\nfrom sklearn.model_selection import GroupKFold\nimport lightgbm as lgb\nfrom joblib import Parallel, delayed\nfrom tqdm.notebook import tqdm\nfrom sklearn.model_selection import train_test_split\nimport shap\n\nimport soundfile as sf\n# Librosa Libraries\nimport librosa\nimport librosa.display\nimport IPython.display as ipd\nimport matplotlib.pyplot as plt\n\nfrom sklearn.metrics import roc_auc_score, accuracy_score\n\nimport xgboost","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"train_files = glob.glob( '../input/rfcx-species-audio-detection/train/*.flac' )\ntest_files = glob.glob( '../input/rfcx-species-audio-detection/test/*.flac' )\nlen(train_files), len(test_files), train_files[0]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"def extract_features(fn):\n    data, samplerate = sf.read(fn)\n    data_fft = np.fft.fft(data)\n    data_fft = data_fft[:(len(data)//2)]\n    varfft = np.abs(data_fft)\n    x = np.linspace(0, len(varfft), num=len(varfft), endpoint=True)\n    f1 = interp1d(x, varfft, kind='cubic')\n    x = np.linspace(0, len(varfft), num=1000, endpoint=True)\n    varfft = f1(x)\n    \n    return varfft","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_fft_features = Parallel(n_jobs=4)(delayed(extract_features)(fn) for fn in tqdm(train_files))\ntrain_fft_features = np.stack(train_fft_features)\ngc.collect()\n\ntrain_fft_features.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"test_fft_features = Parallel(n_jobs=4)(delayed(extract_features)(fn) for fn in tqdm(test_files))\ntest_fft_features = np.stack(test_fft_features)\ngc.collect()\n\ntest_fft_features.shape","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"target = np.hstack([np.ones(train_fft_features.shape[0]), np.zeros(train_fft_features.shape[0])])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_test = np.vstack([train_fft_features, test_fft_features])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"index = list(range(train_test.shape[0]))\nrandom.shuffle(index)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train_test = train_test[index, :]\ntarget = target[index]","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train, test, y_train, y_test = train_test_split(train_test, target, test_size=0.33, random_state=42)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"train = xgboost.DMatrix(train, label=y_train)\ntest = xgboost.DMatrix(test, label=y_test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\nparam = {\n    'eta': 0.05,\n    'max_depth': 10,\n    'subsample': 0.8,\n    'colsample_bytree': 0.7,\n    'objective': 'reg:logistic',\n    'eval_metric': 'auc',\n    'tree_method': 'gpu_hist', \n    'predictor': 'gpu_predictor'\n}\nclf = xgboost.train(param, train, 600)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"preds = clf.predict(test)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"roc_auc_score(y_test, preds)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"%%time\nshap_preds = clf.predict(test, pred_contribs=True)","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"shap.initjs()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"shap.summary_plot(shap_preds[:,:1000])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"shap.summary_plot(shap_preds[:,:1000], plot_type=\"bar\")","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}