{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\ntez_path = '../input/tez-modified-tqdm/'\neffnet_path = '../input/pytorch-efficientnet'\nimport sys,glob\nsys.path.append(tez_path)\nsys.path.append(effnet_path)\nimport librosa\nimport cv2,os\nimport numpy as np # linear algebra\nimport pandas as pd\nimport seaborn as sns\nfrom pathlib import Path\nimport os,random\nimport matplotlib.pyplot as plt\nimport gc\nimport matplotlib.image as immg\nfrom tqdm.notebook import tqdm\nimport albumentations\nimport tez\nimport torch\nimport torch.nn as nn\nfrom torch.nn import functional as F\nfrom efficientnet_pytorch import EfficientNet","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:44.503859Z","iopub.execute_input":"2021-09-02T18:31:44.504632Z","iopub.status.idle":"2021-09-02T18:31:46.411676Z","shell.execute_reply.started":"2021-09-02T18:31:44.504507Z","shell.execute_reply":"2021-09-02T18:31:46.410596Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import warnings\nwarnings.filterwarnings('ignore')","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:51.608398Z","iopub.execute_input":"2021-09-02T18:31:51.608782Z","iopub.status.idle":"2021-09-02T18:31:51.613071Z","shell.execute_reply.started":"2021-09-02T18:31:51.608744Z","shell.execute_reply":"2021-09-02T18:31:51.611854Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!pip install -q nnAudio","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/g2net-gravitational-wave-detection/training_labels.csv')\ntest = pd.read_csv('../input/g2net-gravitational-wave-detection/sample_submission.csv')\n\ndef get_train_file_path(image_id):\n    return \"../input/g2net-gravitational-wave-detection/train/{}/{}/{}/{}.npy\".format(\n        image_id[0], image_id[1], image_id[2], image_id)\n\ndef get_test_file_path(image_id):\n    return \"../input/g2net-gravitational-wave-detection/test/{}/{}/{}/{}.npy\".format(\n        image_id[0], image_id[1], image_id[2], image_id)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:53.838656Z","iopub.execute_input":"2021-09-02T18:31:53.839011Z","iopub.status.idle":"2021-09-02T18:31:54.374289Z","shell.execute_reply.started":"2021-09-02T18:31:53.838980Z","shell.execute_reply":"2021-09-02T18:31:54.373268Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train['file_path'] = train['id'].apply(get_train_file_path)\ntest['file_path'] = test['id'].apply(get_test_file_path)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:54.375804Z","iopub.execute_input":"2021-09-02T18:31:54.376106Z","iopub.status.idle":"2021-09-02T18:31:55.095124Z","shell.execute_reply.started":"2021-09-02T18:31:54.376077Z","shell.execute_reply":"2021-09-02T18:31:55.094002Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:55.097230Z","iopub.execute_input":"2021-09-02T18:31:55.097688Z","iopub.status.idle":"2021-09-02T18:31:55.115751Z","shell.execute_reply.started":"2021-09-02T18:31:55.097642Z","shell.execute_reply":"2021-09-02T18:31:55.114675Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:55.117821Z","iopub.execute_input":"2021-09-02T18:31:55.118479Z","iopub.status.idle":"2021-09-02T18:31:55.132172Z","shell.execute_reply.started":"2021-09-02T18:31:55.118427Z","shell.execute_reply":"2021-09-02T18:31:55.131150Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"qtransform_params={\"sr\": 2048, \"fmin\": 20, \"fmax\": 1024, \"hop_length\": 32, \"bins_per_octave\": 8}","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:55.133447Z","iopub.execute_input":"2021-09-02T18:31:55.133800Z","iopub.status.idle":"2021-09-02T18:31:55.141931Z","shell.execute_reply.started":"2021-09-02T18:31:55.133768Z","shell.execute_reply":"2021-09-02T18:31:55.141144Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import torch\nfrom nnAudio.Spectrogram import CQT1992v2\n\ndef apply_qtransform(waves, transform=CQT1992v2(**qtransform_params)):\n    #waves = np.hstack(waves)\n    waves = waves / np.max(waves)\n    waves = torch.from_numpy(waves).float()\n    image = transform(waves)\n    return image\n\nfor i in range(5):\n    waves = np.load(train.loc[i, 'file_path'])\n    image = apply_qtransform(waves)\n    target = train.loc[i, 'target']\n    plt.imshow(image[0])\n    plt.title(f\"target: {target}\")\n    plt.show()\n","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:55.143247Z","iopub.execute_input":"2021-09-02T18:31:55.143675Z","iopub.status.idle":"2021-09-02T18:31:55.964955Z","shell.execute_reply.started":"2021-09-02T18:31:55.143638Z","shell.execute_reply":"2021-09-02T18:31:55.963662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"image.shape","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:55.966662Z","iopub.execute_input":"2021-09-02T18:31:55.967042Z","iopub.status.idle":"2021-09-02T18:31:55.973217Z","shell.execute_reply.started":"2021-09-02T18:31:55.967005Z","shell.execute_reply":"2021-09-02T18:31:55.972140Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#from https://www.kaggle.com/daisukelab/creating-fat2019-preprocessed-data\ndef mono_to_color(X, mean=None, std=None, norm_max=None, norm_min=None, eps=1e-6):\n    # Standardize\n    mean = mean or X.mean()\n    X = X - mean\n    std = std or X.std()\n    Xstd = X / (std + eps)\n    _min, _max = Xstd.min(), Xstd.max()\n    norm_max = norm_max or _max\n    norm_min = norm_min or _min\n    if (_max - _min) > eps:\n        # Normalize to [0, 255]\n        V = Xstd\n        V[V < norm_min] = norm_min\n        V[V > norm_max] = norm_max\n        V = 255 * (V - norm_min) / (norm_max - norm_min)\n        V = V.astype(np.uint8)\n    else:\n        # Just zero\n        V = np.zeros_like(Xstd, dtype=np.uint8)\n    return V\n\ndef build_spectrogram(file_loc,ax=1):\n    waves = np.load(file_loc)\n    image1,image2,image3 = apply_qtransform(waves[0]),apply_qtransform(waves[1]),apply_qtransform(waves[2])\n    M1,M2,M3 = image1.permute(1,2,0).numpy()[:,:,0],image2.permute(1,2,0).numpy()[:,:,0],image3.permute(1,2,0).numpy()[:,:,0]\n    M1,M2,M3 = mono_to_color(M1),mono_to_color(M2),mono_to_color(M3)\n    return np.concatenate([M1,M2,M3],axis=ax)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:55.974857Z","iopub.execute_input":"2021-09-02T18:31:55.975148Z","iopub.status.idle":"2021-09-02T18:31:55.986915Z","shell.execute_reply.started":"2021-09-02T18:31:55.975120Z","shell.execute_reply":"2021-09-02T18:31:55.985787Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img = build_spectrogram(train.loc[3566,'file_path'],);img.shape","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:55.988406Z","iopub.execute_input":"2021-09-02T18:31:55.988693Z","iopub.status.idle":"2021-09-02T18:31:56.021208Z","shell.execute_reply.started":"2021-09-02T18:31:55.988665Z","shell.execute_reply":"2021-09-02T18:31:56.020156Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(img)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:56.022433Z","iopub.execute_input":"2021-09-02T18:31:56.022762Z","iopub.status.idle":"2021-09-02T18:31:56.171535Z","shell.execute_reply.started":"2021-09-02T18:31:56.022728Z","shell.execute_reply":"2021-09-02T18:31:56.170549Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nimport gc\nimport zipfile\nfrom joblib import Parallel, delayed","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:56.258136Z","iopub.execute_input":"2021-09-02T18:31:56.258506Z","iopub.status.idle":"2021-09-02T18:31:56.263273Z","shell.execute_reply.started":"2021-09-02T18:31:56.258477Z","shell.execute_reply":"2021-09-02T18:31:56.262531Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train.head()","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:56.468191Z","iopub.execute_input":"2021-09-02T18:31:56.468725Z","iopub.status.idle":"2021-09-02T18:31:56.479247Z","shell.execute_reply.started":"2021-09-02T18:31:56.468680Z","shell.execute_reply":"2021-09-02T18:31:56.478292Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"files = train['file_path'].values\nOUT_TRAIN = 'TrainG2NET.zip'","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:57.034103Z","iopub.execute_input":"2021-09-02T18:31:57.034545Z","iopub.status.idle":"2021-09-02T18:31:57.039027Z","shell.execute_reply.started":"2021-09-02T18:31:57.034505Z","shell.execute_reply":"2021-09-02T18:31:57.038006Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"x_tot,x2_tot = [],[]\nbatch = 50\nwith zipfile.ZipFile(OUT_TRAIN, 'w') as img_out:\n    for idx in tqdm(range(0,len(files),batch)):\n        names = files[idx:idx+batch]\n        out = Parallel(n_jobs=-1)(delayed(build_spectrogram)(i) for i in names)\n        for s in range(len(out)):\n            img = out[s]\n            x_tot.append((img/255.0).mean())\n            x2_tot.append(((img/255.0)**2).mean()) \n            name = names[s].split('/')[-1].split('.')[0]\n            img = cv2.imencode('.png',img)[1]\n            img_out.writestr(name + '.png', img)","metadata":{"execution":{"iopub.status.busy":"2021-09-02T18:31:57.608330Z","iopub.execute_input":"2021-09-02T18:31:57.608827Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"img_avr =  np.array(x_tot).mean()\nimg_std =  np.sqrt(np.array(x2_tot).mean() - img_avr**2)\nprint('mean:',img_avr, ', std:', img_std)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split,StratifiedKFold","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df = train.sample(frac=1.,random_state=2021).copy()\ntrain_df['kfold'] = -1\ny = train_df['target'].values\nkf = StratifiedKFold(n_splits=5,random_state = 2021,shuffle = True)\nfor fold ,(trn_,val_ )in enumerate(kf.split(X=train_df,y=y)):\n    train_df.loc[val_,'kfold'] = fold","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.to_csv('g2net_train_kfold.csv',index=False)","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}