{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Introduction\n\nThis Notebook runs for the G2Net Gravitational Wave Detection Kaggle competition started early July 2021.\n\nThere are already several interesting public EDAs for the data. Thanks these help me greatly.<BR>\nI will be kinda quick on this part here. \n\nThe main focus here is building a TensorFlow pipeline mostly relying on tf.data.Dataset and other build in mechanism preparing provided data in order to feed a Machine Learning model.\n ","metadata":{"_kg_hide-input":false}},{"cell_type":"markdown","source":"# TODO\n### TO FIX\n- Tensorflow io installation facing issues.\n- Model  training seems to heavily bottleneck on CPU and would last forever. \nLinked to tfio issue or Dataset handling here? <BR>\nSame code @home 4790K/RTX2060 runs 5-15% SSD, 20-30%CPU and 90-100%GPU? To be investigated.\n\n### Spectrogram\n- Learning and tuning\n- In depth data analysis\n\n### ML Model architecture and tuning \n- First have to fix GPU handling\n    \n## Latest changes\n    - Few typos and code presentation\n    - Spectrogram output size lowered to (64, 65)\n    - Train dataset spectrogram plotting layout\n    - Model fit from loaded dataset\n    - Model example simplified and minimum records for fast execution w/o GPU \n    \n### Changes history","metadata":{}},{"cell_type":"code","source":"# ### 09/07/2021\n#    - Saved dataset from tf.data.experimental save method could not be read back correctly.\n#     => Fixed by upgrading tensorflow to version 2.5, can now be loaded with the load method. \n#     - Lowered saved dataset records count keeping in 5-6 GB range size for confort. \n\n\n","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-07-09T16:33:39.748848Z","iopub.execute_input":"2021-07-09T16:33:39.749520Z","iopub.status.idle":"2021-07-09T16:33:39.755600Z","shell.execute_reply.started":"2021-07-09T16:33:39.749460Z","shell.execute_reply":"2021-07-09T16:33:39.754573Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Upgrade to TensorFlow >= 2.5\n!pip uninstall pytorch-lightning -y\n!pip install tensorflow -U","metadata":{"scrolled":true,"_kg_hide-output":true,"execution":{"iopub.status.busy":"2021-07-09T16:33:42.056406Z","iopub.execute_input":"2021-07-09T16:33:42.056816Z","iopub.status.idle":"2021-07-09T16:33:53.537368Z","shell.execute_reply.started":"2021-07-09T16:33:42.056782Z","shell.execute_reply":"2021-07-09T16:33:53.535993Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Build Dataset \n\n## Data presentation","metadata":{}},{"cell_type":"code","source":"# This Python 3 environment comes with many helpful analytics libraries installed\n# It is defined by the kaggle/python Docker image: https://github.com/kaggle/docker-python\n# For example, here's several helpful packages to load\n\nimport numpy as np # linear algebra\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport seaborn as sns\nfrom matplotlib import pyplot as plt\n# Input data files are available in the read-only \"../input/\" directory\n# For example, running this (by clicking run or pressing Shift+Enter) will list all files under the input directory\n\nimport os\n# for dirname, _, filenames in os.walk('/kaggle/input'):\n# for dirname, _, filenames in os.walk('../input'):\n#     for filename in filenames:\n#         print(os.path.join(dirname, filename))\n\n# import math\nfrom sklearn.utils import shuffle\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nimport tensorflow.keras.layers as Layer\nfrom tensorflow.keras.preprocessing import image\n\n\n# You can write up to 20GB to the current directory (/kaggle/working/) that gets preserved as output when you create a version using \"Save & Run All\" \n# You can also write temporary files to /kaggle/temp/, but they won't be saved outside of the current session","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-input":true,"execution":{"iopub.status.busy":"2021-07-09T16:33:53.539859Z","iopub.execute_input":"2021-07-09T16:33:53.540257Z","iopub.status.idle":"2021-07-09T16:33:57.722348Z","shell.execute_reply.started":"2021-07-09T16:33:53.540223Z","shell.execute_reply":"2021-07-09T16:33:57.720768Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Tensorflow version: \", tf.__version__)","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:33:57.724266Z","iopub.execute_input":"2021-07-09T16:33:57.724596Z","iopub.status.idle":"2021-07-09T16:33:57.734202Z","shell.execute_reply.started":"2021-07-09T16:33:57.724565Z","shell.execute_reply":"2021-07-09T16:33:57.732797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Base repository\nBASE_PATH = '../input/g2net-gravitational-wave-detection/'\nSPLIT_RATIO = .9\nRANDOM_STATE = 42","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:33:57.736920Z","iopub.execute_input":"2021-07-09T16:33:57.737684Z","iopub.status.idle":"2021-07-09T16:33:57.764630Z","shell.execute_reply.started":"2021-07-09T16:33:57.737634Z","shell.execute_reply":"2021-07-09T16:33:57.763698Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"## Consolidate the data in a pandas DataFrame ","metadata":{}},{"cell_type":"code","source":"# \ndf = pd.read_csv(os.path.join(BASE_PATH, 'training_labels.csv'))\n#sample_submission = pd.read_csv('../input/g2net-gravitational-wave-detection/sample_submission.csv')\ndf.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:33:57.765975Z","iopub.execute_input":"2021-07-09T16:33:57.766490Z","iopub.status.idle":"2021-07-09T16:33:58.458474Z","shell.execute_reply.started":"2021-07-09T16:33:57.766444Z","shell.execute_reply":"2021-07-09T16:33:58.457380Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def apply_raw_path(row, is_train=True): \n    file_name = row[0]\n    if is_train:\n        return os.path.join(\n            BASE_PATH, 'train',\n            file_name[0],\n            file_name[1],\n            file_name[2],\n            file_name + \".npy\")\n    else:\n        return os.path.join(\n            BASE_PATH, 'test',\n            file_name[0],\n            file_name[1],\n            file_name[2],\n            file_name + \".npy\")\n\ndf['file_path'] = df.apply(apply_raw_path, args=(True,), axis=1)\ndf['target'] = df['target'].astype('int8')","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:33:58.460161Z","iopub.execute_input":"2021-07-09T16:33:58.460528Z","iopub.status.idle":"2021-07-09T16:34:07.343773Z","shell.execute_reply.started":"2021-07-09T16:33:58.460485Z","shell.execute_reply":"2021-07-09T16:34:07.342419Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:07.345507Z","iopub.execute_input":"2021-07-09T16:34:07.345884Z","iopub.status.idle":"2021-07-09T16:34:07.359511Z","shell.execute_reply.started":"2021-07-09T16:34:07.345804Z","shell.execute_reply":"2021-07-09T16:34:07.358458Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's have a look at the data","metadata":{}},{"cell_type":"code","source":"sample_num = 2579\n\ndef plot_samples(sample_list):\n    fig,a =  plt.subplots(len(sample_list), 3, figsize=(15,2 * len(sample_list)))\n    plt.subplots_adjust(top = 0.99, bottom=0.01, hspace=.5, wspace=0.2)\n    cpt = 0\n    for sample_num in sample_list:\n        sample = np.load(df.loc[sample_num]['file_path'])\n        a[cpt, 0].plot(sample[0],color='red')\n        a[cpt, 1].plot(sample[1],color='green')\n        a[cpt, 2].plot(sample[2],color='cyan')\n        a[cpt, 1].set_title(\n            'sample ' + str(sample_num) +\n            ', label ' + str(df.loc[sample_num]['target']),\n            fontsize=16\n        )\n        cpt = cpt + 1\n    plt.show()\n    return sample.shape\n\nsamples = [0, 1, 2, 2547]\nplot_samples(samples)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:07.362176Z","iopub.execute_input":"2021-07-09T16:34:07.362507Z","iopub.status.idle":"2021-07-09T16:34:08.817578Z","shell.execute_reply.started":"2021-07-09T16:34:07.362473Z","shell.execute_reply":"2021-07-09T16:34:08.816440Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(\"Sample shape: \",str(np.load(df.loc[2547]['file_path']).shape))\nprint(\"Number of records:\", str(len(df)))","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:08.819478Z","iopub.execute_input":"2021-07-09T16:34:08.819820Z","iopub.status.idle":"2021-07-09T16:34:08.828660Z","shell.execute_reply.started":"2021-07-09T16:34:08.819779Z","shell.execute_reply":"2021-07-09T16:34:08.827352Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"So now we have acces to the data, that is\n- 560 000 records \n- Of shape (3, 4096)\nEach record includes 3 channels each containing a wave signal sampled at 2048Hz over 2 seconds.\nEach channel comes from either one of the two LIGOs either the VIRGO\n\nNow let's build a TensorFlow DataSet preparing data for futur use.","metadata":{}},{"cell_type":"markdown","source":"# TensorFlow Dataset preparation","metadata":{"execution":{"iopub.status.busy":"2021-07-08T15:52:13.377827Z","iopub.execute_input":"2021-07-08T15:52:13.378482Z","iopub.status.idle":"2021-07-08T15:52:13.74373Z","shell.execute_reply.started":"2021-07-08T15:52:13.378428Z","shell.execute_reply":"2021-07-08T15:52:13.742059Z"}}},{"cell_type":"markdown","source":"First thing, dataset are only working sequentially and cannot skip (there is a skip method yet it only discards, not really skip).\n\nConsidering the whole data volumetry:\n- If any full shuffling required it has to be performed on the ds_train dataframe as prior,\n- If any validation data to be reserved, rather prepare a spedcific dataset from a dedicated dataframe.\n\nTherefore we will first shuffle the df dataframe then split in train and validation.  ","metadata":{}},{"cell_type":"code","source":"df.head()","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:08.830289Z","iopub.execute_input":"2021-07-09T16:34:08.830823Z","iopub.status.idle":"2021-07-09T16:34:08.848114Z","shell.execute_reply.started":"2021-07-09T16:34:08.830785Z","shell.execute_reply":"2021-07-09T16:34:08.846971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### First shuffle and split","metadata":{}},{"cell_type":"code","source":"df_train, df_valid = train_test_split(\n    df,\n    test_size=1-SPLIT_RATIO,\n    train_size=SPLIT_RATIO,\n    random_state=RANDOM_STATE)\nprint(len(df_train))\nprint(len(df_valid))\ndisplay(df_train.head())\nprint(df_train.loc[0])","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:08.849837Z","iopub.execute_input":"2021-07-09T16:34:08.850349Z","iopub.status.idle":"2021-07-09T16:34:09.287865Z","shell.execute_reply.started":"2021-07-09T16:34:08.850304Z","shell.execute_reply":"2021-07-09T16:34:09.286495Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Second step the numpy format\n\nAs per my understanding TF does not provide yet an easy way feeding from numpy file type.\nStill we can work our way out handling known data structure from files leveraging:\ntf.data.FixedLengthRecordDataset\n\nThis will require providing with the specs of the data.\n\nThe below function found somewhere on the Internet will parse the numpy header returning its length.<BR>\nBtw can be found at https://stackoverflow.com/questions/48889482/feeding-npy-numpy-files-into-tensorflow-data-pipeline","metadata":{}},{"cell_type":"code","source":"def npy_header_offset(npy_path):\n    with open(str(npy_path), 'rb') as f:\n        if f.read(6) != b'\\x93NUMPY':\n            raise ValueError('Invalid NPY file.')\n        version_major, version_minor = f.read(2)\n        if version_major == 1:\n            header_len_size = 2\n        elif version_major == 2:\n            header_len_size = 4\n        else:\n            raise ValueError('Unknown NPY file version {}.{}.'.format(version_major, version_minor))\n        header_len = sum(b << (8 * i) for i, b in enumerate(f.read(header_len_size)))\n        header = f.read(header_len)\n        if not header.endswith(b'\\n'):\n            raise ValueError('Invalid NPY file.')\n        return f.tell()","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:09.290123Z","iopub.execute_input":"2021-07-09T16:34:09.290506Z","iopub.status.idle":"2021-07-09T16:34:09.298417Z","shell.execute_reply.started":"2021-07-09T16:34:09.290472Z","shell.execute_reply":"2021-07-09T16:34:09.297430Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# records chosen in train df indexes / raise error if was allocated to test df\nfor record_num in [0, 1, 3, 25, 2586, 54863, 258412, 559998]:\n    file_length = os.path.getsize(df_train.loc[record_num]['file_path'])\n    header_size = npy_header_offset(df_train.loc[record_num]['file_path'])\n    print(\n        \"File length:\", file_length,\n        \", Header size: \",  header_size,\n        \", Data length\", file_length - header_size\n    )","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:09.299717Z","iopub.execute_input":"2021-07-09T16:34:09.300251Z","iopub.status.idle":"2021-07-09T16:34:09.401214Z","shell.execute_reply.started":"2021-07-09T16:34:09.300220Z","shell.execute_reply":"2021-07-09T16:34:09.399753Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Good let's rock\nwe need: \n- FixedLengthRecordDataset will read the file (can leverage threading btw preventing from cpu/io bottleneck)\n- Next tf.io.decode_raw will unflat the data to our expected format (3, 4096)\nStill 3 x 4096 != 98304, in fact 98304 / ( 3x4096) = 8 ;<BR> Guess we are facing float 64 data type assuming file type is byte.\n","metadata":{}},{"cell_type":"code","source":"ds_train_data = tf.data.FixedLengthRecordDataset(\n    df_train['file_path'],\n    98304,\n    header_bytes=128,\n    num_parallel_reads=4)\nds_train_data = ds_train_data.map(lambda s: tf.reshape(tf.io.decode_raw(s, tf.float64), (3,4096)))\nds_train_data = ds_train_data.map(lambda s: tf.cast(s, tf.float32))","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:09.403189Z","iopub.execute_input":"2021-07-09T16:34:09.403641Z","iopub.status.idle":"2021-07-09T16:34:10.018696Z","shell.execute_reply.started":"2021-07-09T16:34:09.403593Z","shell.execute_reply":"2021-07-09T16:34:10.017606Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Let's add the labels to our data the zip way","metadata":{}},{"cell_type":"code","source":"ds_train_label = tf.data.Dataset.from_tensor_slices(df_train['target'])\nds_train = tf.data.Dataset.zip((ds_train_data, ds_train_label)) ","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:10.020160Z","iopub.execute_input":"2021-07-09T16:34:10.020479Z","iopub.status.idle":"2021-07-09T16:34:10.033007Z","shell.execute_reply.started":"2021-07-09T16:34:10.020450Z","shell.execute_reply":"2021-07-09T16:34:10.032072Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Does it work?<BR>\nLet's adapt previous plot function to work with dataset data:","metadata":{}},{"cell_type":"code","source":"def plot_from_dataset(data, label):\n    fig,a =  plt.subplots(1, 3, figsize=(15,2))\n    plt.subplots_adjust(top = 0.99, bottom=0.01, hspace=.5, wspace=0.2)\n    a[0].plot(data[0],color='red')\n    a[1].plot(data[1],color='green')\n    a[2].plot(data[2],color='cyan')\n    a[1].set_title(', label ' + str(label), fontsize=16)\n    plt.show()\n    return data.shape","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:10.034605Z","iopub.execute_input":"2021-07-09T16:34:10.035253Z","iopub.status.idle":"2021-07-09T16:34:10.044511Z","shell.execute_reply.started":"2021-07-09T16:34:10.035201Z","shell.execute_reply":"2021-07-09T16:34:10.043473Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for data, label in ds_train.take(3):\n    plot_from_dataset(data.numpy(), label.numpy())","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:11.856317Z","iopub.execute_input":"2021-07-09T16:34:11.856725Z","iopub.status.idle":"2021-07-09T16:34:12.976833Z","shell.execute_reply.started":"2021-07-09T16:34:11.856692Z","shell.execute_reply":"2021-07-09T16:34:12.975558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"df_train.index[100]","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:15.601777Z","iopub.execute_input":"2021-07-09T16:34:15.602208Z","iopub.status.idle":"2021-07-09T16:34:15.609217Z","shell.execute_reply.started":"2021-07-09T16:34:15.602171Z","shell.execute_reply":"2021-07-09T16:34:15.608173Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Good looks pretty solid,\nLet's check number wise:\n\nNote: relying on df_train.index matching the right element from dataset hence the:<BR>\n    df_train.loc[df_train.index[100]]","metadata":{}},{"cell_type":"code","source":"check_game_num = 15\nfor data, label in ds_train.skip(check_game_num).take(1):\n    data_from_ds = data.numpy()\n    print(data_from_ds)\n    print(\"label:\", label.numpy() )\n\ndata_from_file = np.load(\n    df_train.loc[df_train.index[check_game_num]]['file_path']).astype(np.float32)\nprint(data_from_file)\nprint(\"label:\", df_train.loc[df_train.index[check_game_num]]['target'])\n\nprint(\"\")\nif (data_from_ds==data_from_file).all():\n    print(\"Hello World!\")\nelse:\n    print(\"Something went wrong.\")","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:17.375481Z","iopub.execute_input":"2021-07-09T16:34:17.375903Z","iopub.status.idle":"2021-07-09T16:34:17.464794Z","shell.execute_reply.started":"2021-07-09T16:34:17.375867Z","shell.execute_reply":"2021-07-09T16:34:17.463295Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset transformation (Spectrogram)","metadata":{}},{"cell_type":"markdown","source":"To sum up we now have at disposal a shuffled train TensorFlow Dataset able to read data from provided files.\n- Data have shape (3,4096) float 32 type\n- Labels have shape (1) type int8\n\nNow imagine we'd like to leverage a convolutional neural network could be a ResNet, EfficientNet, or other famous ones expecting 2 dimensional image like inputs in RGB mode.\nThis because I saw in other notebooks here some spectrograms transformation sounds quite relevant for the classification task we target.\n\nI will below propose an implementation idea I have no clue whether it's relevant or not.<BR>\nWorth mentioning here I'm new to Machine Learning and worse in regards to signal analysis and tranformation, so please forgive the chosen values if aberrant.\n    \nSo my basic idea is why not transforming each wave into a spectrogram and leveraging one of the 3 RGB channel per wave. \n3 RGB channel xpected, 3 waves at disposal... KIS\n    \nThe data shape tranformation mechanic is the focus here.    \n","metadata":{}},{"cell_type":"code","source":"# We start from: \nprint(ds_train.cardinality)\nprint(ds_train_data.cardinality)","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:21.607804Z","iopub.execute_input":"2021-07-09T16:34:21.608246Z","iopub.status.idle":"2021-07-09T16:34:21.616600Z","shell.execute_reply.started":"2021-07-09T16:34:21.608209Z","shell.execute_reply":"2021-07-09T16:34:21.614878Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# lets leverage the tfio.audio TensorFlow tfio library\n!pip install  tensorflow_io\nimport tensorflow_io as tfio\n","metadata":{"_kg_hide-output":true,"scrolled":true,"execution":{"iopub.status.busy":"2021-07-09T16:34:22.777353Z","iopub.execute_input":"2021-07-09T16:34:22.777906Z","iopub.status.idle":"2021-07-09T16:34:36.015299Z","shell.execute_reply.started":"2021-07-09T16:34:22.777870Z","shell.execute_reply":"2021-07-09T16:34:36.013717Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_train_data = tf.data.FixedLengthRecordDataset(\n    df_train['file_path'],\n    98304,\n    header_bytes=128,\n    num_parallel_reads=4)\nds_train_data = ds_train_data.map(\n    lambda s: tf.reshape(\n        tf.io.decode_raw(\n            s, tf.float64), (3,4096)))\nds_train_data = ds_train_data.map(\n    lambda s: tf.cast(s, tf.float32))\nds_train_data = ds_train_data.map(\n    lambda s: tfio.audio.spectrogram(\n        s,\n        nfft=128,\n        window=256,\n        stride=64))\nprint(ds_train_data.cardinality)\n\n# The nfft and stride selected here allow for output size from (4096) to (256,129)\n# My guess is the transformation should consider both expected relevant signal transformation and sizing complying with models ","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:36.018900Z","iopub.execute_input":"2021-07-09T16:34:36.019397Z","iopub.status.idle":"2021-07-09T16:34:36.565084Z","shell.execute_reply.started":"2021-07-09T16:34:36.019357Z","shell.execute_reply":"2021-07-09T16:34:36.563863Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for data in ds_train_data.take(1):\n    print(data)","metadata":{"scrolled":true,"execution":{"iopub.status.busy":"2021-07-09T16:34:36.566727Z","iopub.execute_input":"2021-07-09T16:34:36.567162Z","iopub.status.idle":"2021-07-09T16:34:36.672955Z","shell.execute_reply.started":"2021-07-09T16:34:36.567128Z","shell.execute_reply":"2021-07-09T16:34:36.672120Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# tfio.audio also provides with a melscale method might be leveraged","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:39.967725Z","iopub.execute_input":"2021-07-09T16:34:39.968566Z","iopub.status.idle":"2021-07-09T16:34:39.973759Z","shell.execute_reply.started":"2021-07-09T16:34:39.968509Z","shell.execute_reply":"2021-07-09T16:34:39.972256Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# The below applies a log transformation\n# the main goal is transforming inputs to a [0,1] compliant with expected inputs hence the shift and scale\n# The clip_by_value prevents from log(0) nan.\n# Obviously the clip values will have to be set from consistant data analysis # TODO #  ","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:40.424659Z","iopub.execute_input":"2021-07-09T16:34:40.425198Z","iopub.status.idle":"2021-07-09T16:34:40.430868Z","shell.execute_reply.started":"2021-07-09T16:34:40.425155Z","shell.execute_reply":"2021-07-09T16:34:40.429178Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"ds_train_data = ds_train_data.map(\n    lambda s: (\n        tf.math.log(\n            tf.clip_by_value(s,1e-30, 1e-17)) + 60)/25)\nprint(ds_train_data.cardinality)","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:42.445456Z","iopub.execute_input":"2021-07-09T16:34:42.445908Z","iopub.status.idle":"2021-07-09T16:34:42.500546Z","shell.execute_reply.started":"2021-07-09T16:34:42.445870Z","shell.execute_reply":"2021-07-09T16:34:42.498577Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# this has yet to be transposed for each wave becoming one of the RGB channels expected last by TensorFlow\nds_train_rgb = ds_train_data.map(lambda s: tf.transpose(s))\nprint(ds_train_rgb.cardinality)\n\n# include labels in dataset the zip way\nds_train = tf.data.Dataset.zip((ds_train_rgb, ds_train_label))","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:34:43.492837Z","iopub.execute_input":"2021-07-09T16:34:43.493339Z","iopub.status.idle":"2021-07-09T16:34:43.574301Z","shell.execute_reply.started":"2021-07-09T16:34:43.493297Z","shell.execute_reply":"2021-07-09T16:34:43.572905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Let's try plotting this","metadata":{"execution":{"iopub.status.busy":"2021-07-08T15:44:16.506584Z","iopub.execute_input":"2021-07-08T15:44:16.506878Z","iopub.status.idle":"2021-07-08T15:44:16.515961Z","shell.execute_reply.started":"2021-07-08T15:44:16.506851Z","shell.execute_reply":"2021-07-08T15:44:16.514991Z"}}},{"cell_type":"code","source":"fig,a =  plt.subplots(2,5, figsize=(18,8))\ncpt = 0\nmin_value = 1\nmax_value = 0\nfor data, label in ds_train.skip(15).take(10):\n#     data = tf.math.log(data)\n    local_max = np.amax(data.numpy())\n    if  local_max > max_value:\n        max_value = local_max \n    local_min = np.amin(data.numpy())\n    if  local_min < min_value:\n        min_value = local_min \n#     print(data)\n#     a.imshow(tf.math.log(data[0]))\n    line = cpt % 5\n    col = cpt // 5\n    a[col, line].imshow(data)\n    a[col, line].set_title(\n            'Label: ' + str(label.numpy()),\n            fontsize=16)\n    cpt = cpt + 1\n\nplt.show()\nprint(\"Min value: \", min_value)\nprint(\"Max value: \", max_value)\n","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:36:49.042793Z","iopub.execute_input":"2021-07-09T16:36:49.043279Z","iopub.status.idle":"2021-07-09T16:36:50.599424Z","shell.execute_reply.started":"2021-07-09T16:36:49.043243Z","shell.execute_reply":"2021-07-09T16:36:50.597952Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Build validation dataset ","metadata":{}},{"cell_type":"code","source":"# Build validation\nds_valid_label = tf.data.Dataset.from_tensor_slices(df_valid['target'])\nds_valid_data = tf.data.FixedLengthRecordDataset(\n    df_valid['file_path'],\n    98304,\n    header_bytes=128,\n    num_parallel_reads=4)\nds_valid_data = ds_valid_data.map(\n    lambda s: tf.reshape(tf.io.decode_raw(s, tf.float64), (3,4096)))\nds_valid_data = ds_valid_data.map(\n    lambda s: tf.cast(s, tf.float32))\nds_valid_data = ds_valid_data.map(\n    lambda s: tfio.audio.spectrogram(\n        s,\n        nfft=128,\n        window=256,\n        stride=64))\nds_valid_data = ds_valid_data.map(\n    lambda s: (tf.math.log(tf.clip_by_value(s,1e-30, 1e-17)) + 60)/25)\nds_valid_rgb = ds_valid_data.map(\n    lambda s: tf.transpose(s))\nds_valid = tf.data.Dataset.zip((ds_valid_rgb, ds_valid_label))","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:37:32.696693Z","iopub.execute_input":"2021-07-09T16:37:32.697189Z","iopub.status.idle":"2021-07-09T16:37:32.914578Z","shell.execute_reply.started":"2021-07-09T16:37:32.697138Z","shell.execute_reply":"2021-07-09T16:37:32.913630Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(ds_train.cardinality)\nprint(ds_valid.cardinality)","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:37:35.873160Z","iopub.execute_input":"2021-07-09T16:37:35.873806Z","iopub.status.idle":"2021-07-09T16:37:35.880129Z","shell.execute_reply.started":"2021-07-09T16:37:35.873766Z","shell.execute_reply":"2021-07-09T16:37:35.878920Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Dataset save and load\nThis is informative and should not be necessary yet might help identifying model fitting performances issues ","metadata":{}},{"cell_type":"code","source":"# Save  \ntf.data.experimental.save(ds_train.take(120000), './ds_train', compression=None) \ntf.data.experimental.save(ds_valid.take(12000), './ds_valid', compression=None) \n","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:37:40.902529Z","iopub.execute_input":"2021-07-09T16:37:40.903142Z","iopub.status.idle":"2021-07-09T16:38:47.450314Z","shell.execute_reply.started":"2021-07-09T16:37:40.903104Z","shell.execute_reply":"2021-07-09T16:38:47.448729Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Load \ntrain = tf.data.experimental.load('./ds_train')\nvalid = tf.data.experimental.load('./ds_valid')","metadata":{"execution":{"iopub.status.busy":"2021-07-09T16:39:59.547307Z","iopub.execute_input":"2021-07-09T16:39:59.547718Z","iopub.status.idle":"2021-07-09T16:39:59.582198Z","shell.execute_reply.started":"2021-07-09T16:39:59.547684Z","shell.execute_reply":"2021-07-09T16:39:59.581045Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Fit a RESNET 50 ","metadata":{}},{"cell_type":"code","source":"from tensorflow.keras.applications.resnet_v2 import ResNet50V2\nfrom tensorflow.keras.applications.resnet_v2 import preprocess_input\n\ndef prep(img, label):\n    ret = preprocess_input(img)\n    return ret,label","metadata":{"execution":{"iopub.status.busy":"2021-07-09T17:55:03.584099Z","iopub.execute_input":"2021-07-09T17:55:03.584516Z","iopub.status.idle":"2021-07-09T17:55:03.590641Z","shell.execute_reply.started":"2021-07-09T17:55:03.584485Z","shell.execute_reply":"2021-07-09T17:55:03.589560Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Limiting to 2000 record for fast run without CPU\nBATCH_SIZE = 16\n\ntrain_data = train.take(1000).map(prep).batch(BATCH_SIZE).prefetch(buffer_size=64)\nvalid_data = valid.take(100).map(prep).cache().batch(BATCH_SIZE).prefetch(buffer_size=64)","metadata":{"execution":{"iopub.status.busy":"2021-07-09T17:55:04.880718Z","iopub.execute_input":"2021-07-09T17:55:04.881135Z","iopub.status.idle":"2021-07-09T17:55:04.911441Z","shell.execute_reply.started":"2021-07-09T17:55:04.881091Z","shell.execute_reply":"2021-07-09T17:55:04.910492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"base_model = ResNet50V2(\n    weights='imagenet',\n    include_top=False,\n    input_shape=(65,64,3),\n    pooling='avg')\nbase_model.trainable=True\n# base_model.trainable=False\n\n\nfine_tune_at = 171\nfor layer in base_model.layers[:fine_tune_at]:\n    layer.trainable =  False","metadata":{"execution":{"iopub.status.busy":"2021-07-09T17:55:05.980008Z","iopub.execute_input":"2021-07-09T17:55:05.980423Z","iopub.status.idle":"2021-07-09T17:55:08.316960Z","shell.execute_reply.started":"2021-07-09T17:55:05.980391Z","shell.execute_reply":"2021-07-09T17:55:08.315989Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"metrics = [tf.keras.metrics.AUC(name='AUC'),\n           tf.keras.metrics.SparseCategoricalAccuracy(name='Accuracy'),\n          ]\n\ntb_callback = tf.keras.callbacks.TensorBoard(\"logs\")\n\ncallbacks= [\n    tb_callback,\n]\n\ndef create_model(alpha):\n    model = tf.keras.Sequential([\n        Layer.InputLayer(input_shape=(65,64,3)),\n        base_model,\n        Layer.Dropout(.3),\n        Layer.LeakyReLU(alpha=alpha),\n        Layer.Dense(64),\n        Layer.Dropout(.3),\n        Layer.LeakyReLU(alpha=alpha),\n        Layer.Dense(8),\n        Layer.Dropout(.3),\n        Layer.LeakyReLU(alpha=alpha),\n        Layer.Dense(1, activation='sigmoid'),\n    ])\n    return model\n","metadata":{"execution":{"iopub.status.busy":"2021-07-09T17:55:08.319672Z","iopub.execute_input":"2021-07-09T17:55:08.320159Z","iopub.status.idle":"2021-07-09T17:55:08.338482Z","shell.execute_reply.started":"2021-07-09T17:55:08.320110Z","shell.execute_reply":"2021-07-09T17:55:08.337175Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model = create_model(0)\nmodel.summary()","metadata":{"execution":{"iopub.status.busy":"2021-07-09T17:55:09.018575Z","iopub.execute_input":"2021-07-09T17:55:09.018973Z","iopub.status.idle":"2021-07-09T17:55:09.530693Z","shell.execute_reply.started":"2021-07-09T17:55:09.018937Z","shell.execute_reply":"2021-07-09T17:55:09.529505Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.compile(optimizer=tf.keras.optimizers.Adam(\n    learning_rate=1e-5,\n    ),\n    loss='binary_crossentropy', metrics=metrics)","metadata":{"execution":{"iopub.status.busy":"2021-07-09T17:55:10.690503Z","iopub.execute_input":"2021-07-09T17:55:10.690870Z","iopub.status.idle":"2021-07-09T17:55:10.705780Z","shell.execute_reply.started":"2021-07-09T17:55:10.690840Z","shell.execute_reply":"2021-07-09T17:55:10.704756Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.fit(\n    train_data,\n    epochs=2,\n    validation_data=valid_data,\n    callbacks = callbacks,\n    )","metadata":{"execution":{"iopub.status.busy":"2021-07-09T17:55:18.105951Z","iopub.execute_input":"2021-07-09T17:55:18.106325Z","iopub.status.idle":"2021-07-09T18:03:58.823109Z","shell.execute_reply.started":"2021-07-09T17:55:18.106295Z","shell.execute_reply":"2021-07-09T18:03:58.821995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{"execution":{"iopub.status.busy":"2021-07-09T17:18:17.410572Z","iopub.execute_input":"2021-07-09T17:18:17.411170Z","iopub.status.idle":"2021-07-09T17:18:17.423806Z","shell.execute_reply.started":"2021-07-09T17:18:17.411134Z","shell.execute_reply":"2021-07-09T17:18:17.422856Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}