{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# fastai training with resnet34\nfastai is a great tool to create a strong baseline quickly. I'm learning about signal processing so there may be big errors in my approach :) ","metadata":{"papermill":{"duration":0.052055,"end_time":"2021-02-24T04:34:38.361345","exception":false,"start_time":"2021-02-24T04:34:38.30929","status":"completed"},"tags":[]}},{"cell_type":"code","source":"import pandas as pd\nimport numpy as np\nfrom fastai.vision.all import *\nimport pickle\nimport os\nimport torch\nimport librosa","metadata":{"_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","papermill":{"duration":4.169069,"end_time":"2021-02-24T04:35:14.564093","exception":false,"start_time":"2021-02-24T04:35:10.395024","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-01T15:03:22.443755Z","iopub.execute_input":"2021-07-01T15:03:22.444455Z","iopub.status.idle":"2021-07-01T15:03:24.105498Z","shell.execute_reply.started":"2021-07-01T15:03:22.444315Z","shell.execute_reply":"2021-07-01T15:03:24.104147Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def sample_to_mel(_id, is_test):\n    x = np.load(id2path(_id, is_test))\n    spectrogram = []\n    for i in range(3):\n        mel = librosa.feature.melspectrogram(x[i]/x[i].max(), \n                                                sr=2048,\n                                                n_mels=16,\n                                               hop_length=16)\n        mel = mel[:,:256]\n        mel = librosa.power_to_db(mel).astype(np.float32)\n        mel = mel.reshape(64,64)\n        spectrogram.append(mel)\n    spectrogram = np.stack(spectrogram)\n    return spectrogram","metadata":{"execution":{"iopub.status.busy":"2021-07-01T15:03:24.108353Z","iopub.execute_input":"2021-07-01T15:03:24.108786Z","iopub.status.idle":"2021-07-01T15:03:24.116753Z","shell.execute_reply.started":"2021-07-01T15:03:24.108752Z","shell.execute_reply":"2021-07-01T15:03:24.115546Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/g2net-gravitational-wave-detection/training_labels.csv')\ntest_df = pd.read_csv('../input/g2net-gravitational-wave-detection/sample_submission.csv')","metadata":{"papermill":{"duration":0.519139,"end_time":"2021-02-24T04:35:18.946309","exception":false,"start_time":"2021-02-24T04:35:18.42717","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-01T15:03:24.118432Z","iopub.execute_input":"2021-07-01T15:03:24.118771Z","iopub.status.idle":"2021-07-01T15:03:24.714688Z","shell.execute_reply.started":"2021-07-01T15:03:24.118737Z","shell.execute_reply":"2021-07-01T15:03:24.713515Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def id2path(id, is_test):\n    a, b, c = id[0], id[1], id[2]\n    if is_test: return f'../input/g2net-gravitational-wave-detection/test/{a}/{b}/{c}/{id}.npy'\n    return f'../input/g2net-gravitational-wave-detection/train/{a}/{b}/{c}/{id}.npy'","metadata":{"papermill":{"duration":4.157218,"end_time":"2021-02-24T04:35:23.420229","exception":false,"start_time":"2021-02-24T04:35:19.263011","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-01T15:03:24.716092Z","iopub.execute_input":"2021-07-01T15:03:24.716472Z","iopub.status.idle":"2021-07-01T15:03:24.723061Z","shell.execute_reply.started":"2021-07-01T15:03:24.716438Z","shell.execute_reply":"2021-07-01T15:03:24.722019Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = train.sample(frac=0.3).reset_index(drop=True)","metadata":{"execution":{"iopub.status.busy":"2021-07-01T15:03:24.724476Z","iopub.execute_input":"2021-07-01T15:03:24.724834Z","iopub.status.idle":"2021-07-01T15:03:24.789176Z","shell.execute_reply.started":"2021-07-01T15:03:24.72479Z","shell.execute_reply":"2021-07-01T15:03:24.788093Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class NumpyDataset(torch.utils.data.Dataset):\n    def __init__(self, df, is_test=False):\n        self.df,self.is_test = df,is_test\n        \n    def __getitem__(self, i):\n        image_id = self.df['id'].loc[i]\n        img = sample_to_mel(image_id, self.is_test)\n        if self.is_test:\n            tgt = 0 if i < 10 else 1\n            return (torch.tensor(img, dtype=torch.float), torch.tensor(tgt, dtype=torch.long))\n        else:\n            tgt = self.df['target'].loc[i]\n            return (torch.tensor(img, dtype=torch.float), torch.tensor(tgt, dtype=torch.long))\n    \n    def __len__(self): return len(self.df)","metadata":{"execution":{"iopub.status.busy":"2021-07-01T15:03:24.790737Z","iopub.execute_input":"2021-07-01T15:03:24.791053Z","iopub.status.idle":"2021-07-01T15:03:24.800582Z","shell.execute_reply.started":"2021-07-01T15:03:24.791022Z","shell.execute_reply":"2021-07-01T15:03:24.79924Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cut = int(0.8 * len(train))\ntrain_df = train[:cut].reset_index(drop=True)\nvalid_df = train[cut:].reset_index(drop=True)\nlen(train_df), len(valid_df)","metadata":{"execution":{"iopub.status.busy":"2021-07-01T15:03:24.801951Z","iopub.execute_input":"2021-07-01T15:03:24.802282Z","iopub.status.idle":"2021-07-01T15:03:24.83082Z","shell.execute_reply.started":"2021-07-01T15:03:24.802244Z","shell.execute_reply":"2021-07-01T15:03:24.829589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_ds = NumpyDataset(train_df, is_test=False)\nvalid_ds = NumpyDataset(valid_df, is_test=False)\ntest_ds = NumpyDataset(test_df, is_test=True)","metadata":{"papermill":{"duration":0.28233,"end_time":"2021-02-24T04:35:26.014346","exception":false,"start_time":"2021-02-24T04:35:25.732016","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-01T15:03:24.833691Z","iopub.execute_input":"2021-07-01T15:03:24.834023Z","iopub.status.idle":"2021-07-01T15:03:24.841792Z","shell.execute_reply.started":"2021-07-01T15:03:24.833991Z","shell.execute_reply":"2021-07-01T15:03:24.840342Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"dls = DataLoaders.from_dsets(train_ds, valid_ds, bs=16)\ndls.c = 1","metadata":{"execution":{"iopub.status.busy":"2021-07-01T15:03:24.843882Z","iopub.execute_input":"2021-07-01T15:03:24.844271Z","iopub.status.idle":"2021-07-01T15:03:24.856285Z","shell.execute_reply.started":"2021-07-01T15:03:24.844197Z","shell.execute_reply":"2021-07-01T15:03:24.855438Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn = cnn_learner(dls, resnet34, loss_func=BCEWithLogitsLossFlat(), metrics=RocAucBinary())","metadata":{"papermill":{"duration":5.522109,"end_time":"2021-02-24T04:35:31.85855","exception":false,"start_time":"2021-02-24T04:35:26.336441","status":"completed"},"tags":[],"execution":{"iopub.status.busy":"2021-07-01T15:03:24.857344Z","iopub.execute_input":"2021-07-01T15:03:24.857763Z","iopub.status.idle":"2021-07-01T15:03:25.737799Z","shell.execute_reply.started":"2021-07-01T15:03:24.857731Z","shell.execute_reply":"2021-07-01T15:03:25.736883Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.fit_one_cycle(1, 3e-4)","metadata":{"execution":{"iopub.status.busy":"2021-07-01T15:03:25.73893Z","iopub.execute_input":"2021-07-01T15:03:25.739406Z","iopub.status.idle":"2021-07-01T16:14:53.313298Z","shell.execute_reply.started":"2021-07-01T15:03:25.739372Z","shell.execute_reply":"2021-07-01T16:14:53.31174Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.save('model')","metadata":{"execution":{"iopub.status.busy":"2021-07-01T16:14:53.315793Z","iopub.execute_input":"2021-07-01T16:14:53.316296Z","iopub.status.idle":"2021-07-01T16:14:53.56014Z","shell.execute_reply.started":"2021-07-01T16:14:53.316242Z","shell.execute_reply":"2021-07-01T16:14:53.559038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"learn.recorder.plot_loss()","metadata":{"execution":{"iopub.status.busy":"2021-07-01T16:14:53.561522Z","iopub.execute_input":"2021-07-01T16:14:53.561837Z","iopub.status.idle":"2021-07-01T16:14:53.884352Z","shell.execute_reply.started":"2021-07-01T16:14:53.561808Z","shell.execute_reply":"2021-07-01T16:14:53.883314Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Inference","metadata":{}},{"cell_type":"code","source":"test_dl = DataLoader(test_ds, bs=16, shuffle=False, drop_last=False)\npreds, _ = learn.get_preds(dl=test_dl)","metadata":{"execution":{"iopub.status.busy":"2021-07-01T16:14:53.886001Z","iopub.execute_input":"2021-07-01T16:14:53.88641Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.target = np.array(preds)\ntest_df.to_csv('submission.csv', index=False)\ntest_df.head()","metadata":{},"execution_count":null,"outputs":[]}]}