{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import os\nimport numpy as np\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport pandas as pd\nimport math\nimport cv2\nimport librosa \nimport librosa.display\nimport IPython.display as ipd \nfrom sklearn.preprocessing import LabelEncoder\nfrom sklearn.model_selection import train_test_split\n\nos.environ['TF_CPP_MIN_LOG_LEVEL'] = '3' \n\nimport tensorflow as tf\nfrom tensorflow import keras\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nfrom tensorflow.keras.models import Sequential, load_model\nfrom tensorflow.keras.layers import *\nfrom tensorflow.keras import backend as K","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","execution":{"iopub.status.busy":"2022-03-06T21:50:22.134797Z","iopub.execute_input":"2022-03-06T21:50:22.135075Z","iopub.status.idle":"2022-03-06T21:50:29.009156Z","shell.execute_reply.started":"2022-03-06T21:50:22.135044Z","shell.execute_reply":"2022-03-06T21:50:29.008369Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def merge_history(hlist):\n    history = {}\n    for k in hlist[0].history.keys():\n        history[k] = sum([h.history[k] for h in hlist], [])\n    return history","metadata":{"execution":{"iopub.status.busy":"2022-03-05T22:21:42.90923Z","iopub.execute_input":"2022-03-05T22:21:42.909637Z","iopub.status.idle":"2022-03-05T22:21:42.914309Z","shell.execute_reply.started":"2022-03-05T22:21:42.909602Z","shell.execute_reply":"2022-03-05T22:21:42.913673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def vis_training(h, start=1):\n    epoch_range = range(start, len(h['loss'])+1)\n    s = slice(start-1, None)\n\n    plt.figure(figsize=[16,4])\n\n    n = int(len(h.keys()) / 2)\n\n    for i in range(n):\n        k = list(h.keys())[i]\n        plt.subplot(1,n,i+1)\n        plt.plot(epoch_range, h[k][s], label='Training')\n        plt.plot(epoch_range, h['val_' + k][s], label='Validation')\n        plt.xlabel('Epoch'); plt.ylabel(k); plt.title(k)\n        plt.grid()\n        plt.legend()\n\n    plt.tight_layout()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2022-03-05T22:21:44.700957Z","iopub.execute_input":"2022-03-05T22:21:44.701388Z","iopub.status.idle":"2022-03-05T22:21:44.708943Z","shell.execute_reply.started":"2022-03-05T22:21:44.701349Z","shell.execute_reply":"2022-03-05T22:21:44.708184Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train = pd.read_csv('../input/g2net-mel-spectrograms-20x9/spectrograms/training_labels.csv')\nprint(train.shape, '\\n')\ntrain.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-06T21:50:45.846247Z","iopub.execute_input":"2022-03-06T21:50:45.846824Z","iopub.status.idle":"2022-03-06T21:50:46.210215Z","shell.execute_reply.started":"2022-03-06T21:50:45.846785Z","shell.execute_reply":"2022-03-06T21:50:46.209504Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test = pd.read_csv('../input/g2net-mel-spectrograms-20x9/spectrograms/sample_submission.csv')\nprint(test.shape)\ntest.head()","metadata":{"execution":{"iopub.status.busy":"2022-03-06T00:45:53.596976Z","iopub.execute_input":"2022-03-06T00:45:53.597233Z","iopub.status.idle":"2022-03-06T00:45:53.746038Z","shell.execute_reply.started":"2022-03-06T00:45:53.597202Z","shell.execute_reply":"2022-03-06T00:45:53.745219Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SPEC_PATH = '../input/g2net-mel-spectrograms-20x9/spectrograms'\n\nclass DataGenerator(keras.utils.Sequence):\n    \n    def __init__(self, df, batch_size=32, img_size=(20, 9), shuffle=True, is_train=True):\n        self.df = df\n        self.n = len(df)\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.shuffle = shuffle\n        self.is_train = is_train\n        self.on_epoch_end()\n        \n    def on_epoch_end(self):\n        self.indices = np.arange(self.n)\n        if self.shuffle == True:\n            np.random.shuffle(self.indices)   \n    \n    def __len__(self):     \n        return math.ceil( self.n / self.batch_size )\n    \n    def __getitem__(self, batch_index):\n        start = batch_index * self.batch_size\n        end = (batch_index + 1) * self.batch_size\n        indices = self.indices[start:end]\n        \n        return self.__data_generation(indices)\n    \n    def __data_generation(self, batch_indices):\n        batch_size = len(batch_indices)\n        \n        X = np.zeros(shape=(batch_size, self.img_size[0], self.img_size[1], 3))\n        y = np.zeros(batch_size)\n        \n        for i, idx in enumerate(batch_indices):\n            ID = self.df.id.values[idx]\n            y[i] = self.df.target.values[idx]\n            \n            SET = 'train_spec' if self.is_train else 'test_spec'\n            path = f'{SPEC_PATH}/{SET}/{ID}.npy'\n            data_array = np.load(path)\n            \n            X[i,:,:,:] = data_array\n            \n        return X, y\n    \n\nGENERATOR_TEST = True\n\nif GENERATOR_TEST:\n    temp_gen = DataGenerator(train, batch_size=8, shuffle=False)\n    X,y = temp_gen.__getitem__(0)\n\n    print(X.shape)\n    print(y)","metadata":{"execution":{"iopub.status.busy":"2022-03-06T21:51:44.22446Z","iopub.execute_input":"2022-03-06T21:51:44.224729Z","iopub.status.idle":"2022-03-06T21:51:44.299408Z","shell.execute_reply.started":"2022-03-06T21:51:44.224699Z","shell.execute_reply":"2022-03-06T21:51:44.298637Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df, valid_df = train_test_split(train, test_size=0.15)\n\nprint(train_df.shape)\nprint(valid_df.shape)","metadata":{"execution":{"iopub.status.busy":"2022-03-05T22:22:18.084441Z","iopub.execute_input":"2022-03-05T22:22:18.085107Z","iopub.status.idle":"2022-03-05T22:22:18.194835Z","shell.execute_reply.started":"2022-03-05T22:22:18.08507Z","shell.execute_reply":"2022-03-05T22:22:18.193958Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test","metadata":{"execution":{"iopub.status.busy":"2022-03-06T00:47:24.050314Z","iopub.execute_input":"2022-03-06T00:47:24.050599Z","iopub.status.idle":"2022-03-06T00:47:24.064002Z","shell.execute_reply.started":"2022-03-06T00:47:24.050569Z","shell.execute_reply":"2022-03-06T00:47:24.063238Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_loader = DataGenerator(train_df, batch_size=2048, shuffle=True)\nvalid_loader = DataGenerator(valid_df, batch_size=2048, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-05T22:22:19.894659Z","iopub.execute_input":"2022-03-05T22:22:19.895213Z","iopub.status.idle":"2022-03-05T22:22:19.913376Z","shell.execute_reply.started":"2022-03-05T22:22:19.895172Z","shell.execute_reply":"2022-03-05T22:22:19.912605Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"np.random.seed(1)\ncnn = Sequential()\n\ncnn.add(Conv2D(32, (3,3), activation = 'elu', padding = 'same', input_shape=(20,9,3)))\ncnn.add(Conv2D(32, (3,3), activation = 'elu', padding = 'same'))\ncnn.add(MaxPooling2D(2,2))\ncnn.add(Dropout(0.20))\ncnn.add(BatchNormalization())\n\ncnn.add(Conv2D(64, (3,3), activation = 'elu', padding = 'same'))\ncnn.add(Conv2D(64, (3,3), activation = 'elu', padding = 'same'))\ncnn.add(MaxPooling2D(2,2))\ncnn.add(Dropout(0.20))\ncnn.add(BatchNormalization())\n\ncnn.add(Conv2D(128, (3,3), activation = 'elu', padding = 'same'))\ncnn.add(Conv2D(128, (3,3), activation = 'elu', padding = 'same'))\ncnn.add(MaxPooling2D(2,2))\ncnn.add(Dropout(0.20))\ncnn.add(BatchNormalization())\n\ncnn.add(Conv2D(128, (3,3), activation = 'elu', padding = 'same'))\ncnn.add(Conv2D(128, (3,3), activation = 'elu', padding = 'same'))\ncnn.add(Dropout(0.20))\ncnn.add(BatchNormalization())\n\ncnn.add(Flatten())\n\ncnn.add(Dense(128, activation='elu'))\ncnn.add(Dropout(0.20))\ncnn.add(BatchNormalization())\n\ncnn.add(Dense(64, activation='elu'))\ncnn.add(Dropout(0.10))\ncnn.add(BatchNormalization())\n\ncnn.add(Dense(1, activation='sigmoid'))\n\ncnn.summary()","metadata":{"execution":{"iopub.status.busy":"2022-03-05T22:22:23.068463Z","iopub.execute_input":"2022-03-05T22:22:23.068709Z","iopub.status.idle":"2022-03-05T22:22:25.576555Z","shell.execute_reply.started":"2022-03-05T22:22:23.068682Z","shell.execute_reply":"2022-03-05T22:22:25.575839Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"%%time\n\nopt = tf.keras.optimizers.Adam(0.01)\ncnn.compile(loss='binary_crossentropy', optimizer=opt, metrics=['accuracy', tf.keras.metrics.AUC()])\n\nh1 = cnn.fit(train_loader, epochs=10, validation_data=valid_loader, verbose=1)","metadata":{"execution":{"iopub.status.busy":"2022-03-05T22:24:10.201206Z","iopub.execute_input":"2022-03-05T22:24:10.201932Z","iopub.status.idle":"2022-03-06T00:37:46.230115Z","shell.execute_reply.started":"2022-03-05T22:24:10.201893Z","shell.execute_reply":"2022-03-06T00:37:46.229354Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"history = merge_history([h1])\nvis_training(history)","metadata":{"execution":{"iopub.status.busy":"2022-03-06T00:37:59.121014Z","iopub.execute_input":"2022-03-06T00:37:59.12127Z","iopub.status.idle":"2022-03-06T00:37:59.636116Z","shell.execute_reply.started":"2022-03-06T00:37:59.121241Z","shell.execute_reply":"2022-03-06T00:37:59.635435Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"cnn.save('MelModel1.h')","metadata":{"execution":{"iopub.status.busy":"2022-03-06T00:40:33.944395Z","iopub.execute_input":"2022-03-06T00:40:33.945113Z","iopub.status.idle":"2022-03-06T00:40:38.326609Z","shell.execute_reply.started":"2022-03-06T00:40:33.945078Z","shell.execute_reply":"2022-03-06T00:40:38.325795Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SPEC_PATH = '../input/g2net-mel-spectrograms-20x9/spectrograms'\n\nclass DataGenerator2(keras.utils.Sequence):\n    \n    def __init__(self, df, batch_size=32, img_size=(20, 9), shuffle=True, is_train=True):\n        self.df = df\n        self.n = len(df)\n        self.batch_size = batch_size\n        self.img_size = img_size\n        self.shuffle = shuffle\n        self.is_train = is_train\n        self.on_epoch_end()\n        \n    def on_epoch_end(self):\n        self.indices = np.arange(self.n)\n        if self.shuffle == True:\n            np.random.shuffle(self.indices)   \n    \n    def __len__(self):     \n        return math.ceil( self.n / self.batch_size )\n    \n    def __getitem__(self, batch_index):\n        start = batch_index * self.batch_size\n        end = (batch_index + 1) * self.batch_size\n        indices = self.indices[start:end]\n        \n        return self.__data_generation(indices)\n    \n    def __data_generation(self, batch_indices):\n        batch_size = len(batch_indices)\n        \n        X = np.zeros(shape=(batch_size, self.img_size[0], self.img_size[1], 3))\n        y = np.zeros(batch_size)\n        \n        for i, idx in enumerate(batch_indices):\n            ID = self.df.id.values[idx]\n            y[i] = self.df.target.values[idx]\n            \n            SET = 'test_spec' if self.is_train else 'train_spec'\n            path = f'{SPEC_PATH}/{SET}/{ID}.npy'\n            data_array = np.load(path)\n            \n            X[i,:,:,:] = data_array\n            \n        return X, y\n    \n\nGENERATOR_TEST = True\n\nif GENERATOR_TEST:\n    temp_gen = DataGenerator(train, batch_size=8, shuffle=False)\n    X,y = temp_gen.__getitem__(0)\n\n    print(X.shape)\n    print(y)","metadata":{"execution":{"iopub.status.busy":"2022-03-06T01:04:37.409359Z","iopub.execute_input":"2022-03-06T01:04:37.409739Z","iopub.status.idle":"2022-03-06T01:04:37.439172Z","shell.execute_reply.started":"2022-03-06T01:04:37.409706Z","shell.execute_reply":"2022-03-06T01:04:37.438478Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_loader = DataGenerator2(test, batch_size=2048, shuffle=False)","metadata":{"execution":{"iopub.status.busy":"2022-03-06T01:04:45.787665Z","iopub.execute_input":"2022-03-06T01:04:45.789418Z","iopub.status.idle":"2022-03-06T01:04:45.793787Z","shell.execute_reply.started":"2022-03-06T01:04:45.789362Z","shell.execute_reply":"2022-03-06T01:04:45.793035Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_prob = cnn.predict(test_loader)","metadata":{"execution":{"iopub.status.busy":"2022-03-06T01:04:48.542728Z","iopub.execute_input":"2022-03-06T01:04:48.543046Z","iopub.status.idle":"2022-03-06T01:30:16.119971Z","shell.execute_reply.started":"2022-03-06T01:04:48.543008Z","shell.execute_reply":"2022-03-06T01:30:16.119185Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(test_prob)","metadata":{"execution":{"iopub.status.busy":"2022-03-06T01:31:18.280167Z","iopub.execute_input":"2022-03-06T01:31:18.280693Z","iopub.status.idle":"2022-03-06T01:31:18.286022Z","shell.execute_reply.started":"2022-03-06T01:31:18.280655Z","shell.execute_reply":"2022-03-06T01:31:18.2852Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission = pd.read_csv('../input/g2net-mel-spectrograms-20x9/spectrograms/sample_submission.csv')\nsubmission.target = test_prob[:,0]\nsubmission","metadata":{"execution":{"iopub.status.busy":"2022-03-06T01:49:19.854944Z","iopub.execute_input":"2022-03-06T01:49:19.855221Z","iopub.status.idle":"2022-03-06T01:49:20.004172Z","shell.execute_reply.started":"2022-03-06T01:49:19.855189Z","shell.execute_reply":"2022-03-06T01:49:20.003409Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"submission.to_csv('submission.csv', header=True, index= False)","metadata":{"execution":{"iopub.status.busy":"2022-03-06T01:49:22.003772Z","iopub.execute_input":"2022-03-06T01:49:22.004021Z","iopub.status.idle":"2022-03-06T01:49:22.55926Z","shell.execute_reply.started":"2022-03-06T01:49:22.003992Z","shell.execute_reply":"2022-03-06T01:49:22.558431Z"},"trusted":true},"execution_count":null,"outputs":[]}]}