{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import numpy as np\nimport pandas as pd # data processing, CSV file I/O (e.g. pd.read_csv)\nimport os\nimport matplotlib.pyplot as plt\nimport matplotlib.image as mpimg\nimport pickle\nimport seaborn as sns","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:28.811050Z","iopub.execute_input":"2021-10-27T08:44:28.811359Z","iopub.status.idle":"2021-10-27T08:44:29.840586Z","shell.execute_reply.started":"2021-10-27T08:44:28.811327Z","shell.execute_reply":"2021-10-27T08:44:29.839579Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Load Train and Test CSVs which contain path to images and their labels - fire or no fire.","metadata":{}},{"cell_type":"code","source":"train_df = pd.read_csv('/kaggle/input/nnfl-2021-assignment-1/fire_videos/train.csv')\ntest_df = pd.read_csv('/kaggle/input/nnfl-2021-assignment-1/fire_videos/test.csv')","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:29.843002Z","iopub.execute_input":"2021-10-27T08:44:29.843356Z","iopub.status.idle":"2021-10-27T08:44:29.961348Z","shell.execute_reply.started":"2021-10-27T08:44:29.843309Z","shell.execute_reply":"2021-10-27T08:44:29.959882Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:29.963076Z","iopub.execute_input":"2021-10-27T08:44:29.963372Z","iopub.status.idle":"2021-10-27T08:44:29.987839Z","shell.execute_reply.started":"2021-10-27T08:44:29.963340Z","shell.execute_reply":"2021-10-27T08:44:29.986853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_df['True_Label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:29.990029Z","iopub.execute_input":"2021-10-27T08:44:29.990289Z","iopub.status.idle":"2021-10-27T08:44:30.014905Z","shell.execute_reply.started":"2021-10-27T08:44:29.990259Z","shell.execute_reply":"2021-10-27T08:44:30.013575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:30.016353Z","iopub.execute_input":"2021-10-27T08:44:30.017085Z","iopub.status.idle":"2021-10-27T08:44:30.030598Z","shell.execute_reply.started":"2021-10-27T08:44:30.017039Z","shell.execute_reply":"2021-10-27T08:44:30.029462Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Clearly, we have a data-imbalance and needs to be handled appropriately to avoid bias towards a particular class.","metadata":{}},{"cell_type":"code","source":"def stratified_sample_df(df, col, n_samples):\n    n = min(n_samples, df[col].value_counts().min())\n    df_ = df.groupby(col).apply(lambda x: x.sample(n))\n    df_.index = df_.index.droplevel(0)\n    return df_","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:30.032211Z","iopub.execute_input":"2021-10-27T08:44:30.032532Z","iopub.status.idle":"2021-10-27T08:44:30.041288Z","shell.execute_reply.started":"2021-10-27T08:44:30.032485Z","shell.execute_reply":"2021-10-27T08:44:30.039873Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def lambda_1_sample_df(df, _lambda=2):\n    dfa = df[df['True_Label'] == 'fire']\n    dfb = df[df['True_Label'] == 'not_fire'].sample(int(len(dfa)*_lambda))\n    bigdata = dfa.append(dfb, ignore_index=True)\n    return bigdata\n\ntrain_df = lambda_1_sample_df(train_df, 2)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:30.042993Z","iopub.execute_input":"2021-10-27T08:44:30.043264Z","iopub.status.idle":"2021-10-27T08:44:30.095509Z","shell.execute_reply.started":"2021-10-27T08:44:30.043226Z","shell.execute_reply":"2021-10-27T08:44:30.094750Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df_ = pd.DataFrame()\n# train_df_ = train_df.sample(train_df['True_Label'].value_counts().min()*3)\n# train_df_['True_Label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:30.096999Z","iopub.execute_input":"2021-10-27T08:44:30.097520Z","iopub.status.idle":"2021-10-27T08:44:30.102102Z","shell.execute_reply.started":"2021-10-27T08:44:30.097476Z","shell.execute_reply":"2021-10-27T08:44:30.100901Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df = train_df_","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:30.103687Z","iopub.execute_input":"2021-10-27T08:44:30.104131Z","iopub.status.idle":"2021-10-27T08:44:30.116288Z","shell.execute_reply.started":"2021-10-27T08:44:30.104089Z","shell.execute_reply":"2021-10-27T08:44:30.115701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# train_df = stratified_sample_df(train_df, 'True_Label', 30000)\ntrain_df['True_Label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:30.118454Z","iopub.execute_input":"2021-10-27T08:44:30.118691Z","iopub.status.idle":"2021-10-27T08:44:30.140591Z","shell.execute_reply.started":"2021-10-27T08:44:30.118665Z","shell.execute_reply":"2021-10-27T08:44:30.139559Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.countplot(x='True_Label', data=train_df)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:30.141734Z","iopub.execute_input":"2021-10-27T08:44:30.142116Z","iopub.status.idle":"2021-10-27T08:44:30.465034Z","shell.execute_reply.started":"2021-10-27T08:44:30.142075Z","shell.execute_reply":"2021-10-27T08:44:30.463922Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def display_image(num_images=10, df=train_df, phase='train'):\n    plt.figure(figsize=(25, 16))\n    for i in range(1, num_images+1):\n        plt.subplot(num_images // 5, 5, i)\n        index = np.random.randint(0, high=len(df))\n        path = '/kaggle/input/nnfl-2021-assignment-1/fire_videos/' + phase + '/' + df['File'].iloc[index]\n        img = mpimg.imread(path)\n        img = img.mean(-1)\n        imgplot = plt.imshow(img, cmap='gray', vmin=0, vmax=255)\n        if phase == 'train':\n            plt.title(df['True_Label'].iloc[index])\n    plt.show()\n\ndisplay_image(df=train_df[train_df['True_Label'] == 'fire'])\ndisplay_image(df=train_df[train_df['True_Label'] == 'not_fire'])","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:30.466270Z","iopub.execute_input":"2021-10-27T08:44:30.466529Z","iopub.status.idle":"2021-10-27T08:44:34.965795Z","shell.execute_reply.started":"2021-10-27T08:44:30.466502Z","shell.execute_reply":"2021-10-27T08:44:34.964701Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import tensorflow as tf\nfrom tensorflow import keras\nimport cv2\nfrom tqdm.notebook import tqdm","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:34.968302Z","iopub.execute_input":"2021-10-27T08:44:34.969460Z","iopub.status.idle":"2021-10-27T08:44:40.933302Z","shell.execute_reply.started":"2021-10-27T08:44:34.969363Z","shell.execute_reply":"2021-10-27T08:44:40.932599Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMG_HEIGHT, IMG_WIDTH = (28, 28)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:40.934312Z","iopub.execute_input":"2021-10-27T08:44:40.934919Z","iopub.status.idle":"2021-10-27T08:44:40.940226Z","shell.execute_reply.started":"2021-10-27T08:44:40.934877Z","shell.execute_reply":"2021-10-27T08:44:40.939198Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"LABEL_MAPPING = {\n    'fire': 1,\n    'not_fire': 0\n}","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:40.942284Z","iopub.execute_input":"2021-10-27T08:44:40.942574Z","iopub.status.idle":"2021-10-27T08:44:40.953765Z","shell.execute_reply.started":"2021-10-27T08:44:40.942538Z","shell.execute_reply":"2021-10-27T08:44:40.952936Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def get_random_crop(image, crop_height=IMG_HEIGHT, crop_width=IMG_WIDTH):\n    max_x = image.shape[1] - crop_width\n    max_y = image.shape[0] - crop_height\n    x = np.random.randint(0, max_x)\n    y = np.random.randint(0, max_y)\n    crop = image[y: y + crop_height, x: x + crop_width]\n    return crop\n\ndef display_crops(num_images=4, num_crops=5, df=train_df[train_df['True_Label']==\"not_fire\"], phase='train'):\n    plt.figure(figsize=(25, 16))\n    for i in range(1, num_images+1):\n        index = np.random.randint(0, high=len(df))\n        path = '/kaggle/input/nnfl-2021-assignment-1/fire_videos/' + phase + '/' + df['File'].iloc[index]\n        img = mpimg.imread(path)\n        for j in range(1, num_crops+1):\n            random_crop = get_random_crop(img)\n            plt.subplot(num_images, num_images * num_crops, (i-1) * num_crops + j)\n            imgplot = plt.imshow(random_crop)\n            if phase == 'train':\n                plt.title(df['True_Label'].iloc[index])\n    plt.show()\n\ndisplay_crops()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:40.955487Z","iopub.execute_input":"2021-10-27T08:44:40.956441Z","iopub.status.idle":"2021-10-27T08:44:43.114629Z","shell.execute_reply.started":"2021-10-27T08:44:40.956379Z","shell.execute_reply":"2021-10-27T08:44:43.113485Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_not_fire_dataset(df, num_samples=18215, each_sample_crops = 3):\n    new_dataset = []\n    for _ in range(num_samples):\n        index = np.random.randint(0, high=len(df))\n        path = '/kaggle/input/nnfl-2021-assignment-1/fire_videos/' + 'train' + '/' + df['File'].iloc[index]\n        img = mpimg.imread(path)\n        random_crops = np.hstack([[get_random_crop(img)] * each_sample_crops])\n        new_dataset = np.hstack([new_dataset, random_crops])\n    return new_dataset\n\n# create_not_fire_dataset(train_df[train_df['True_Label']==\"not_fire\"])","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:44:43.116059Z","iopub.execute_input":"2021-10-27T08:44:43.116311Z","iopub.status.idle":"2021-10-27T08:44:43.123276Z","shell.execute_reply.started":"2021-10-27T08:44:43.116283Z","shell.execute_reply":"2021-10-27T08:44:43.122590Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_dataset(df, phase):\n    img_data_array=[]\n    class_name=[]\n    img_folder = '/kaggle/input/nnfl-2021-assignment-1/fire_videos/'\n    for index, row in tqdm(df.iterrows(), total=df.shape[0]):\n        image_path = os.path.join(img_folder, phase, row['File'])\n        image = cv2.imread(image_path, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (IMG_HEIGHT, IMG_WIDTH), interpolation = cv2.INTER_AREA)\n        image = np.array(image)\n        image = image.astype('float32')\n        image /= 255\n        img_data_array.append(image)\n        if phase == 'train':\n            class_name.append(LABEL_MAPPING[row['True_Label']])\n    return img_data_array, class_name","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:49:17.172976Z","iopub.execute_input":"2021-10-27T08:49:17.173301Z","iopub.status.idle":"2021-10-27T08:49:17.182184Z","shell.execute_reply.started":"2021-10-27T08:49:17.173270Z","shell.execute_reply":"2021-10-27T08:49:17.181008Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"HIGH_RES_IMG_HEIGHT, HIGH_RES_IMG_WIDTH = (256, 256)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:45:35.033650Z","iopub.execute_input":"2021-10-27T08:45:35.034352Z","iopub.status.idle":"2021-10-27T08:45:35.038550Z","shell.execute_reply.started":"2021-10-27T08:45:35.034316Z","shell.execute_reply":"2021-10-27T08:45:35.037465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_high_res_dataset(df, phase):\n    img_data_array=[]\n    class_name=[]\n    img_folder = '/kaggle/input/nnfl-2021-assignment-1/fire_videos/'\n    for index, row in tqdm(df.iterrows(), total=df.shape[0]):\n        image_path = os.path.join(img_folder, phase, row['File'])\n        image = cv2.imread(image_path, cv2.COLOR_BGR2RGB)\n        image = cv2.resize(image, (HIGH_RES_IMG_HEIGHT, HIGH_RES_IMG_WIDTH), interpolation = cv2.INTER_AREA)\n        image = np.array(image)\n        image = image.astype('float32')\n        image /= 255 \n        img_data_array.append(image)\n        if phase == 'train':\n            class_name.append(LABEL_MAPPING[row['True_Label']])\n    return img_data_array, class_name","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:45:35.369834Z","iopub.execute_input":"2021-10-27T08:45:35.370114Z","iopub.status.idle":"2021-10-27T08:45:35.378310Z","shell.execute_reply.started":"2021-10-27T08:45:35.370086Z","shell.execute_reply":"2021-10-27T08:45:35.376906Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"SAVE_AS_PICKLE = False\nif SAVE_AS_PICKLE:\n    X_train, y_train = create_dataset(train_df, 'train')\n    import pickle\n    training_data = {\n        'X': X_train,\n        'y': y_train\n    }\n    with open('2_1_class_high_res_train_X_y.pickle', 'wb') as handle:\n        pickle.dump(training_data, handle, protocol=pickle.HIGHEST_PROTOCOL)\nelse:\n    with open('/kaggle/input/nnfl-assignment-train-pickle/equal_class_train_X_y.pickle', 'rb') as handle:\n#     with open('./equal_class_high_res_train_X_y.pickle', 'rb') as handle:\n        training_data = pickle.load(handle)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:45:37.760087Z","iopub.execute_input":"2021-10-27T08:45:37.761086Z","iopub.status.idle":"2021-10-27T08:45:41.196772Z","shell.execute_reply.started":"2021-10-27T08:45:37.761037Z","shell.execute_reply":"2021-10-27T08:45:41.195516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras import layers\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator, img_to_array, load_img","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:45:45.702192Z","iopub.execute_input":"2021-10-27T08:45:45.702532Z","iopub.status.idle":"2021-10-27T08:45:45.708150Z","shell.execute_reply.started":"2021-10-27T08:45:45.702500Z","shell.execute_reply":"2021-10-27T08:45:45.706980Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"### Data Splitting into train-val set","metadata":{}},{"cell_type":"code","source":"training_data['X'] = np.array(training_data['X'])\ntraining_data['X'].shape","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:45:44.410442Z","iopub.execute_input":"2021-10-27T08:45:44.410740Z","iopub.status.idle":"2021-10-27T08:45:44.609009Z","shell.execute_reply.started":"2021-10-27T08:45:44.410712Z","shell.execute_reply":"2021-10-27T08:45:44.607985Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.model_selection import train_test_split\nX_train, X_test, y_train, y_test = train_test_split(np.array(training_data['X']), np.array(training_data['y']), test_size=0.2, random_state=42)\nprint('Training Samples:', len(X_train), '\\nValidation Samples:', len(X_test))\ndel training_data","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:45:56.751257Z","iopub.execute_input":"2021-10-27T08:45:56.752072Z","iopub.status.idle":"2021-10-27T08:45:57.252392Z","shell.execute_reply.started":"2021-10-27T08:45:56.752019Z","shell.execute_reply":"2021-10-27T08:45:57.251255Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from tensorflow.keras.models import Sequential\nfrom tensorflow.keras.layers import Dense, Conv2D, Dropout, Flatten, MaxPooling2D, Input, BatchNormalization\nfrom tensorflow.keras import regularizers","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:45:58.101679Z","iopub.execute_input":"2021-10-27T08:45:58.102030Z","iopub.status.idle":"2021-10-27T08:45:58.107787Z","shell.execute_reply.started":"2021-10-27T08:45:58.101995Z","shell.execute_reply":"2021-10-27T08:45:58.106945Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"METRICS = [\n      keras.metrics.BinaryAccuracy(name='accuracy'),\n      keras.metrics.Precision(name='precision'),\n      keras.metrics.Recall(name='recall'),\n      keras.metrics.AUC(name='auc'),\n      keras.metrics.AUC(name='prc', curve='PR'), # precision-recall curve\n]\n\ndef make_baseline_model(output_bias):\n    model = Sequential()\n    model.add(Conv2D(20, kernel_size=(3,3), input_shape=(IMG_HEIGHT, IMG_WIDTH, 3)))\n    model.add(MaxPooling2D(pool_size=(2, 2)))\n    model.add(Flatten())\n    model.add(Dense(128, activation=tf.nn.relu))\n    model.add(Dropout(0.1))\n    model.add(Dense(1, activation=tf.nn.sigmoid, bias_initializer=output_bias))\n    return model\n\ndef make_model(model_fn, lr=1e-3, metrics=METRICS, output_bias=None):\n    if output_bias is not None:\n        output_bias = tf.keras.initializers.Constant(output_bias)\n    model = model_fn(output_bias)\n    model.compile(\n      optimizer=keras.optimizers.Adam(learning_rate=lr),\n      loss=keras.losses.BinaryCrossentropy(),\n      metrics=metrics)\n    return model","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:53:49.017969Z","iopub.execute_input":"2021-10-27T08:53:49.018280Z","iopub.status.idle":"2021-10-27T08:53:49.061854Z","shell.execute_reply.started":"2021-10-27T08:53:49.018251Z","shell.execute_reply":"2021-10-27T08:53:49.060971Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"baseline_model = make_model(make_baseline_model)\nbaseline_model.summary()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:53:50.563698Z","iopub.execute_input":"2021-10-27T08:53:50.564853Z","iopub.status.idle":"2021-10-27T08:53:50.650900Z","shell.execute_reply.started":"2021-10-27T08:53:50.564808Z","shell.execute_reply":"2021-10-27T08:53:50.649939Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def plot_loss(history, label, n):\n    # Use a log scale on y-axis to show the wide range of values.\n    plt.semilogy(history.epoch, history.history['loss'],\n               color=colors[n], label='Train ' + label)\n    plt.semilogy(history.epoch, history.history['val_loss'],\n               color=colors[n], label='Val ' + label,\n               linestyle=\"--\")\n    plt.xlabel('Epoch')\n    plt.ylabel('Loss')","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:53:56.673337Z","iopub.execute_input":"2021-10-27T08:53:56.673664Z","iopub.status.idle":"2021-10-27T08:53:56.680039Z","shell.execute_reply.started":"2021-10-27T08:53:56.673630Z","shell.execute_reply":"2021-10-27T08:53:56.679094Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"X_train.shape","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:53:57.926336Z","iopub.execute_input":"2021-10-27T08:53:57.927242Z","iopub.status.idle":"2021-10-27T08:53:57.934208Z","shell.execute_reply.started":"2021-10-27T08:53:57.927186Z","shell.execute_reply":"2021-10-27T08:53:57.933190Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"BATCH_SIZE = 2048\nEPOCHS = 15\n\nearly_stopping = keras.callbacks.EarlyStopping(\n    monitor='val_prc', \n    verbose=1,\n    patience=3,\n    mode='max',\n    restore_best_weights=True)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:53:58.082224Z","iopub.execute_input":"2021-10-27T08:53:58.083346Z","iopub.status.idle":"2021-10-27T08:53:58.088786Z","shell.execute_reply.started":"2021-10-27T08:53:58.083277Z","shell.execute_reply":"2021-10-27T08:53:58.087703Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"baseline_history = baseline_model.fit(\n    X_train, y_train,\n    batch_size=BATCH_SIZE,\n    epochs=5,\n    callbacks=[early_stopping],\n    validation_data=(X_test, y_test))","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:56:10.327329Z","iopub.execute_input":"2021-10-27T08:56:10.327676Z","iopub.status.idle":"2021-10-27T08:56:46.418128Z","shell.execute_reply.started":"2021-10-27T08:56:10.327644Z","shell.execute_reply":"2021-10-27T08:56:46.417492Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"colors = ['red', 'green', 'yellow', 'black']\n\ndef plot_metrics(history):\n    metrics = ['loss', 'prc', 'precision', 'recall']\n    plt.figure(figsize=(16, 9))\n    for n, metric in enumerate(metrics):\n        name = metric.replace(\"_\",\" \").capitalize()\n        plt.subplot(2,2,n+1)\n        plt.plot(history.epoch, history.history[metric], color=colors[0], label='Train')\n        plt.plot(history.epoch, history.history['val_'+metric],\n                 color=colors[1], linestyle=\"--\", label='Val')\n        plt.xlabel('Epoch')\n        plt.ylabel(name)\n        if metric == 'loss':\n            plt.ylim([0, plt.ylim()[1]])\n        elif metric == 'auc':\n            plt.ylim([0.8,1])\n        else:\n            plt.ylim([0,1])\n        plt.legend()\n    plt.show()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:55:52.923375Z","iopub.execute_input":"2021-10-27T08:55:52.924357Z","iopub.status.idle":"2021-10-27T08:55:52.934839Z","shell.execute_reply.started":"2021-10-27T08:55:52.924308Z","shell.execute_reply":"2021-10-27T08:55:52.934098Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_metrics(baseline_history)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:55:53.051312Z","iopub.execute_input":"2021-10-27T08:55:53.052382Z","iopub.status.idle":"2021-10-27T08:55:53.703458Z","shell.execute_reply.started":"2021-10-27T08:55:53.052324Z","shell.execute_reply":"2021-10-27T08:55:53.700516Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from sklearn.metrics import confusion_matrix\n\ndef plot_cm(labels, predictions, p=0.5):\n    cm = confusion_matrix(labels, predictions > p)\n    plt.figure(figsize=(5,5))\n    sns.heatmap(cm, annot=True, fmt=\"d\")\n    plt.title('Confusion matrix @{:.2f}'.format(p))\n    plt.ylabel('Actual label')\n    plt.xlabel('Predicted label')","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:56:46.421629Z","iopub.execute_input":"2021-10-27T08:56:46.423298Z","iopub.status.idle":"2021-10-27T08:56:46.430258Z","shell.execute_reply.started":"2021-10-27T08:56:46.423262Z","shell.execute_reply":"2021-10-27T08:56:46.429552Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"train_predictions_baseline = baseline_model.predict(X_train, batch_size=BATCH_SIZE)\nval_predictions_baseline = baseline_model.predict(X_test, batch_size=BATCH_SIZE)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:56:46.431559Z","iopub.execute_input":"2021-10-27T08:56:46.432031Z","iopub.status.idle":"2021-10-27T08:56:49.618172Z","shell.execute_reply.started":"2021-10-27T08:56:46.432001Z","shell.execute_reply":"2021-10-27T08:56:49.617360Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(train_predictions_baseline)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:56:49.620873Z","iopub.execute_input":"2021-10-27T08:56:49.621800Z","iopub.status.idle":"2021-10-27T08:56:49.962575Z","shell.execute_reply.started":"2021-10-27T08:56:49.621738Z","shell.execute_reply":"2021-10-27T08:56:49.961446Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"sns.histplot(val_predictions_baseline)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:56:49.964336Z","iopub.execute_input":"2021-10-27T08:56:49.965275Z","iopub.status.idle":"2021-10-27T08:56:50.268938Z","shell.execute_reply.started":"2021-10-27T08:56:49.965235Z","shell.execute_reply":"2021-10-27T08:56:50.268316Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_cm(y_train, train_predictions_baseline, 0.5)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:56:50.270042Z","iopub.execute_input":"2021-10-27T08:56:50.270795Z","iopub.status.idle":"2021-10-27T08:56:50.662033Z","shell.execute_reply.started":"2021-10-27T08:56:50.270762Z","shell.execute_reply":"2021-10-27T08:56:50.661384Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plot_cm(y_test, val_predictions_baseline, 0.5)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:56:50.663269Z","iopub.execute_input":"2021-10-27T08:56:50.663671Z","iopub.status.idle":"2021-10-27T08:56:50.963389Z","shell.execute_reply.started":"2021-10-27T08:56:50.663641Z","shell.execute_reply":"2021-10-27T08:56:50.962745Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_dataset, _ = create_dataset(test_df, 'test')\ntest_dataset = np.array(test_dataset)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:56:50.964630Z","iopub.execute_input":"2021-10-27T08:56:50.965028Z","iopub.status.idle":"2021-10-27T08:57:21.903680Z","shell.execute_reply.started":"2021-10-27T08:56:50.964997Z","shell.execute_reply":"2021-10-27T08:57:21.902534Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_results = baseline_model.predict(test_dataset, batch_size=BATCH_SIZE)\nfinal_results = (final_results > 0.5).astype(int)","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:57:21.906566Z","iopub.execute_input":"2021-10-27T08:57:21.906829Z","iopub.status.idle":"2021-10-27T08:57:22.168798Z","shell.execute_reply.started":"2021-10-27T08:57:21.906799Z","shell.execute_reply":"2021-10-27T08:57:22.167979Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def create_results_df(df, results):\n    rs_df = pd.DataFrame({\n        'File': df['File'],\n        'Label': results.reshape(-1)\n    })\n    rs_df['Label'] = rs_df['Label'].astype(str)\n    rs_df['Label'] = rs_df['Label'].map({'0': 'not_fire', '1': 'fire'})\n    rs_df['Label'].value_counts()\n    rs_df.head()\n    return rs_df","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:57:22.170406Z","iopub.execute_input":"2021-10-27T08:57:22.171467Z","iopub.status.idle":"2021-10-27T08:57:22.178830Z","shell.execute_reply.started":"2021-10-27T08:57:22.171372Z","shell.execute_reply":"2021-10-27T08:57:22.177736Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rs_df = create_results_df(test_df, final_results)\nrs_df.to_csv('out_larger_smaller_kernals.csv', index=False)\nrs_df['Label'].value_counts()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:57:22.190597Z","iopub.execute_input":"2021-10-27T08:57:22.191545Z","iopub.status.idle":"2021-10-27T08:57:22.217300Z","shell.execute_reply.started":"2021-10-27T08:57:22.191505Z","shell.execute_reply":"2021-10-27T08:57:22.216051Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"rs_df.head()","metadata":{"execution":{"iopub.status.busy":"2021-10-27T08:57:24.991103Z","iopub.execute_input":"2021-10-27T08:57:24.991405Z","iopub.status.idle":"2021-10-27T08:57:25.001910Z","shell.execute_reply.started":"2021-10-27T08:57:24.991377Z","shell.execute_reply":"2021-10-27T08:57:25.000838Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}