{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# ****Thoughts on why my model isn't learning?**** \n\nThis is my first Kaggle competition, I've only been coding for 9 months, and nearly everything I've learned has been from youtube videos. There are definitly some issues with the code below, and I would greatly apprecaite feedback, but the most important issue I'm having is that my model won't learn. Other tensorflow models I've built have been able to learn, and my model is almost identical to the model presented in the tutorial. \n\nPlease help!","metadata":{}},{"cell_type":"code","source":"import os\nimport gc\nimport glob\nimport random\nimport numpy as np\nimport pandas as pd\nfrom tqdm import tqdm\nimport tensorflow as tf\nimport PIL.Image as Image\n# from sklearn import metrics\nimport matplotlib.pyplot as plt\nimport matplotlib.patches as patches\nfrom keras import layers, models, optimizers\n","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","scrolled":true,"execution":{"iopub.status.busy":"2023-04-18T18:10:20.559344Z","iopub.execute_input":"2023-04-18T18:10:20.559625Z","iopub.status.idle":"2023-04-18T18:10:28.352626Z","shell.execute_reply.started":"2023-04-18T18:10:20.559597Z","shell.execute_reply":"2023-04-18T18:10:28.351529Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"PREFIX = '/kaggle/input/vesuvius-challenge-ink-detection/train/1/'\nplt.imshow(Image.open(PREFIX+\"ir.png\"), cmap=\"gray\")\n\n# These are the layers I'm adding to my test set.\n# image_selections = [1,5,10,15,20,25,28,31,34,37,40,45,50,55,60,65]\nimage_selections = [25,26,27,28,29,30,31,32,33,34,35,36,37,38,39,40]\nimg_layers = sorted(glob.glob(PREFIX+\"surface_volume/*.tif\"))\n\nbuffer = 5 # This is +/- on each side of a center pixel.","metadata":{"execution":{"iopub.status.busy":"2023-04-18T18:10:33.269676Z","iopub.execute_input":"2023-04-18T18:10:33.270417Z","iopub.status.idle":"2023-04-18T18:10:37.308315Z","shell.execute_reply.started":"2023-04-18T18:10:33.270375Z","shell.execute_reply":"2023-04-18T18:10:37.307355Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"We'll start with the model because it's almost identical (as far as I can tell) to the Vesuvius Challenge: Ink Detection Tutorial, which can be found here: https://www.kaggle.com/code/jpposma/vesuvius-challenge-ink-detection-tutorial\n\nThe example seems to learn quite well, but my model does not. Both models have a 3x3x3 kernel, 'same' padding, three 3D convolutional layers followed by 2x2 maxpooling layers, and a 128 neuron dense layer. I don't believe this is where the issue lies, but would appreciate a sanity check. ","metadata":{}},{"cell_type":"code","source":"# Initializing a 3D convolutional TF model:\nmodel = models.Sequential()\nmodel.add(layers.Conv3D(16,(3,3,3), padding='same', activation='relu', \n                        input_shape=(len(image_selections)+1, 2*buffer+1, 2*buffer+1, 1)))\nmodel.add(layers.MaxPool3D(pool_size=2, padding='same'))\nmodel.add(layers.Conv3D(32,(3,3,3), padding='same', activation='relu'))\nmodel.add(layers.MaxPool3D(pool_size=2, padding='same'))         \nmodel.add(layers.Conv3D(64,(3,3,3), padding='same', activation='relu'))\nmodel.add(layers.MaxPool3D(pool_size=2, padding='same'))\nmodel.add(layers.Flatten())\nmodel.add(layers.Dense(128, activation='relu'))\nmodel.add(layers.Dense(1, activation='sigmoid'))\n\nmodel.compile(optimizer=tf.keras.optimizers.Adam(lr=.003),\n              loss=tf.keras.losses.BinaryCrossentropy(),\n              metrics=tf.keras.metrics.BinaryAccuracy())","metadata":{"execution":{"iopub.status.busy":"2023-04-18T18:10:37.310076Z","iopub.execute_input":"2023-04-18T18:10:37.311534Z","iopub.status.idle":"2023-04-18T18:10:39.712638Z","shell.execute_reply.started":"2023-04-18T18:10:37.311491Z","shell.execute_reply":"2023-04-18T18:10:39.711651Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"In this section I'm setting up my data, and I'm doing this in a very different way than the tutorial because I had trouble following the example code and couldn't make what I thought were interesting changes to it. \n\nThe tutorial code isolated a box within the JPEG for validation, and then sampled around it. I'm choosing as many random points on the image that 30 gigs of RAM will hold to build a dataset (if my frame is 5x5 I can hold 1MM examples), and then attempting to predict the rest of the picture with the model. ","metadata":{}},{"cell_type":"code","source":"# This is called the \"imbalanced data generator\" because I also had a version that picked x samples\n# with a 0 label and x with a 1. Neither set learned well, but this is simpler.\ndef imbalanced_data_generator():\n    \n    mask = np.array(Image.open(PREFIX+\"mask.png\"))\n    label = np.array(Image.open(PREFIX+\"inklabels.png\"))\n    \n    # First we're getting a randomized list of pixels.        \n    length, width = label.shape\n    print(\"Selecting random pixels\")\n    pixel_list = []\n    sample = 0\n    while sample < 7500000/((2*buffer+1)**2): # This **should** max samples w/o exceeding RAM.\n        # doing the \"1 + \" and the \"-2\" from length ensures we don't pick a border spot. \n        pixel_y = buffer + random.randrange(1,length-(2*buffer))\n        pixel_x = buffer + random.randrange(1,width-(2*buffer))\n        \n        for frame_height in range(pixel_y - buffer, pixel_y + buffer + 1):\n            for frame_width in range(pixel_x - buffer, pixel_x + buffer + 1):\n                pixel_list.append([sample, frame_height, frame_width])\n                                            \n        sample += 1\n    \n    # This adds the label and mask data to the list of random pixels.\n    for pixel in pixel_list:\n        pixel.append(label[pixel[1]][pixel[2]].item())\n        pixel.append(mask[pixel[1]][pixel[2]])\n\n    del mask, label\n    \n    # This adds the image layer data to the list of random pixels.\n    print(\"Adding image layers\")\n    for i in tqdm(image_selections):\n        image = [np.array(Image.open(img_layers[i-1]), dtype=np.float32)/65535.0][0]\n        for line in pixel_list:\n            line.append(image[line[1]][line[2]])       \n    \n    del image\n    \n    pixel_list = np.array(pixel_list)\n    \n    # This is isolating the labels, which is the center pixel in the grid.\n    all_y = []\n    i = (2*buffer+1)*buffer+buffer # This is the mid-point in a b*b array, starting from pos 0\n    while i < len(pixel_list):\n        all_y.append(np.array(pixel_list[i][3]))\n        i += (2 * buffer + 1) ** 2\n    all_y = np.array(all_y)\n\n    # This is creating a set of 3D inputs to train on. \n    i = 0\n    all_x = []\n    while i <len(pixel_list):\n        all_x.append(pixel_list[i:i+((2*buffer+1)**2),4:].reshape(17, buffer*2+1, buffer*2+1, 1))\n        i += (2 * buffer + 1) ** 2\n    all_x =np.array(all_x)\n    \n    return all_x, all_y\n\n\nall_x, all_y = imbalanced_data_generator()","metadata":{"execution":{"iopub.status.busy":"2023-04-18T18:10:39.716518Z","iopub.execute_input":"2023-04-18T18:10:39.716805Z","iopub.status.idle":"2023-04-18T18:12:30.579185Z","shell.execute_reply.started":"2023-04-18T18:10:39.716778Z","shell.execute_reply":"2023-04-18T18:12:30.578128Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"The model is run below. The accuracy looks good initially, but in reality the model is just predicting that every pixel is a zero. Also, the model isn't learning as it runs. ","metadata":{}},{"cell_type":"code","source":"model.build(input_shape=(len(image_selections)+1, 2*buffer+1, 2*buffer+1, 1))\nmodel.fit(all_x, all_y, epochs=10)\n# model.evaluate(valid_xb, valid_y)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T18:12:30.581414Z","iopub.execute_input":"2023-04-18T18:12:30.581797Z","iopub.status.idle":"2023-04-18T18:14:54.565729Z","shell.execute_reply.started":"2023-04-18T18:12:30.581738Z","shell.execute_reply":"2023-04-18T18:14:54.564527Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for i in range(100):\n    del all_x, all_y\n    all_x, all_y = imbalanced_data_generator()\n    model.fit(all_x, all_y, epochs=10)","metadata":{"execution":{"iopub.status.busy":"2023-04-18T18:14:54.775883Z","iopub.execute_input":"2023-04-18T18:14:54.776516Z","iopub.status.idle":"2023-04-18T21:19:52.022307Z","shell.execute_reply.started":"2023-04-18T18:14:54.776479Z","shell.execute_reply":"2023-04-18T21:19:52.015797Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"model.save('tutorial.model')","metadata":{"execution":{"iopub.status.busy":"2023-04-18T21:20:01.688097Z","iopub.execute_input":"2023-04-18T21:20:01.689246Z","iopub.status.idle":"2023-04-18T21:20:03.101800Z","shell.execute_reply.started":"2023-04-18T21:20:01.689190Z","shell.execute_reply":"2023-04-18T21:20:03.100796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"From here down, I'm tyring to have the model make predictions on the full image. My original code only looked at one pixel at a time and I was able to produce a nice image, but now that the voxel is larger I'm struggling to get an image to appear. That part I can figure out though. My larger concern is getting the model to learn. \n\nCan anyone help me understand what I may be doing wrong?\n_____________________________________________________________________________________________________","metadata":{}},{"cell_type":"code","source":"model = tf.keras.models.load_model('/kaggle/working/tutorial.model')","metadata":{"execution":{"iopub.status.busy":"2023-04-17T18:40:39.892794Z","iopub.execute_input":"2023-04-17T18:40:39.896632Z","iopub.status.idle":"2023-04-17T18:40:43.697762Z","shell.execute_reply.started":"2023-04-17T18:40:39.896581Z","shell.execute_reply":"2023-04-17T18:40:43.696688Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Now we're trying to load test data to see if we can make a full prediction.\n# This has to be done in batches because we'll run out of RAM real quick.\n\nTEST_PREFIX = '/kaggle/input/vesuvius-challenge-ink-detection/train/3/'\nmask = np.array(Image.open(TEST_PREFIX+\"mask.png\"))\nprint(mask.shape)\n\nimage_selections = [1,5,10,15,20,25,28,31,34,37,40,45,50,55,60,65]\nimg_layers = sorted(glob.glob(TEST_PREFIX+\"surface_volume/*.tif\"))\n\npixel_column_batch = 200\nstart_column = 0\n\nbuffer = 2 # Should match buffer above!","metadata":{"execution":{"iopub.status.busy":"2023-04-17T18:40:43.701918Z","iopub.execute_input":"2023-04-17T18:40:43.702245Z","iopub.status.idle":"2023-04-17T18:40:43.858875Z","shell.execute_reply.started":"2023-04-17T18:40:43.702215Z","shell.execute_reply":"2023-04-17T18:40:43.857512Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This will remove the predictions file if you want to make a fresh one.\n# os.remove(\"/kaggle/working/predictions.csv\")","metadata":{"execution":{"iopub.status.busy":"2023-04-17T18:43:27.045515Z","iopub.execute_input":"2023-04-17T18:43:27.046168Z","iopub.status.idle":"2023-04-17T18:43:27.056490Z","shell.execute_reply.started":"2023-04-17T18:43:27.046129Z","shell.execute_reply":"2023-04-17T18:43:27.055465Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"while start_column < mask.shape[1]:\n    stop_column = min(start_column + pixel_column_batch, mask.shape[1])\n       \n    # This creates the batch mask\n    batch_mask = np.pad(mask[:,start_column:stop_column], buffer)\n    \n    print(\"Adding layers to mask\")\n    # This adds the image layer data to the batch mask\n    for i in image_selections:\n        image = np.pad([np.array(Image.open(img_layers[1]), dtype=np.float32)/65535.0][0][:,start_column:stop_column], buffer)\n        batch_mask = np.dstack((batch_mask, image))\n    \n    print(\"Building model input\")\n    reshaped_input = []\n    for column in range(buffer, stop_column - start_column + buffer):\n        for row in range(buffer, mask.shape[0]):\n            model_input = []\n            for frame_height in range(row - buffer, row + buffer + 1):\n                for frame_width in range(column - buffer, column + buffer + 1):\n                    model_input.append(batch_mask[row][column])\n            reshaped_input.append(np.array(model_input).reshape(17, buffer*2+1, buffer*2+1, 1))\n    \n    del batch_mask\n    reshaped_input = np.array(reshaped_input)\n    \n    batch_preds = model.predict(reshaped_input)\n\n    del reshaped_input    \n    \n    print(\"Saving predictions\")\n    df = pd.DataFrame(batch_preds)\n    df.to_csv('predictions.csv', mode='a', index=False, header=False, sep=';')\n    \n    del batch_preds, df\n    gc.collect()\n    \n    start_column = stop_column\n    print(\"Completed \" + str(stop_column) + \" out of \"+ str(mask.shape[1]) +\" columns\")","metadata":{"execution":{"iopub.status.busy":"2023-04-17T18:43:35.524020Z","iopub.execute_input":"2023-04-17T18:43:35.524802Z","iopub.status.idle":"2023-04-17T19:59:12.626458Z","shell.execute_reply.started":"2023-04-17T18:43:35.524760Z","shell.execute_reply":"2023-04-17T19:59:12.625362Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This is just a reminder of what the mask and labels look like:\nTEST_PREFIX = '/kaggle/input/vesuvius-challenge-ink-detection/train/3/'\nmask = np.array(Image.open(TEST_PREFIX+\"mask.png\").convert('1'))\nlabel = np.array(Image.open(TEST_PREFIX+\"inklabels.png\"))\nfig, (ax1, ax2) = plt.subplots(1, 2)\nax1.set_title(\"mask.png\")\nax1.imshow(mask, cmap='gray')\nax2.set_title(\"inklabels.png\")\nax2.imshow(label, cmap='gray')\nplt.show()","metadata":{"execution":{"iopub.status.busy":"2023-04-17T20:14:39.624842Z","iopub.execute_input":"2023-04-17T20:14:39.625491Z","iopub.status.idle":"2023-04-17T20:14:42.758319Z","shell.execute_reply.started":"2023-04-17T20:14:39.625439Z","shell.execute_reply":"2023-04-17T20:14:42.756976Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_preds = pd.read_csv(\"/kaggle/working/predictions.csv\", sep=\";\")\nfinal_preds = np.array(final_preds)\nprint(final_preds.shape)\n","metadata":{"execution":{"iopub.status.busy":"2023-04-17T20:16:00.746305Z","iopub.execute_input":"2023-04-17T20:16:00.746732Z","iopub.status.idle":"2023-04-17T20:16:01.654275Z","shell.execute_reply.started":"2023-04-17T20:16:00.746693Z","shell.execute_reply":"2023-04-17T20:16:01.653027Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_preds_np = []\nfor pred in final_preds:\n    final_preds_np.append(pred)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T20:16:10.920305Z","iopub.execute_input":"2023-04-17T20:16:10.921628Z","iopub.status.idle":"2023-04-17T20:16:14.415178Z","shell.execute_reply.started":"2023-04-17T20:16:10.921574Z","shell.execute_reply":"2023-04-17T20:16:14.414029Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"final_preds_np = np.array(final_preds_np)\nprint(final_preds_np.shape)","metadata":{"execution":{"iopub.status.busy":"2023-04-17T20:16:26.025006Z","iopub.execute_input":"2023-04-17T20:16:26.025470Z","iopub.status.idle":"2023-04-17T20:16:30.450648Z","shell.execute_reply.started":"2023-04-17T20:16:26.025427Z","shell.execute_reply":"2023-04-17T20:16:30.449378Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"length, width = label.shape\n# This creates a picture out of the prediction.\n# final_preds_np = np.array(final_preds).flatten()\nwidth = round(len(final_preds)/length)\nTHRESHOLD = .05 #This controls the 'exposure' of the image.\npixels = np.where(final_preds_np > THRESHOLD, 1, 0).astype(np.uint8)\nreconstituted_picture = []\n# This makes a column\nstart_pixel = 0\nfor j in range(width):\n    new_column = []\n    for i in range(length):\n        new_column.append(pixels[i+start_pixel])\n    reconstituted_picture.append(new_column)\n    start_pixel += length\n\nreconstituted_picture = np.array(reconstituted_picture).reshape(length, width)\n\nplt.imshow(reconstituted_picture.reshape(length, width), cmap=\"gray\")","metadata":{"execution":{"iopub.status.busy":"2023-04-17T20:17:34.876573Z","iopub.execute_input":"2023-04-17T20:17:34.877677Z","iopub.status.idle":"2023-04-17T20:17:44.644143Z","shell.execute_reply.started":"2023-04-17T20:17:34.877620Z","shell.execute_reply":"2023-04-17T20:17:44.642813Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"print(reconstituted_picture.shape)\nprint(reconstituted_picture[5500][300])\nprint(final_preds_np.max())","metadata":{"execution":{"iopub.status.busy":"2023-04-17T20:19:29.277501Z","iopub.execute_input":"2023-04-17T20:19:29.278632Z","iopub.status.idle":"2023-04-17T20:19:29.292881Z","shell.execute_reply.started":"2023-04-17T20:19:29.278574Z","shell.execute_reply":"2023-04-17T20:19:29.291653Z"},"trusted":true},"execution_count":null,"outputs":[]}]}