{"cells":[{"metadata":{},"cell_type":"markdown","source":"This creates a model that takes up ~1GB memory, then loops through the parquet files to load the data and train the model on it. I am trying to free up memory after each loop, but it still accumulates and I end up exceeding the 13GB limit in the 3rd loop. Could someone provide some feedback on things I could be doing differently to stay under the limit?"},{"metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","trusted":true},"cell_type":"code","source":"import numpy as np \nimport pandas as pd \nimport tensorflow as tf\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\nimport math\nimport gc\nimport cv2","execution_count":null,"outputs":[]},{"metadata":{"_uuid":"d629ff2d2480ee46fbb7e2d37f6b5fab8052498a","_cell_guid":"79c7e3d0-c299-4dcb-8224-4455121ee9b0","trusted":true},"cell_type":"code","source":"#Credit to iafoss for the preprocessing code\nHEIGHT = 137\nWIDTH = 236\nSIZE = 128\n\ndef bbox(img):\n    rows = np.any(img, axis=1)\n    cols = np.any(img, axis=0)\n    rmin, rmax = np.where(rows)[0][[0, -1]]\n    cmin, cmax = np.where(cols)[0][[0, -1]]\n    return rmin, rmax, cmin, cmax\n\ndef crop_resize(img0, size=SIZE, pad=16):\n    ymin,ymax,xmin,xmax = bbox(img0[5:-5,5:-5] > 80)\n    xmin = xmin - 13 if (xmin > 13) else 0\n    ymin = ymin - 10 if (ymin > 10) else 0\n    xmax = xmax + 13 if (xmax < WIDTH - 13) else WIDTH\n    ymax = ymax + 10 if (ymax < HEIGHT - 10) else HEIGHT\n    img = img0[ymin:ymax,xmin:xmax]\n    img[img < 28] = 0\n    lx, ly = xmax-xmin,ymax-ymin\n    l = max(lx,ly) + pad\n    img = np.pad(img, [((l-ly)//2,), ((l-lx)//2,)], mode='constant')\n    return cv2.resize(img,(size,size))","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"densenodes = 1000\nconvchannels = 50\nkernsize = 3\npoolsize = 6\nbatchsize = 100\n\nmodel1 = tf.keras.Sequential([\ntf.keras.layers.Conv2D(convchannels, (kernsize,kernsize), padding='valid', activation=tf.nn.relu, input_shape=(128,128,1)),\ntf.keras.layers.MaxPooling2D((poolsize, poolsize), strides=2),\ntf.keras.layers.Flatten(),\ntf.keras.layers.Dense(densenodes, activation=tf.nn.relu),\ntf.keras.layers.Dense(168,  activation=tf.nn.softmax)\n])\n\nmodel1.compile(optimizer=eval('tf.keras.optimizers.RMSprop(lr=0.001)'), loss='sparse_categorical_crossentropy', metrics=['accuracy'])","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"for i in range(4):\n    df = pd.read_parquet('/kaggle/input/bengaliai-cv19/train_image_data_' + str(i) + '.parquet')\n    trainimgs = df.iloc[:,0]\n    data = 255 - df.iloc[:, 1:].values.reshape(-1, HEIGHT, WIDTH).astype(np.uint8)\n    resized = []\n    for idx in range(len(df)):\n        img = (data[idx]*(255.0/data[idx].max())).astype(np.uint8)\n        resized.append(crop_resize(img))\n    del df\n    del data\n    gc.collect()\n    train = np.array(resized).reshape(len(resized),128,128,1)\n    del resized\n    gc.collect()\n    trainlbl = pd.read_csv('/kaggle/input/bengaliai-cv19/train.csv').merge(trainimgs).iloc[:,1]\n    train_data_gen = ImageDataGenerator(rescale=1./255).flow(train,trainlbl)\n    del train\n    gc.collect()\n    num_train_examples = len(trainimgs)\n    model1.fit_generator(train_data_gen, epochs=10, steps_per_epoch=math.ceil(num_train_examples/batchsize), verbose=1)\n    del train_data_gen\n    gc.collect()","execution_count":null,"outputs":[]},{"metadata":{"trusted":true},"cell_type":"code","source":"","execution_count":null,"outputs":[]}],"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat":4,"nbformat_minor":4}