{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"https://github.com/CarterWoolsey/woodentify","metadata":{}},{"cell_type":"markdown","source":"I was wondering why this exercis is so hard to crack...\nThe problem from black ink on papyrus, is that it once carbonized the black ink being also carbon is not very distinguishable from the carbonize papyrus...","metadata":{}},{"cell_type":"markdown","source":"The exercise, we don't need to recognize the letters, bizar enough\nwe don't need to find the difference between letters and background\nwe need to recognize the text lines..\nHence since papyrus is a kind of woodpattern, \nand we need to recognize the difference between de textlines and the inbetween lines, \nwe have to recognize woodpatterns ?\n\nwhy not..;-)\n","metadata":{}},{"cell_type":"code","source":"import cv2\nfrom matplotlib import pyplot as plt\nimport numpy as np\nfrom os.path import isfile, join\nfrom os import listdir\n\ndef load_image_folder(filepath, limit=None):\n    onlyfiles = [f for f in listdir(filepath) if isfile(join(filepath,f)) and '.tif' in f]\n    print(filepath)\n    print('Num pics in folder: {}'.format(len(onlyfiles)))\n    if limit != None:\n        num_files = limit\n        print('Only {} images being used'.format(num_files))\n    else:\n        num_files = len(onlyfiles)\n        print('All images being used')\n    images = np.empty(num_files, dtype=object)\n    for n in range(0, num_files):\n      images[n] = cv2.resize(cv2.imread(join(filepath,onlyfiles[n])), (1008, 756))\n      images[n] = cv2.cvtColor(images[n], cv2.COLOR_BGR2RGB)\n    return images\n\ndef crop_image(image, size):\n    image_list = []\n\n    #Find height, width of input image\n    height = image.shape[0]\n    width = image.shape[1]\n\n    #Find out how many slices we can do for height and width based on size\n    height_slices = int(height / size)\n    width_slices = int(width / size)\n\n    #Set start and stop parameters for image cropping\n    h_start, h_stop = 0, size\n    w_start, w_stop = 0, size\n\n    #This for loop will\n    for i in range(height_slices):\n        for j in range(width_slices):\n            crop_img = image[h_start:h_stop, w_start:w_stop]\n            image_list.append(crop_img)\n            w_start += size\n            w_stop += size\n        h_start += size\n        h_stop += size\n        w_start = 0\n        w_stop = size\n\n    return image_list\n\ndef crop_image_list(image_list, size):\n    new_image_list = []\n    for i in image_list:\n        cropped_images = crop_image(i, size)\n        for j in cropped_images:\n            new_image_list.append(j)\n    return new_image_list\n\ndef rotate_images_4x(image_list):\n    new_image_list = []\n\n    #Takes the image from the list, converts to NP array, flips it 90 degrees\n    #and saves it for 4 total images\n    for i in image_list:\n        img = np.array(i)\n        new_image_list.append(img)\n        for j in range(3):\n            img = np.rot90(img)\n            new_image_list.append(img)\n    return new_image_list\n\ndef mirror_images(image_list):\n    new_image_list = []\n    for i in image_list:\n        img = np.array(i)\n        new_image_list.append(img)\n        img = np.fliplr(img)\n        new_image_list.append(img)\n    return new_image_list\n\ndef save_images(image_list):\n    for i in range(len(image_list)):\n        cv2.imwrite('test_images/img_{}.jpg'.format(i), image_list[i])\n\ndef prep_pipeline(folder, size):\n    images = load_image_folder(folder)\n    cropped = crop_image_list(images, size)\n    rot_crop = rotate_images_4x(cropped)\n    return mirror_images(rot_crop)\n\ndef prep_total_pipeline(folder_list, size, limit=None):\n    X = 0\n    y = 0\n    for i in range(len(folder_list)):\n        if limit == None:\n            images = load_image_folder(folder_list[i])\n        else:\n            images = load_image_folder(folder_list[i], limit)\n        cropped = crop_image_list(images, size)\n        rot_crop = rotate_images_4x(cropped)\n        mirrored = mirror_images(rot_crop)\n        if type(X) == int:\n            X = np.array(mirrored)\n        else:\n            X = np.vstack((X, np.array(mirrored)))\n        if type(y) == int:\n            y = np.zeros(len(mirrored))\n        else:\n            y_arr = np.zeros(len(mirrored))\n            y_arr.fill(i)\n            y = np.append(y, y_arr)\n        print('X shape: {} -=-=-=-= y shape: {}'.format(X.shape, y.shape))\n    return np.array(X), y\n\n\n\nif True:\n    PATH = \"/kaggle/input/vesuvius-challenge-ink-detection/train/2/ir.png\"\n    image = cv2.imread(PATH)\n\n    #change the colors from BGR to RGB - this is unnecessary because imwrite flips it back\n    # BGRflags = [flag for flag in dir(cv2) if flag.startswith('COLOR_BGR') ]\n    # image = cv2.cvtColor(image, cv2.COLOR_BGR2RGB)\n\n    print(\"Height:\\t\\t%i pixels\\nWidth:\\t\\t%i pixels\\nChannels:\\t%i\" % image.shape)\n    print(\"pixel at (0,0) [B,G,R]:\\t[%i,%i,%i]\" % tuple(image[0,0,:]))\n    print(\"data-type: %s \" % image.dtype)\n\n    #This is how to resize the image\n    resized_image = cv2.resize(image, (100, 50))\n    # plt.imshow(resized_image)\n    # plt.show()\n\n    #Pipeline Test\n    l1 = []\n    l1.append(image)\n    rotated = rotate_images_4x(l1)\n    mirrored_rotated = mirror_images(rotated)","metadata":{"execution":{"iopub.status.busy":"2023-04-06T19:45:40.120841Z","iopub.execute_input":"2023-04-06T19:45:40.121310Z","iopub.status.idle":"2023-04-06T19:45:47.670916Z","shell.execute_reply.started":"2023-04-06T19:45:40.121273Z","shell.execute_reply":"2023-04-06T19:45:47.669784Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"plt.imshow(np.asarray(rotated[0],dtype=('int')) )","metadata":{"execution":{"iopub.status.busy":"2023-04-06T19:45:47.673468Z","iopub.execute_input":"2023-04-06T19:45:47.674010Z","iopub.status.idle":"2023-04-06T19:46:11.057590Z","shell.execute_reply.started":"2023-04-06T19:45:47.673957Z","shell.execute_reply":"2023-04-06T19:46:11.056595Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import cv2\nfrom matplotlib import pyplot as plt\nimport numpy as np\n#from img_preprocess import *\nfrom keras.models import Sequential\nfrom keras.layers import Dense, Dropout, Activation, Flatten\nfrom keras.layers import Convolution2D, MaxPooling2D\nfrom keras.layers.convolutional import Conv2D\nfrom keras.utils import np_utils\nfrom keras import backend as K\nfrom sklearn.model_selection import train_test_split\nnp.random.seed(1337)  # for reproducibility\nimport sys\n\nif True:\n    img_rows, img_cols = 50, 50 # input image dimensions\n    folder_list = ['/kaggle/input/vesuvius-challenge-ink-detection/train/3/surface_volume', '/kaggle/input/vesuvius-challenge-ink-detection/train/2/surface_volume', '/kaggle/input/vesuvius-challenge-ink-detection/train/1/surface_volume']\n    X, y = prep_total_pipeline(folder_list, img_rows, limit=5)\n    X_train, X_test, y_train, y_test = train_test_split(X, y, test_size=0.2, random_state=42)\n\n    batch_size = 100\n    num_classes = len(folder_list)\n    num_epochs = 20\n    nb_filters = 32 # number of convolutional filters to use\n    pool_size = (2, 2) # size of pooling area for max pooling\n                       # decreases image size, and helps to avoid overfitting\n\n    if K.image_data_format() == 'th':\n        X_train = X_train.reshape(X_train.shape[0], X_train.shape[3], img_rows, img_cols)\n        X_test = X_test.reshape(X_test.shape[0], X_test.shape[3], img_rows, img_cols)\n        input_shape = (X_test.shape[3], img_rows, img_cols)\n    else:\n        X_train = X_train.reshape(X_train.shape[0], img_rows, img_cols, X_train.shape[3])\n        X_test = X_test.reshape(X_test.shape[0], img_rows, img_cols, X_test.shape[3])\n        input_shape = (img_rows, img_cols, X_test.shape[3])\n\n    # don't change conversion or normalization\n    X_train = X_train.astype('float32') # data was uint8 [0-255]\n    X_test = X_test.astype('float32')  # data was uint8 [0-255]\n    X_train /= 255 # normalizing (scaling from 0 to 1)\n    X_test /= 255  # normalizing (scaling from 0 to 1)\n    print('X_train shape:', X_train.shape)\n    print(X_train.shape[0], 'train samples')\n    print(X_test.shape[0], 'test samples')\n\n    # convert class vectors to binary class matrices (don't change)\n    Y_train = np_utils.to_categorical(y_train, num_classes) # cool\n    Y_test = np_utils.to_categorical(y_test, num_classes)   # cool * 2\n\n    model = Sequential()\n    model.add(Conv2D(nb_filters, (5, 5),\n                        padding='valid',\n                        input_shape=input_shape))\n    model.add(Activation('relu')) # Activation specification necessary for Conv2D and Dense layers\n    model.add(Conv2D(nb_filters, (3,3)))\n    model.add(Activation('relu'))\n    model.add(Conv2D(nb_filters, (2,2)))\n    model.add(Activation('relu'))\n    model.add(MaxPooling2D(pool_size=pool_size)) # decreases size, helps prevent overfitting\n    model.add(Dropout(0.5)) # zeros out some fraction of inputs, helps prevent overfitting\n\n    model.add(Flatten()) # necessary to flatten before going into conventional dense layer (keep layer)\n    print('Model flattened out to ', model.output_shape)\n\n    # now start a typical neural network\n    model.add(Dense(128))\n    model.add(Activation('tanh'))\n    model.add(Dropout(0.15))\n    model.add(Dense(num_classes)) # 10 final nodes (one for each class) (keep layer)\n    model.add(Activation('softmax')) # keep softmax at end to pick between classes 0-9\n    model.compile(loss='categorical_crossentropy',\n                  optimizer='adam',\n                  metrics=['accuracy'])\n\n    # during fit process watch train and test error simultaneously\n    model.fit(X_train, Y_train, batch_size=batch_size, epochs=num_epochs,\n              verbose=1, validation_data=(X_test, Y_test))\n\n    score = model.evaluate(X_test, Y_test, verbose=0)\n    print('Test score:', score[0])\n    print('Test accuracy:', score[1]) # this is the one we care about","metadata":{"execution":{"iopub.status.busy":"2023-04-06T19:46:11.058862Z","iopub.execute_input":"2023-04-06T19:46:11.059831Z","iopub.status.idle":"2023-04-06T20:36:06.947906Z","shell.execute_reply.started":"2023-04-06T19:46:11.059788Z","shell.execute_reply":"2023-04-06T20:36:06.945558Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"so the maximum accuracy should be 78%-82% IMHO","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}