{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# <h1>Denoising using Autoencoder model</h1>\n\nDataset from : https://www.kaggle.com/c/denoising-dirty-documents","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19"}},{"cell_type":"code","source":"# import libraries\n\nimport numpy as np\nimport cv2\nimport random\nfrom tqdm import tqdm\nfrom sklearn.model_selection import train_test_split\n\nimport tensorflow as tf\nfrom tensorflow.keras import layers, Sequential\nfrom tensorflow.keras.preprocessing.image import ImageDataGenerator\n\nimport os\nimport glob\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:05:43.641426Z","iopub.execute_input":"2022-07-17T02:05:43.641815Z","iopub.status.idle":"2022-07-17T02:05:43.648692Z","shell.execute_reply.started":"2022-07-17T02:05:43.641752Z","shell.execute_reply":"2022-07-17T02:05:43.647604Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <h1> Data Preparation</h1>","metadata":{}},{"cell_type":"code","source":"# create dirs\n\ntrain_input_path = '/kaggle/working/train/input_images/'\ntrain_target_path = '/kaggle/working/train/target_images/'\n\ntest_input_path = '/kaggle/working/test/input_images/'\ntest_target_path = '/kaggle/working/test/target_images/'\n\n\nos.makedirs(train_input_path+\"train/\")\nos.makedirs(train_target_path+\"train_cleaned/\")\n\nos.makedirs(test_input_path)\nos.makedirs(test_target_path)\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:05:43.660282Z","iopub.execute_input":"2022-07-17T02:05:43.660963Z","iopub.status.idle":"2022-07-17T02:05:43.669413Z","shell.execute_reply.started":"2022-07-17T02:05:43.660894Z","shell.execute_reply":"2022-07-17T02:05:43.668448Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Add new data\n\nos.makedirs('/kaggle/working/denoising-dirty-documents/')\n\n!unzip /kaggle/input/denoising-dirty-documents/train_cleaned.zip -d /kaggle/working/denoising-dirty-documents/\n!unzip /kaggle/input/denoising-dirty-documents/train.zip -d /kaggle/working/denoising-dirty-documents/\n\n!cp -ar /kaggle/working/denoising-dirty-documents/train/* /kaggle/working/train/input_images/train\n!cp -ar /kaggle/working/denoising-dirty-documents/train_cleaned/* /kaggle/working/train/target_images/train_cleaned\n\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:05:43.683877Z","iopub.execute_input":"2022-07-17T02:05:43.684200Z","iopub.status.idle":"2022-07-17T02:05:47.698765Z","shell.execute_reply.started":"2022-07-17T02:05:43.684155Z","shell.execute_reply":"2022-07-17T02:05:47.697518Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# training params\n\nbatch_size = 32\nepoch_size = 150","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:05:47.703817Z","iopub.execute_input":"2022-07-17T02:05:47.704429Z","iopub.status.idle":"2022-07-17T02:05:47.711335Z","shell.execute_reply.started":"2022-07-17T02:05:47.704390Z","shell.execute_reply":"2022-07-17T02:05:47.710285Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create training generators\n\ntrain_input_data_gen = ImageDataGenerator(rescale=1./255)\ntrain_target_data_gen = ImageDataGenerator(rescale=1./255)\n\ntrain_input_image_generator = train_input_data_gen.flow_from_directory(\n    train_input_path,\n    batch_size=batch_size,\n    color_mode = 'grayscale',\n    target_size=(400, 400),\n    class_mode=None,\n    shuffle=False,\n    seed=0)\n\ntrain_target_image_generator = train_target_data_gen.flow_from_directory(\n    train_target_path,\n    batch_size=batch_size,\n    color_mode = 'grayscale',\n    target_size=(400, 400),\n    class_mode=None,\n    shuffle=False,\n    seed=0)\n\ntrain_generator = zip(train_input_image_generator, train_target_image_generator)\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:05:47.713143Z","iopub.execute_input":"2022-07-17T02:05:47.714725Z","iopub.status.idle":"2022-07-17T02:05:47.951964Z","shell.execute_reply.started":"2022-07-17T02:05:47.714643Z","shell.execute_reply":"2022-07-17T02:05:47.951055Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display some training images and target images\n\nfrom matplotlib import pyplot as plt\n\nn = 0\nfor train, target in zip(train_input_image_generator, train_target_image_generator):\n    plt.figure()\n    plt.subplot(121)\n    plt.imshow((train[0][:,:,0]*255).astype('uint8'),cmap='gray')\n    plt.subplot(122)\n    plt.imshow((target[0][:,:,0]*255).astype('uint8'),cmap='gray')\n    n+=1\n    if n >5:\n        break\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:05:47.955692Z","iopub.execute_input":"2022-07-17T02:05:47.955989Z","iopub.status.idle":"2022-07-17T02:05:51.200882Z","shell.execute_reply.started":"2022-07-17T02:05:47.955949Z","shell.execute_reply":"2022-07-17T02:05:51.199853Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <h1> Create and train model </h1>","metadata":{}},{"cell_type":"code","source":"# Create model\n\ndef autoencoder():\n\n    model = Sequential()\n\n    # input layer\n    model.add(layers.Input(shape=(400,400, 1)))\n\n    # encoder section\n    model.add(layers.Conv2D(32, (3, 3), activation='relu',strides=2,padding='same'))\n    model.add(layers.Conv2D(64, (3, 3), activation='relu',strides=2,padding='same'))\n    model.add(layers.BatchNormalization())\n    \n\n    # decoder section\n    model.add(layers.Conv2DTranspose(64, (3, 3), activation='relu',strides=2,padding='same'))\n    model.add(layers.Conv2DTranspose(32, (3, 3), activation='relu',strides=2,padding='same'))\n    model.add(layers.BatchNormalization())\n    model.add(layers.Conv2DTranspose(1, (3, 3), activation='sigmoid',strides=1, padding='same'))\n\n    # compile model\n    model.compile(optimizer='adam' , loss='mean_squared_error', metrics=['mae'])\n\n    #print model summary\n    model.summary()\n\n    return model\n\n# create model\nmodel = autoencoder()","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:05:51.202547Z","iopub.execute_input":"2022-07-17T02:05:51.203190Z","iopub.status.idle":"2022-07-17T02:05:54.577508Z","shell.execute_reply.started":"2022-07-17T02:05:51.203138Z","shell.execute_reply":"2022-07-17T02:05:54.576361Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# fit model\n\ntraining_sample = train_input_image_generator.samples\n\nmodel.fit(\n        train_generator,\n        steps_per_epoch=np.ceil(training_sample/batch_size),\n        epochs=epoch_size,\n        )\n","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:05:54.580830Z","iopub.execute_input":"2022-07-17T02:05:54.581358Z","iopub.status.idle":"2022-07-17T02:09:17.620244Z","shell.execute_reply.started":"2022-07-17T02:05:54.581323Z","shell.execute_reply":"2022-07-17T02:09:17.619207Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# <h1> Predict clean image and get submission file</h1>","metadata":{}},{"cell_type":"code","source":"%cd /kaggle/working/","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:09:17.622103Z","iopub.execute_input":"2022-07-17T02:09:17.622722Z","iopub.status.idle":"2022-07-17T02:09:17.632298Z","shell.execute_reply.started":"2022-07-17T02:09:17.622675Z","shell.execute_reply":"2022-07-17T02:09:17.631102Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!git clone https://github.com/kwcckw/shabby_images/","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:09:17.634118Z","iopub.execute_input":"2022-07-17T02:09:17.634553Z","iopub.status.idle":"2022-07-17T02:09:36.792372Z","shell.execute_reply.started":"2022-07-17T02:09:17.634510Z","shell.execute_reply":"2022-07-17T02:09:36.791176Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"test_input_path = \"/kaggle/working/shabby_images/Datasets/test\"","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:09:36.794390Z","iopub.execute_input":"2022-07-17T02:09:36.795134Z","iopub.status.idle":"2022-07-17T02:09:36.800506Z","shell.execute_reply.started":"2022-07-17T02:09:36.795083Z","shell.execute_reply":"2022-07-17T02:09:36.799232Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def preprocess(path):\n\n    img = cv2.imread(path, cv2.IMREAD_GRAYSCALE)\n    img = np.asarray(img, dtype=\"float32\")\n    img = img/255.0 #Scaling the pixel values\n    \n    return img.reshape(400,400,1)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:09:36.804751Z","iopub.execute_input":"2022-07-17T02:09:36.805401Z","iopub.status.idle":"2022-07-17T02:09:36.817264Z","shell.execute_reply.started":"2022-07-17T02:09:36.805356Z","shell.execute_reply":"2022-07-17T02:09:36.816037Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get testing image\n\nimg_test_path = sorted(glob.glob(test_input_path+'/input/*'))\n\ntest_imgs = []\nfor file_path in img_test_path:\n    test_imgs.append(preprocess(file_path))\ntest_imgs = np.asarray(test_imgs)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:09:36.818719Z","iopub.execute_input":"2022-07-17T02:09:36.819629Z","iopub.status.idle":"2022-07-17T02:09:38.362701Z","shell.execute_reply.started":"2022-07-17T02:09:36.819546Z","shell.execute_reply":"2022-07-17T02:09:38.361679Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get cleaned images using trained model\n\nimg_predicted = model.predict(test_imgs, batch_size=2)\nfor i, (predicted, testing_path) in enumerate(zip(img_predicted, img_test_path)):\n    predicted_sequeeze = (np.squeeze(predicted) * 255).astype(\"uint8\")\n    cv2.imwrite(test_target_path+os.path.basename(testing_path), predicted_sequeeze)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:09:38.364405Z","iopub.execute_input":"2022-07-17T02:09:38.364805Z","iopub.status.idle":"2022-07-17T02:09:41.145190Z","shell.execute_reply.started":"2022-07-17T02:09:38.364759Z","shell.execute_reply":"2022-07-17T02:09:41.144123Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get cleaned images (optional)\n\nfrom IPython.display import FileLink\n\n!zip -r output_images.zip /kaggle/working/test/target_images/\nFileLink(r'output_images.zip')","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:09:41.147188Z","iopub.execute_input":"2022-07-17T02:09:41.147504Z","iopub.status.idle":"2022-07-17T02:09:43.060209Z","shell.execute_reply.started":"2022-07-17T02:09:41.147451Z","shell.execute_reply":"2022-07-17T02:09:43.058977Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# Display some input testing image and cleaned image from the model\n\nfrom matplotlib import pyplot as plt\n\nn = 0\nfor noisy_path in img_test_path:\n    \n    clean_path = test_target_path + os.path.basename(noisy_path)\n    \n    img_noisy = cv2.imread(noisy_path, cv2.IMREAD_GRAYSCALE)\n    img_clean = cv2.imread(clean_path, cv2.IMREAD_GRAYSCALE)\n    \n    plt.figure()\n    plt.subplot(121)\n    plt.imshow(img_noisy,cmap='gray')\n    plt.subplot(122)\n    plt.imshow(img_clean,cmap='gray')\n    n+=1\n    if n >5:\n        break\n    ","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:09:43.062275Z","iopub.execute_input":"2022-07-17T02:09:43.062941Z","iopub.status.idle":"2022-07-17T02:09:45.200694Z","shell.execute_reply.started":"2022-07-17T02:09:43.062868Z","shell.execute_reply":"2022-07-17T02:09:45.199575Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# create submission file\n\n\ncleaned_images_dir = '/kaggle/working/test/target_images/'\n\ndef select_pixels(img):\n    y,x = img.shape\n\n    pixels = list()\n\n    for i in range(10000):\n        pixel = (random.randrange(y), random.randrange(x))\n\n        if pixel not in pixels:\n            pixels.append(pixel)\n\n    return pixels\n\n\nrandom.seed(0)\n\ncleaned_images = sorted(os.listdir(cleaned_images_dir))\n\nwith open(\"submission.csv\", \"w\") as submission_file:\n    submission_file.write(\"id,predicted\\n\")\n\n    print(\"Processing images...\")\n    filenum = 1\n    for image in tqdm(cleaned_images):\n        \n        img = cv2.imread(cleaned_images_dir + image, cv2.IMREAD_GRAYSCALE)\n        pixels = select_pixels(img)\n\n        for pixel in pixels:\n            y,x = pixel\n            submission_file.write(\"{}_{}_{},{}\\n\".format(filenum, y, x, img[y][x]/255.0))\n\n        filenum += 1\n    print('Done!')","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:49:04.010497Z","iopub.execute_input":"2022-07-17T02:49:04.010815Z","iopub.status.idle":"2022-07-17T02:55:56.122501Z","shell.execute_reply.started":"2022-07-17T02:49:04.010768Z","shell.execute_reply":"2022-07-17T02:55:56.121368Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# get submission file\n\nfrom IPython.display import FileLink\n\nFileLink(r'submission.csv')","metadata":{"execution":{"iopub.status.busy":"2022-07-17T02:55:59.485660Z","iopub.execute_input":"2022-07-17T02:55:59.485970Z","iopub.status.idle":"2022-07-17T02:55:59.494024Z","shell.execute_reply.started":"2022-07-17T02:55:59.485928Z","shell.execute_reply":"2022-07-17T02:55:59.492725Z"},"trusted":true},"execution_count":null,"outputs":[]}]}