{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Generate images from text using OpenAI GLIDE","metadata":{}},{"cell_type":"code","source":"# Run this line in Colab to install the package if it is\n# not already installed.\n!pip install git+https://github.com/openai/glide-text2im","metadata":{"execution":{"iopub.status.busy":"2022-09-03T11:45:58.410258Z","iopub.execute_input":"2022-09-03T11:45:58.410731Z","iopub.status.idle":"2022-09-03T11:46:16.092693Z","shell.execute_reply.started":"2022-09-03T11:45:58.410609Z","shell.execute_reply":"2022-09-03T11:46:16.091406Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"from typing import Tuple\nfrom PIL import Image\nfrom IPython.display import display\nimport torch as th\nimport cv2\nimport random\nimport numpy as np\nimport torchvision\n\nimport torch as th\nimport torch.nn.functional as F\n\nfrom glide_text2im.download import load_checkpoint\nfrom glide_text2im.model_creation import (\n    create_model_and_diffusion,\n    model_and_diffusion_defaults,\n    model_and_diffusion_defaults_upsampler\n)","metadata":{"_kg_hide-input":true,"execution":{"iopub.status.busy":"2022-09-03T11:57:55.940886Z","iopub.execute_input":"2022-09-03T11:57:55.94123Z","iopub.status.idle":"2022-09-03T11:57:55.948243Z","shell.execute_reply.started":"2022-09-03T11:57:55.941183Z","shell.execute_reply":"2022-09-03T11:57:55.946789Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import requests\n\nword_site = \"https://www.mit.edu/~ecprice/wordlist.10000\"\n\nresponse = requests.get(word_site)\nWORDS = response.content.splitlines()\n\nlen(WORDS)","metadata":{"execution":{"iopub.status.busy":"2022-09-03T11:46:18.962632Z","iopub.execute_input":"2022-09-03T11:46:18.963027Z","iopub.status.idle":"2022-09-03T11:46:19.622714Z","shell.execute_reply.started":"2022-09-03T11:46:18.962968Z","shell.execute_reply":"2022-09-03T11:46:19.621434Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import os\nparent_dir = '/kaggle/working/glide_gen_inpainting'\nos.mkdir(parent_dir)","metadata":{"execution":{"iopub.status.busy":"2022-09-03T11:49:14.10532Z","iopub.execute_input":"2022-09-03T11:49:14.106181Z","iopub.status.idle":"2022-09-03T11:49:14.112665Z","shell.execute_reply.started":"2022-09-03T11:49:14.106149Z","shell.execute_reply":"2022-09-03T11:49:14.111086Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# This notebook supports both CPU and GPU.\n# On CPU, generating one sample may take on the order of 20 minutes.\n# On a GPU, it should be under a minute.\n\nhas_cuda = th.cuda.is_available()\ndevice = th.device('cpu' if not has_cuda else 'cuda')\n\n\n\nfrom glob import glob\nfrom tqdm.notebook import tqdm\n\nsource_images = glob(\"/kaggle/input/imagenetmini-1000/imagenet-mini/*/*/*.JPEG\")\n\n\n\n# if not os.path.is_dir(parent_dir):\n\n\n\n# sample parameters\n\nupsample_temp = 0.997\nbatch_size = 16\nn_batches = 100\nguidance_scale = 5.0\n\n\n#def show_images(batch: th.Tensor):\n # \"\"\" Display a batch of images inline. \"\"\"\n #   scaled = ((batch + 1)*127.5).round().clamp(0,255).to(th.uint8).cpu()\n #   reshaped = scaled.permute(2, 0, 3, 1).reshape([batch.shape[2], -1, 3])\n #   display(Image.fromarray(reshaped.numpy()))\n\ndef save_images(batch: th.Tensor, nth_batch, parent_dir = parent_dir):\n#    \"\"\" Display a batch of images inline. \"\"\"\n#     scaled = ((batch + 1)*127.5).round().clamp(0,255).to(th.uint8).cpu()\n#     reshaped = scaled.permute(2, 0, 3, 1).reshape([batch.shape[2], -1, 3])\n    for i, img in enumerate(batch):\n        save_path = os.path.join(parent_dir, f'{nth_batch}_{i}.png')\n        torchvision.utils.save_image(img, fp = save_path)\n#         cv2.imwrite(save_path, reshaped.numpy())    \n    \n    \ndef read_image(path: str, size: int = 256) -> Tuple[th.Tensor, th.Tensor]:\n    pil_img = Image.open(path).convert('RGB')\n    pil_img = pil_img.resize((size, size), resample=Image.BICUBIC)\n    img = np.array(pil_img)\n    return th.from_numpy(img)[None].permute(0, 3, 1, 2).float() / 127.5 - 1\n\n\n# Create an classifier-free guidance sampling function\ndef model_fn(x_t, ts, **kwargs):\n    half = x_t[: len(x_t) // 2]\n    combined = th.cat([half, half], dim=0)\n    model_out = model(combined, ts, **kwargs)\n    eps, rest = model_out[:, :3], model_out[:, 3:]\n    cond_eps, uncond_eps = th.split(eps, len(eps) // 2, dim=0)\n    half_eps = uncond_eps + guidance_scale * (cond_eps - uncond_eps)\n    eps = th.cat([half_eps, half_eps], dim=0)\n    return th.cat([eps, rest], dim=1)\n\ndef denoised_fn(x_start):\n    # Force the model to have the exact right x_start predictions\n    # for the part of the image which is known.\n    return (\n        x_start * (1 - model_kwargs['inpaint_mask'])\n        + model_kwargs['inpaint_image'] * model_kwargs['inpaint_mask']\n    )\n\n\n# Create base model.\noptions = model_and_diffusion_defaults()\noptions['inpaint'] = True\noptions['use_fp16'] = has_cuda\noptions['timestep_respacing'] = '100' # use 100 diffusion steps for fast sampling\nmodel, diffusion = create_model_and_diffusion(**options)\nmodel.eval()\nif has_cuda:\n    model.convert_to_fp16()\nmodel.to(device)\nmodel.load_state_dict(load_checkpoint('base-inpaint', device))\nprint('total base parameters', sum(x.numel() for x in model.parameters()))\n\n\n\n# Create upsampler model.\noptions_up = model_and_diffusion_defaults_upsampler()\noptions_up['inpaint'] = True\noptions_up['use_fp16'] = has_cuda\noptions_up['timestep_respacing'] = 'fast27' # use 27 diffusion steps for very fast sampling\nmodel_up, diffusion_up = create_model_and_diffusion(**options_up)\nmodel_up.eval()\nif has_cuda:\n    model_up.convert_to_fp16()\nmodel_up.to(device)\nmodel_up.load_state_dict(load_checkpoint('upsample-inpaint', device))\nprint('total upsampler parameters', sum(x.numel() for x in model_up.parameters()))\n\n\n\n# Tune this parameter to control the sharpness of 256x256 images.\n# A value of 1.0 is sharper, but sometimes results in grainy artifacts.\n# upsample_temp = 0.997\n\n\n\n# Visualize the image we are inpainting\n# show_images(source_image_256 * source_mask_256)\n\n\n\n\n\n\n\n\nfor i in tqdm(range(n_batches)):\n    # Random prompt\n    prompt = random.choice(WORDS).decode(\"utf-8\") \n    # Source image we are inpainting\n    img_path = random.choice(source_images)\n    source_image_256 = read_image(img_path, size=256)\n    source_image_64 = read_image(img_path, size=64)\n\n    # The mask should always be a boolean 64x64 mask, and then we\n    # can upsample it for the second stage.\n    \n    source_mask_64 = th.ones_like(source_image_64)[:, :1]\n    source_mask_64[:, :, 20:] = 0\n    source_mask_256 = F.interpolate(source_mask_64, (256, 256), mode='nearest')\n    \n    # Create the text tokens to feed to the model.\n    tokens = model.tokenizer.encode(prompt)\n    tokens, mask = model.tokenizer.padded_tokens_and_mask(\n        tokens, options['text_ctx']\n    )\n\n    \n    # Create the classifier-free guidance tokens (empty)\n    full_batch_size = batch_size * 2\n    uncond_tokens, uncond_mask = model.tokenizer.padded_tokens_and_mask(\n        [], options['text_ctx']\n    )\n\n    \n    # Pack the tokens together into model kwargs.\n    model_kwargs = dict(\n        tokens=th.tensor(\n            [tokens] * batch_size + [uncond_tokens] * batch_size, device=device\n        ),\n        mask=th.tensor(\n            [mask] * batch_size + [uncond_mask] * batch_size,\n            dtype=th.bool,\n            device=device,\n        ),\n        \n        inpaint_image=(source_image_64 * source_mask_64).repeat(full_batch_size, 1, 1, 1).to(device),\n        inpaint_mask=source_mask_64.repeat(full_batch_size, 1, 1, 1).to(device),\n\n    )\n    \n    \n    # Sample from the base model.\n    #     model.del_cache()\n    samples = diffusion.p_sample_loop(\n        model_fn,\n        (full_batch_size, 3, options[\"image_size\"], options[\"image_size\"]),\n        device=device,\n        clip_denoised=True,\n        progress=True,\n        model_kwargs=model_kwargs,\n        cond_fn=None,\n    )[:batch_size]\n    #     model.del_cache()\n    \n    ##############################\n    # Upsample the 64x64 samples #\n    ##############################\n\n    tokens = model_up.tokenizer.encode(prompt)\n    tokens, mask = model_up.tokenizer.padded_tokens_and_mask(\n    tokens, options_up['text_ctx']\n    )\n\n    # Create the model conditioning dict.\n    model_kwargs = dict(\n        # Low-res image to upsample.\n        low_res=((samples+1)*127.5).round()/127.5 - 1,\n\n        # Text tokens\n        tokens=th.tensor(\n            [tokens] * batch_size, device=device\n        ),\n        mask=th.tensor(\n            [mask] * batch_size,\n            dtype=th.bool,\n            device=device,\n        ),\n\n    # Masked inpainting image.\n    inpaint_image=(source_image_256 * source_mask_256).repeat(batch_size, 1, 1, 1).to(device),\n    inpaint_mask=source_mask_256.repeat(batch_size, 1, 1, 1).to(device),\n)\n    # Sample from the base model.\n#     model_up.del_cache()\n    up_shape = (batch_size, 3, options_up[\"image_size\"], options_up[\"image_size\"])\n    up_samples = diffusion_up.ddim_sample_loop(\n        model_up,\n        up_shape,\n        noise=th.randn(up_shape, device=device) * upsample_temp,\n        device=device,\n        clip_denoised=True,\n        progress=True,\n        model_kwargs=model_kwargs,\n        cond_fn=None,\n    )[:batch_size]\n#     model_up.del_cache()\n    \n    # Show the output\n    print(f'[{i}/{n_batches}]{batch_size} images generated for prompt: {prompt}')\n    save_images(up_samples, i)\n","metadata":{"execution":{"iopub.status.busy":"2022-09-03T11:58:03.794195Z","iopub.execute_input":"2022-09-03T11:58:03.794812Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"#upsample_temp = 0.997\n#batch_size = 16\n#n_batches = 100\n#guidance_scale = 3.0","metadata":{},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"I was really impressed with the images that were included in the paper but when I try my own prompts the results are consistently much worse.","metadata":{}},{"cell_type":"markdown","source":"**GLIDE: Towards Photorealistic Image Generation and Editing with Text-Guided Diffusion Models**\n - Paper: https://arxiv.org/abs/2112.10741\n - Repo: https://github.com/openai/glide-text2im\n - Code adapted from https://github.com/openai/glide-text2im/blob/main/notebooks/text2im.ipynb\n ","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}