{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"code","source":"import joblib\nimport json\nimport multiprocessing\nimport os\nimport shutil\nimport zipfile\nimport warnings\nwarnings.simplefilter('ignore', UserWarning)\n\nimport tqdm.auto as tqdm\n\nimport skimage\nimport skimage.io\nimport skimage.color\nimport skimage.transform","metadata":{"execution":{"iopub.status.busy":"2022-02-05T16:44:10.943230Z","iopub.execute_input":"2022-02-05T16:44:10.943553Z","iopub.status.idle":"2022-02-05T16:44:12.795415Z","shell.execute_reply.started":"2022-02-05T16:44:10.943468Z","shell.execute_reply":"2022-02-05T16:44:12.794413Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Add Painter by Numbers competition dataset on Kaggle to the session before running the notebook.\n\nCOCO Dataset downloaded from [here](https://cocodataset.org/#download).\n\nPainter by Numbers competition dataset is [here](https://www.kaggle.com/c/painter-by-numbers/data).\n\nIf you run this locally, download the data beforehand and adjust the paths appropriately.","metadata":{}},{"cell_type":"code","source":"UPDATE = False # set to True if this is an update to the dataset","metadata":{"execution":{"iopub.status.busy":"2022-02-05T16:44:12.797349Z","iopub.execute_input":"2022-02-05T16:44:12.797578Z","iopub.status.idle":"2022-02-05T16:44:12.804416Z","shell.execute_reply.started":"2022-02-05T16:44:12.797552Z","shell.execute_reply":"2022-02-05T16:44:12.802589Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_STORAGE_PATH = \"/kaggle/tmp/coco_wikiart_nst_dataset\"\nos.makedirs(DATASET_STORAGE_PATH)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T16:44:12.805834Z","iopub.execute_input":"2022-02-05T16:44:12.806369Z","iopub.status.idle":"2022-02-05T16:44:12.820284Z","shell.execute_reply.started":"2022-02-05T16:44:12.806312Z","shell.execute_reply":"2022-02-05T16:44:12.819038Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"IMAGE_SIZE = 512\nDATASET_SIZE = 50000 # this fits into the Kaggle /tmp folder, but barely","metadata":{"execution":{"iopub.status.busy":"2022-02-05T16:44:12.821423Z","iopub.execute_input":"2022-02-05T16:44:12.821650Z","iopub.status.idle":"2022-02-05T16:44:12.833389Z","shell.execute_reply.started":"2022-02-05T16:44:12.821623Z","shell.execute_reply":"2022-02-05T16:44:12.832642Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!wget http://images.cocodataset.org/zips/unlabeled2017.zip -O unlabeled.zip\nwith zipfile.ZipFile(\"unlabeled.zip\", \"r\") as archive:\n    for i, member in enumerate(tqdm.tqdm(archive.namelist(), desc=\"Extracting\", unit=\"files\", unit_scale=False)):\n        if i > DATASET_SIZE:\n            break\n        archive.extract(member, f\"{DATASET_STORAGE_PATH}\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T16:44:12.835610Z","iopub.execute_input":"2022-02-05T16:44:12.836202Z","iopub.status.idle":"2022-02-05T16:52:53.970636Z","shell.execute_reply.started":"2022-02-05T16:44:12.836146Z","shell.execute_reply":"2022-02-05T16:52:53.969061Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with zipfile.ZipFile(\"../input/painter-by-numbers/train.zip\", \"r\") as archive:\n    for i, member in enumerate(tqdm.tqdm(archive.namelist(), desc=\"Extracting\", unit=\"files\", unit_scale=False)):\n        if i > DATASET_SIZE:\n            break\n        archive.extract(member, f\"{DATASET_STORAGE_PATH}\")","metadata":{"id":"DNjkSk0Vykl6","outputId":"3c528bea-5fa6-49bf-a831-a0f2574983a9","execution":{"iopub.status.busy":"2022-02-05T16:52:53.972405Z","iopub.execute_input":"2022-02-05T16:52:53.972982Z","iopub.status.idle":"2022-02-05T17:00:00.257998Z","shell.execute_reply.started":"2022-02-05T16:52:53.972922Z","shell.execute_reply":"2022-02-05T17:00:00.256995Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"extracted = len(list(os.listdir(f\"{DATASET_STORAGE_PATH}/train\")))\nprint(f\"Already extracted {extracted} files\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:00:00.259966Z","iopub.execute_input":"2022-02-05T17:00:00.261009Z","iopub.status.idle":"2022-02-05T17:00:00.303002Z","shell.execute_reply.started":"2022-02-05T17:00:00.260928Z","shell.execute_reply":"2022-02-05T17:00:00.301775Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"with zipfile.ZipFile(\"../input/painter-by-numbers/test.zip\", \"r\") as archive:\n    for i, member in enumerate(tqdm.tqdm(archive.namelist(), desc=\"Extracting\", unit=\"files\", unit_scale=False)):\n        if extracted + i > DATASET_SIZE:\n            break\n        archive.extract(member, f\"{DATASET_STORAGE_PATH}\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:00:00.307179Z","iopub.execute_input":"2022-02-05T17:00:00.307484Z","iopub.status.idle":"2022-02-05T17:00:01.109562Z","shell.execute_reply.started":"2022-02-05T17:00:00.307446Z","shell.execute_reply":"2022-02-05T17:00:01.107858Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def resize_image(file, target_dir):\n    file = os.path.abspath(file)\n    fname, ext = os.path.splitext(file)\n    root, fname = os.path.split(fname)\n    try:\n        image = skimage.io.imread(file)\n    except Exception:\n        os.remove(file)\n        return\n    resized = skimage.transform.resize(image, (IMAGE_SIZE, IMAGE_SIZE), anti_aliasing=True)\n    if len(image.shape) == 2:\n        resized = skimage.color.gray2rgb(resized)\n    if image.shape[-1] == 4:\n        resized = skimage.color.rgba2rgb(resized)\n    skimage.io.imsave(os.path.join(target_dir, fname + \".jpg\"), skimage.img_as_ubyte(resized))","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:00:01.111212Z","iopub.execute_input":"2022-02-05T17:00:01.112378Z","iopub.status.idle":"2022-02-05T17:00:01.126350Z","shell.execute_reply.started":"2022-02-05T17:00:01.112290Z","shell.execute_reply":"2022-02-05T17:00:01.125517Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def copy_and_resize_images(paths, target, size):\n    \"\"\"\n    Move dataset images to a single folder while resizing each image.\n    :param paths: list of paths to all directories with images.\n    :param target: target directory where to store all images.\n    \"\"\"\n    os.makedirs(target, exist_ok=True)\n    tqdm_wrapped = tqdm.tqdm(paths, desc=\"Moving\", unit=\"directory\", unit_scale=False)\n    acc = 0\n    for dir in tqdm_wrapped:\n        tqdm_wrapped.set_description(f\"Moving files from {os.path.abspath(dir)}\")\n        acc = 0\n        for root, dirs, files in os.walk(os.path.abspath(dir)):\n            if acc + len(files) > size:\n                files = files[:size - acc]\n            files = [os.path.join(root, i) for i in files]\n            num_of_files = joblib.Parallel(n_jobs=-1)\\\n            (joblib.delayed(resize_image)(e, target)\\\n             for i, e in enumerate(tqdm.tqdm(files, desc=\"Resizing\", unit=\"images\", unit_scale=False)))\n            copied = len(num_of_files)\n            acc += copied","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:00:01.127412Z","iopub.execute_input":"2022-02-05T17:00:01.127658Z","iopub.status.idle":"2022-02-05T17:00:01.148028Z","shell.execute_reply.started":"2022-02-05T17:00:01.127628Z","shell.execute_reply":"2022-02-05T17:00:01.146662Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"Copy your entire `kaggle.json` to a session Secret (`Add-ons -> Secrets`).\n\n(Not applicable for local runs)","metadata":{}},{"cell_type":"code","source":"if os.environ.get('KAGGLE_KERNEL_RUN_TYPE','') != '':\n    from kaggle_secrets import UserSecretsClient\n    user_secrets = UserSecretsClient()\n    kaggle_key = user_secrets.get_secret(\"kaggle_key\")\n    os.makedirs(\"/root/.kaggle\", exist_ok=True)\n    with open(\"/root/.kaggle/kaggle.json\", \"w\") as f:\n        f.write(kaggle_key)\n    del kaggle_key\n    os.chmod(\"/root/.kaggle/kaggle.json\", 600)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:00:01.149539Z","iopub.execute_input":"2022-02-05T17:00:01.149853Z","iopub.status.idle":"2022-02-05T17:00:01.351420Z","shell.execute_reply.started":"2022-02-05T17:00:01.149760Z","shell.execute_reply":"2022-02-05T17:00:01.350036Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"copy_and_resize_images([f\"{DATASET_STORAGE_PATH}/train\", f\"{DATASET_STORAGE_PATH}/test\"], \n                       f\"{DATASET_STORAGE_PATH}/style\", DATASET_SIZE)","metadata":{"execution":{"iopub.status.busy":"2022-02-05T17:00:01.353039Z","iopub.execute_input":"2022-02-05T17:00:01.353309Z","iopub.status.idle":"2022-02-05T18:52:37.653259Z","shell.execute_reply.started":"2022-02-05T17:00:01.353278Z","shell.execute_reply":"2022-02-05T18:52:37.652377Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shutil.rmtree(f\"{DATASET_STORAGE_PATH}/train\")\nshutil.rmtree(f\"{DATASET_STORAGE_PATH}/test\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T18:52:37.655299Z","iopub.execute_input":"2022-02-05T18:52:37.655562Z","iopub.status.idle":"2022-02-05T18:52:40.327363Z","shell.execute_reply.started":"2022-02-05T18:52:37.655532Z","shell.execute_reply":"2022-02-05T18:52:40.326654Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"DATASET_SIZE = len(list(os.listdir(f\"{DATASET_STORAGE_PATH}/style\")))","metadata":{"execution":{"iopub.status.busy":"2022-02-05T18:52:40.331414Z","iopub.execute_input":"2022-02-05T18:52:40.331847Z","iopub.status.idle":"2022-02-05T18:52:40.384148Z","shell.execute_reply.started":"2022-02-05T18:52:40.331795Z","shell.execute_reply":"2022-02-05T18:52:40.383270Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"copy_and_resize_images([f\"{DATASET_STORAGE_PATH}/unlabeled2017\"],\n          f\"{DATASET_STORAGE_PATH}/content\", DATASET_SIZE)","metadata":{"id":"O4Uz5M6wzGH2","execution":{"iopub.status.busy":"2022-02-05T18:52:40.387176Z","iopub.execute_input":"2022-02-05T18:52:40.388141Z","iopub.status.idle":"2022-02-05T19:27:24.488867Z","shell.execute_reply.started":"2022-02-05T18:52:40.388031Z","shell.execute_reply":"2022-02-05T19:27:24.486307Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"shutil.rmtree(f\"{DATASET_STORAGE_PATH}/unlabeled2017\")","metadata":{"execution":{"iopub.status.busy":"2022-02-05T19:27:24.494926Z","iopub.execute_input":"2022-02-05T19:27:24.496872Z","iopub.status.idle":"2022-02-05T19:27:27.070599Z","shell.execute_reply.started":"2022-02-05T19:27:24.496808Z","shell.execute_reply":"2022-02-05T19:27:27.069845Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!ls {DATASET_STORAGE_PATH}","metadata":{"execution":{"iopub.status.busy":"2022-02-05T19:27:27.072179Z","iopub.execute_input":"2022-02-05T19:27:27.072558Z","iopub.status.idle":"2022-02-05T19:27:27.965231Z","shell.execute_reply.started":"2022-02-05T19:27:27.072526Z","shell.execute_reply":"2022-02-05T19:27:27.964151Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!kaggle datasets init -p {DATASET_STORAGE_PATH}\nMETADATA = {\n        \"title\": \"COCO/WikiArt NST Dataset\",\n        \"id\": \"shaorrran/coco-wikiart-nst-dataset-512-100000\",\n        \"description\": \"Dataset for Neural Style Transfer consisting of COCO2017 images \\\n        and Kaggle competition \\\"Painter by Numbers\\\" dataset.\\nThe number of COCO images and style images is the same.\\n\\\n        Intended for use with NST using Adaptive Instance Normalization.\\n\\\n        All respective licenses for used datasets apply to corresponding parts of this dataset.\",\n        \"licenses\": [{\"name\": \"unknown\"}],\n    }\nwith open(f\"{DATASET_STORAGE_PATH}/dataset-metadata.json\", \"w\", encoding=\"utf-8\") as f:\n    json.dump(METADATA, f, ensure_ascii=False, indent=4)\nif UPDATE:\n    !kaggle datasets version -p {DATASET_STORAGE_PATH} -m \"update dataset\" -r zip\nelse:\n    !kaggle datasets create -p {DATASET_STORAGE_PATH} -r zip","metadata":{"id":"xN8D3dBu3gfE","outputId":"ad53b429-d784-4063-da23-5360488247f8","execution":{"iopub.status.busy":"2022-02-05T19:27:27.967163Z","iopub.execute_input":"2022-02-05T19:27:27.967464Z","iopub.status.idle":"2022-02-05T19:31:53.243736Z","shell.execute_reply.started":"2022-02-05T19:27:27.967428Z","shell.execute_reply":"2022-02-05T19:31:53.242464Z"},"trusted":true},"execution_count":null,"outputs":[]}]}