{"metadata":{"kernelspec":{"language":"python","display_name":"Python 3","name":"python3"},"language_info":{"pygments_lexer":"ipython3","nbconvert_exporter":"python","version":"3.6.4","file_extension":".py","codemirror_mode":{"name":"ipython","version":3},"name":"python","mimetype":"text/x-python"}},"nbformat_minor":4,"nbformat":4,"cells":[{"cell_type":"markdown","source":"# Install","metadata":{}},{"cell_type":"code","source":"# install google chrome\n!wget https://dl.google.com/linux/linux_signing_key.pub\n!sudo apt-key add linux_signing_key.pub\n!echo 'deb [arch=amd64] http://dl.google.com/linux/chrome/deb/ stable main' >> /etc/apt/sources.list.d/google-chrome.list\n!sudo apt-get -y update\n!sudo apt-get install -y google-chrome-stable","metadata":{"_uuid":"8f2839f25d086af736a60e9eeb907d3b93b6e0e5","_cell_guid":"b1076dfc-b9ad-4769-8c92-a6c4dae69d19","_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-17T16:38:36.781934Z","iopub.execute_input":"2022-07-17T16:38:36.782276Z","iopub.status.idle":"2022-07-17T16:39:03.335707Z","shell.execute_reply.started":"2022-07-17T16:38:36.782194Z","shell.execute_reply":"2022-07-17T16:39:03.334526Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# install chromedriver\n!wget -O /tmp/chromedriver.zip http://chromedriver.storage.googleapis.com/`curl -sS chromedriver.storage.googleapis.com/LATEST_RELEASE`/chromedriver_linux64.zip\n!unzip /tmp/chromedriver.zip chromedriver -d /usr/local/bin/","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-17T16:39:03.338304Z","iopub.execute_input":"2022-07-17T16:39:03.33869Z","iopub.status.idle":"2022-07-17T16:39:05.152667Z","shell.execute_reply.started":"2022-07-17T16:39:03.33865Z","shell.execute_reply":"2022-07-17T16:39:05.151566Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# install selenium\n!sudo apt install -y python3-selenium\n!pip install selenium==3.141.0 > /dev/null","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-17T16:39:05.154347Z","iopub.execute_input":"2022-07-17T16:39:05.154964Z","iopub.status.idle":"2022-07-17T16:39:31.700278Z","shell.execute_reply.started":"2022-07-17T16:39:05.154925Z","shell.execute_reply":"2022-07-17T16:39:31.699165Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Imports and preparations","metadata":{}},{"cell_type":"code","source":"# import libraries\nimport io\nimport os\nimport time\nimport shutil\nimport hashlib\nimport requests\nimport signal\nimport errno\n\nfrom tqdm import tqdm\nfrom multiprocessing import Pool\nfrom PIL import Image, ImageDraw\nfrom selenium import webdriver\nfrom selenium.webdriver.common.keys import Keys\nfrom selenium.webdriver.common.by import By","metadata":{"execution":{"iopub.status.busy":"2022-07-17T16:39:36.912888Z","iopub.execute_input":"2022-07-17T16:39:36.913622Z","iopub.status.idle":"2022-07-17T16:39:36.9627Z","shell.execute_reply.started":"2022-07-17T16:39:36.913585Z","shell.execute_reply":"2022-07-17T16:39:36.96181Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!mkdir data   # directory for saving all parsed images","metadata":{"execution":{"iopub.status.busy":"2022-07-17T17:58:14.939214Z","iopub.execute_input":"2022-07-17T17:58:14.939575Z","iopub.status.idle":"2022-07-17T17:58:15.60731Z","shell.execute_reply.started":"2022-07-17T17:58:14.939545Z","shell.execute_reply":"2022-07-17T17:58:15.605984Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"chrome_options = webdriver.ChromeOptions()\nchrome_options.add_argument('--no-sandbox')\nchrome_options.add_argument('--headless')\nchrome_options.add_argument('--disable-gpu')\nchrome_options.add_argument('--disable-dev-shm-usage')\nchrome_options.add_argument(\"--window-size=1920,1080\")\ndriver = webdriver.Chrome(options=chrome_options)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T16:39:38.519693Z","iopub.execute_input":"2022-07-17T16:39:38.522347Z","iopub.status.idle":"2022-07-17T16:39:40.281326Z","shell.execute_reply.started":"2022-07-17T16:39:38.522305Z","shell.execute_reply":"2022-07-17T16:39:40.279905Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Functions","metadata":{}},{"cell_type":"code","source":"def fetch_image_urls(query:str, max_links_to_fetch:int, wd:webdriver, sleep_between_interactions:int=1):\n    \"\"\"\n    This function scrolls down and collects links to images in a google search\n    \"\"\"\n    def scroll_to_end(wd):\n        wd.execute_script(\"window.scrollTo(0, document.body.scrollHeight);\")\n        time.sleep(sleep_between_interactions)    \n    \n    # build the google query\n    search_url = \"https://www.google.com/search?safe=off&site=&tbm=isch&source=hp&q={q}&oq={q}&gs_l=img\"\n\n    # load the page\n    wd.get(search_url.format(q=query))\n\n    image_urls = set()\n    image_count = 0\n    results_start = 0\n    while image_count < max_links_to_fetch:\n        scroll_to_end(wd)\n\n        # get all image thumbnail results\n        thumbnail_results = wd.find_elements_by_css_selector(\"img.Q4LuWd\")\n        number_results = len(thumbnail_results)\n        \n        print(f\"Found: {number_results} search results. Extracting links from {results_start}:{number_results}\")\n        \n        for img in thumbnail_results[results_start:number_results]:\n            # try to click every thumbnail such that we can get the real image behind it\n            try:\n                img.click()\n                time.sleep(sleep_between_interactions)\n            except Exception:\n                continue\n\n            # extract image urls    \n            actual_images = wd.find_elements_by_css_selector('img.n3VNCb')\n            for actual_image in actual_images:\n                if actual_image.get_attribute('src') and 'http' in actual_image.get_attribute('src'):\n                    image_urls.add(actual_image.get_attribute('src'))\n                    \n                    print(actual_image.get_attribute('src'))\n                    #print(actual_image.text)\n\n            image_count = len(image_urls)\n\n            if len(image_urls) >= max_links_to_fetch:\n                print(f\"Found: {len(image_urls)} image links, done!\")\n                break\n        else:\n            print(\"Found:\", len(image_urls), \"image links, looking for more ...\")\n            time.sleep(10)\n            #return\n            load_more_button = wd.find_element_by_css_selector(\".mye4qd\")\n            if load_more_button:\n                wd.execute_script(\"document.querySelector('.mye4qd').click();\")\n\n        # move the result startpoint further down\n        results_start = len(thumbnail_results)\n\n    return image_urls","metadata":{"execution":{"iopub.status.busy":"2022-07-17T16:39:43.154232Z","iopub.execute_input":"2022-07-17T16:39:43.154683Z","iopub.status.idle":"2022-07-17T16:39:43.176297Z","shell.execute_reply.started":"2022-07-17T16:39:43.154645Z","shell.execute_reply":"2022-07-17T16:39:43.175052Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def persist_image(folder_path:str,url:str,name=None):\n    \"\"\"\n    This function saves images to desktop in the 'folder_path' directory\n    \"\"\"\n    try:\n        image_content = requests.get(url).content\n\n    except Exception as e:\n        print(f\"ERROR - Could not download {url} - {e}\")\n\n    try:\n        image_file = io.BytesIO(image_content)\n        image = Image.open(image_file).convert('RGB')\n        if name == None:\n            name = hashlib.sha1(image_content).hexdigest()[:10]\n        file_path = os.path.join(folder_path,name + '.jpg')\n        with open(file_path, 'wb') as f:\n            image.save(f, \"JPEG\", quality=85)\n        print(f\"SUCCESS - saved {url} - as {file_path}\")\n    except Exception as e:\n        print(f\"ERROR - Could not save {url} - {e}\")","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:52:33.461534Z","iopub.execute_input":"2022-07-18T10:52:33.46213Z","iopub.status.idle":"2022-07-18T10:52:33.470352Z","shell.execute_reply.started":"2022-07-18T10:52:33.462096Z","shell.execute_reply":"2022-07-18T10:52:33.468723Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"class TimeoutError(Exception):\n    pass\n\nclass timeout:\n    \"\"\"\n    This function interrupts too long functions (when something is bad)\n    \"\"\"\n    def __init__(self, seconds=1, error_message='Timeout'):\n        self.seconds = seconds\n        self.error_message = error_message\n    def handle_timeout(self, signum, frame):\n        raise TimeoutError(self.error_message)\n    def __enter__(self):\n        signal.signal(signal.SIGALRM, self.handle_timeout)\n        signal.alarm(self.seconds)\n    def __exit__(self, type, value, traceback):\n        signal.alarm(0)","metadata":{"execution":{"iopub.status.busy":"2022-07-17T16:39:44.277702Z","iopub.execute_input":"2022-07-17T16:39:44.278784Z","iopub.status.idle":"2022-07-17T16:39:44.286853Z","shell.execute_reply.started":"2022-07-17T16:39:44.278714Z","shell.execute_reply":"2022-07-17T16:39:44.285867Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"def parse_query(query):\n    \"\"\"\n    This function saves images from google search with a given query. Images are being saved in 'data/$query' directory\n    \"\"\"\n    driver = webdriver.Chrome(options=chrome_options)\n\n    links = fetch_image_urls(query, N, driver)\n\n    directory = 'data/{}'.format(query.replace(\" \", \"_\"))\n    try:\n        os.stat(directory)\n    except:\n        os.mkdir(directory) # create directory for this query if it doesn't exist\n\n    for i, link in enumerate(list(links)):\n        with timeout(seconds=5): # if it takes more than 5 sec to save an image, we will skip it\n            persist_image(directory, link, name=str(i))\n","metadata":{"execution":{"iopub.status.busy":"2022-07-18T10:52:23.4221Z","iopub.execute_input":"2022-07-18T10:52:23.422502Z","iopub.status.idle":"2022-07-18T10:52:23.451994Z","shell.execute_reply.started":"2022-07-18T10:52:23.422415Z","shell.execute_reply":"2022-07-18T10:52:23.45108Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Parameters","metadata":{}},{"cell_type":"code","source":"N = 200                 # how many images to parse\nNUMBER_OF_BROUSERS = 2  # The number of brousers that will search queries in parallel \n                        # it mostly depends on the speed of your internet connection","metadata":{"execution":{"iopub.status.busy":"2022-07-17T17:58:24.576875Z","iopub.execute_input":"2022-07-17T17:58:24.577481Z","iopub.status.idle":"2022-07-17T17:58:24.582385Z","shell.execute_reply.started":"2022-07-17T17:58:24.577445Z","shell.execute_reply":"2022-07-17T17:58:24.581341Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"# the following queries will be inserted into google search\n\nOBJECTS = [\n    'apparel & accessories',\n    'packaged goods',\n    'landmarks',\n    'furniture & home decor',\n    'storefronts',\n    'dishes',\n    'artwork',\n    'toys',\n    'memes',\n    'illustrations',\n    'cars'\n]","metadata":{"execution":{"iopub.status.busy":"2022-07-17T17:58:23.755647Z","iopub.execute_input":"2022-07-17T17:58:23.756668Z","iopub.status.idle":"2022-07-17T17:58:23.762943Z","shell.execute_reply.started":"2022-07-17T17:58:23.756627Z","shell.execute_reply":"2022-07-17T17:58:23.761804Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"# Main part","metadata":{}},{"cell_type":"code","source":"if __name__ == '__main__':\n    with Pool(NUMBER_OF_BROUSERS) as p:\n        print(p.map(parse_query, OBJECTS))","metadata":{"_kg_hide-output":true,"execution":{"iopub.status.busy":"2022-07-17T17:58:25.432279Z","iopub.execute_input":"2022-07-17T17:58:25.432956Z","iopub.status.idle":"2022-07-17T17:59:08.014015Z","shell.execute_reply.started":"2022-07-17T17:58:25.432921Z","shell.execute_reply":"2022-07-17T17:59:08.012749Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"for query in OBJECTS:\n    directory = 'data/{}'.format(query.replace(\" \", \"_\"))\n    \n    print(\"There are {} photos of {}\".format(len(os.listdir(directory)), query))","metadata":{"execution":{"iopub.status.busy":"2022-07-17T17:59:14.804347Z","iopub.execute_input":"2022-07-17T17:59:14.804728Z","iopub.status.idle":"2022-07-17T17:59:14.811905Z","shell.execute_reply.started":"2022-07-17T17:59:14.804698Z","shell.execute_reply":"2022-07-17T17:59:14.810796Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"import shutil\nshutil.make_archive('dataset', 'zip', 'data')","metadata":{"execution":{"iopub.status.busy":"2022-07-17T18:00:31.608388Z","iopub.execute_input":"2022-07-17T18:00:31.60898Z","iopub.status.idle":"2022-07-17T18:00:31.939789Z","shell.execute_reply.started":"2022-07-17T18:00:31.608937Z","shell.execute_reply":"2022-07-17T18:00:31.938829Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"code","source":"!rm -rf data\n!rm -rf linux_signing_key.pub","metadata":{"execution":{"iopub.status.busy":"2022-07-17T18:01:11.910844Z","iopub.execute_input":"2022-07-17T18:01:11.911892Z","iopub.status.idle":"2022-07-17T18:01:13.375991Z","shell.execute_reply.started":"2022-07-17T18:01:11.911852Z","shell.execute_reply":"2022-07-17T18:01:13.374673Z"},"trusted":true},"execution_count":null,"outputs":[]},{"cell_type":"markdown","source":"<div class=\"alert alert-warning\">\n    <h1> Warning!","metadata":{}},{"cell_type":"markdown","source":"<div class=\"alert alert-warning\">\n    I would recommend that you create a <mark>private dataset from dataset.zip</mark> so that you can use it in your work and not risk getting banned for violating the <mark>Kaggle Community Guidelines</mark>.\n</div>","metadata":{}},{"cell_type":"code","source":"","metadata":{},"execution_count":null,"outputs":[]}]}