Website Image Downloader logoWebsite Image Downloader

Home › Guides

Extract and download images from a website with Python

DIY: requests + BeautifulSoup

import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin

def largest_from_srcset(srcset):
    best, best_score = None, -1
    for part in srcset.split(", "):
        bits = part.strip().split()
        if not bits or bits[0].startswith(("data:", "blob:")):
            continue
        desc = bits[1] if len(bits) > 1 else "1x"
        try:
            score = float(desc[:-1]) * (1 if desc.endswith("w") else 1000)
        except ValueError:
            score = 1000
        if score > best_score:
            best, best_score = bits[0], score
    return best

def image_urls(page_url):
    html = requests.get(page_url, headers={"User-Agent": "Mozilla/5.0"}, timeout=30).text
    soup = BeautifulSoup(html, "html.parser")
    found = []
    for img in soup.find_all("img"):
        src = (largest_from_srcset(img.get("srcset", "")) or largest_from_srcset(img.get("data-srcset", ""))
               or img.get("data-src") or img.get("data-lazy-src") or img.get("data-original") or img.get("src"))
        if src and not src.startswith("data:"):
            found.append(urljoin(page_url, src))
    for source in soup.select("picture source[srcset]"):
        best = largest_from_srcset(source["srcset"])
        if best:
            found.append(urljoin(page_url, best))
    og = soup.find("meta", property="og:image")
    if og and og.get("content"):
        found.append(urljoin(page_url, og["content"]))
    return list(dict.fromkeys(found))  # dedupe, keep order

print(image_urls("https://www.example.com/"))

Then download each URL, check its size with Pillow (see filtering) and hash it to dedupe.

Hosted: apify-client

from apify_client import ApifyClient  # pip install "apify-client>=3"

client = ApifyClient("YOUR_APIFY_TOKEN")
run = client.actor("sste/website-image-downloader").call(run_input={
    "startUrls": [
        {
            "url": "https://www.example.com/"
        }
    ],
    "minWidth": 300,
    "minHeight": 300
})
for img in client.dataset(run.default_dataset_id).iterate_items():
    print(img["width"], img["height"], img["imageUrl"])

zip_record = client.key_value_store(run.default_key_value_store_id).get_record_as_bytes("images.zip")
if zip_record:  # no ZIP when no image matched the filters
    open("images.zip", "wb").write(zip_record["value"])

More: examples on GitHub, API reference.

Related

Try it on Apify → API docs Use with AI agents (MCP)