Extract and download images from a website with Python
DIY: requests + BeautifulSoup
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
def largest_from_srcset(srcset):
best, best_score = None, -1
for part in srcset.split(", "):
bits = part.strip().split()
if not bits or bits[0].startswith(("data:", "blob:")):
continue
desc = bits[1] if len(bits) > 1 else "1x"
try:
score = float(desc[:-1]) * (1 if desc.endswith("w") else 1000)
except ValueError:
score = 1000
if score > best_score:
best, best_score = bits[0], score
return best
def image_urls(page_url):
html = requests.get(page_url, headers={"User-Agent": "Mozilla/5.0"}, timeout=30).text
soup = BeautifulSoup(html, "html.parser")
found = []
for img in soup.find_all("img"):
src = (largest_from_srcset(img.get("srcset", "")) or largest_from_srcset(img.get("data-srcset", ""))
or img.get("data-src") or img.get("data-lazy-src") or img.get("data-original") or img.get("src"))
if src and not src.startswith("data:"):
found.append(urljoin(page_url, src))
for source in soup.select("picture source[srcset]"):
best = largest_from_srcset(source["srcset"])
if best:
found.append(urljoin(page_url, best))
og = soup.find("meta", property="og:image")
if og and og.get("content"):
found.append(urljoin(page_url, og["content"]))
return list(dict.fromkeys(found)) # dedupe, keep order
print(image_urls("https://www.example.com/"))
Then download each URL, check its size with Pillow (see filtering) and hash it to dedupe.
Hosted: apify-client
from apify_client import ApifyClient # pip install "apify-client>=3"
client = ApifyClient("YOUR_APIFY_TOKEN")
run = client.actor("sste/website-image-downloader").call(run_input={
"startUrls": [
{
"url": "https://www.example.com/"
}
],
"minWidth": 300,
"minHeight": 300
})
for img in client.dataset(run.default_dataset_id).iterate_items():
print(img["width"], img["height"], img["imageUrl"])
zip_record = client.key_value_store(run.default_key_value_store_id).get_record_as_bytes("images.zip")
if zip_record: # no ZIP when no image matched the filters
open("images.zip", "wb").write(zip_record["value"])
More: examples on GitHub, API reference.