How to extract image URLs from a webpage (srcset, lazy-loaded, CSS)
To get every image URL on a page you need to look in more places than <img src>:
| Where | Example |
|---|---|
src | <img src="/a.jpg"> |
srcset (take the largest) | srcset="a-480.jpg 480w, a-1600.jpg 1600w" |
<picture><source> | WebP/AVIF variants |
| Lazy-load attributes | data-src, data-srcset, data-lazy-src, data-original |
<noscript> fallbacks | the real <img> for no-JS clients |
| CSS | style="background-image:url(...)", data-bg |
| Meta | og:image, twitter:image |
| Links | <a href="full-size.jpg"> in galleries |
Always resolve relative URLs against the page URL (and <base href> if present), drop data: URIs and dedupe.
Python
import requests
from bs4 import BeautifulSoup
from urllib.parse import urljoin
def largest_from_srcset(srcset):
best, best_score = None, -1
for part in srcset.split(", "):
bits = part.strip().split()
if not bits or bits[0].startswith(("data:", "blob:")):
continue
desc = bits[1] if len(bits) > 1 else "1x"
try:
score = float(desc[:-1]) * (1 if desc.endswith("w") else 1000)
except ValueError:
score = 1000
if score > best_score:
best, best_score = bits[0], score
return best
def image_urls(page_url):
html = requests.get(page_url, headers={"User-Agent": "Mozilla/5.0"}, timeout=30).text
soup = BeautifulSoup(html, "html.parser")
found = []
for img in soup.find_all("img"):
src = (largest_from_srcset(img.get("srcset", "")) or largest_from_srcset(img.get("data-srcset", ""))
or img.get("data-src") or img.get("data-lazy-src") or img.get("data-original") or img.get("src"))
if src and not src.startswith("data:"):
found.append(urljoin(page_url, src))
for source in soup.select("picture source[srcset]"):
best = largest_from_srcset(source["srcset"])
if best:
found.append(urljoin(page_url, best))
og = soup.find("meta", property="og:image")
if og and og.get("content"):
found.append(urljoin(page_url, og["content"]))
return list(dict.fromkeys(found)) # dedupe, keep order
print(image_urls("https://www.example.com/"))
JavaScript
import * as cheerio from 'cheerio'; // npm i cheerio
const largest = (srcset = '') => srcset.split(/,\s+/).map((p) => p.trim().split(/\s+/))
.filter(([u]) => u && !/^(data|blob):/.test(u))
.map(([u, d = '1x']) => [u, (parseFloat(d) || 1) * (d.endsWith('w') ? 1 : 1000)])
.sort((a, b) => b[1] - a[1])[0]?.[0];
export async function imageUrls(pageUrl) {
const html = await (await fetch(pageUrl, { headers: { 'user-agent': 'Mozilla/5.0' } })).text();
const $ = cheerio.load(html);
const out = new Set();
$('img').each((_, el) => {
const $el = $(el);
const src = largest($el.attr('srcset')) || largest($el.attr('data-srcset'))
|| $el.attr('data-src') || $el.attr('data-lazy-src') || $el.attr('src');
if (src && !src.startsWith('data:')) out.add(new URL(src, pageUrl).href);
});
$('picture source[srcset]').each((_, el) => {
const best = largest($(el).attr('srcset'));
if (best) out.add(new URL(best, pageUrl).href);
});
const og = $('meta[property="og:image"]').attr('content');
if (og) out.add(new URL(og, pageUrl).href);
return [...out];
}
console.log(await imageUrls('https://www.example.com/'));
Without code: list mode
The hosted Website Image Downloader covers all the sources above. With "mode": "list" it only returns URLs (no downloads), at $0.05 per 1,000 URLs:
curl -X POST "https://api.apify.com/v2/acts/sste~website-image-downloader/run-sync-get-dataset-items" \
-H "Authorization: Bearer $APIFY_TOKEN" -H "Content-Type: application/json" \
-d '{"startUrls":[{"url":"https://www.example.com/"}],"mode":"list"}'