feat(tools): add a Wikimedia Commons demo library downloader

tools/demo/fetch-commons-photos.py builds a screenshot library from
Wikimedia Commons: geosearch around ten cities (a ring of nine points, since
each query returns at most 500 files), then only freely licensed JPEG
originals whose own EXIF has a capture date and GPS within range, skipping
placeholder "city centre" positions, at most two per author per city, with
an optional filter on Commons' Quality/Featured pictures. Writes CREDITS.md
and credits.csv, resumes interrupted runs, standard library only. Listed in
the README's useful scripts.

Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com>
Claude-Session: https://claude.ai/code/session_011xpLSeYKKHX6o16jzYLgZv
This commit is contained in:
2026-09-30 14:55:39 -04:00
co-authored by Claude Opus 5.5
parent 1e2e205ef4
commit 4e450e88ae
2 changed files with 267 additions and 0 deletions
+5
View File
@@ -90,6 +90,11 @@ To re-measure: `unzip -v target/pholio.jar | sort -k3 -n -r | head -20` (third c
- `python3 tools/icons/make-macos-icon.py` — regenerates the macOS icon, `icon.png` and `pholio.icns`
from `src/main/resources/images/spo-1024x1024.png`.
- `python3 tools/demo/fetch-commons-photos.py <dir> [--quality] [--places Paris Kyoto …] [--per-place N]` —
downloads a demo library for screenshots from Wikimedia Commons: freely licensed JPEG originals whose own
EXIF has a capture date and GPS, one folder per city, with a `CREDITS.md` (CC BY / BY-SA require the
credit wherever the photos are shown). `--quality` keeps only Commons' Quality/Featured pictures — nicer,
fewer, slower. `--help` for every option.
## Release
+262
View File
@@ -0,0 +1,262 @@
#!/usr/bin/env python3
"""
Downloads a demo photo library from Wikimedia Commons, for Pholio screenshots.
Only photos that carry their own metadata are kept — the point is a library whose timeline, map, places and
camera details look real: every file must be a JPEG whose EXIF (in the file itself, not just on the Commons
page) has a capture date and GPS coordinates, under a free license (public domain, CC0, CC BY, CC BY-SA).
Photos are found by geographic search around a handful of places, so they are geotagged by construction, and
downloaded as originals: Commons thumbnails lose their EXIF. A CREDITS.md and a credits.csv list author,
license and source page of every file — CC BY / BY-SA require that credit wherever the photos are shown.
No dependency beyond Python 3's standard library. Resumable: files already downloaded are skipped.
Usage:
python3 tools/demo/fetch-commons-photos.py ~/Pictures/pholio-demo # defaults: 10 places x 40 photos
python3 tools/demo/fetch-commons-photos.py ~/Pictures/pholio-demo --per-place 10 --places Paris Kyoto
python3 tools/demo/fetch-commons-photos.py --list-places
"""
import argparse
import csv
import html
import json
import math
import random
import re
import sys
import time
import urllib.error
import urllib.parse
import urllib.request
from pathlib import Path
API = "https://commons.wikimedia.org/w/api.php"
# Wikimedia's API policy asks every client to identify itself with a descriptive User-Agent.
USER_AGENT = "PholioDemoLibraryFetcher/1.0 (https://github.com/Imag-In/Pholio; demo screenshots)"
PLACES = {
"Paris": (48.8566, 2.3522),
"Rome": (41.9028, 12.4964),
"Lisbon": (38.7223, -9.1393),
"Reykjavik": (64.1466, -21.9426),
"Kyoto": (35.0116, 135.7681),
"New York": (40.7580, -73.9855),
"Montreal": (45.5019, -73.5674),
"Marrakech": (31.6295, -7.9811),
"Cape Town": (-33.9249, 18.4241),
"Sydney": (-33.8568, 151.2153),
}
# Commons' own curation labels, awarded by its community after review: technically good, well-composed photos.
QUALITY_CATEGORIES = ["Category:Quality images", "Category:Featured pictures on Wikimedia Commons"]
ALLOWED_LICENSE = re.compile(r"^(public domain|pd|cc0|cc[ -]by([ -]sa)?)", re.IGNORECASE)
def api(**params):
params.update(format="json", formatversion="2", maxlag="5")
url = f"{API}?{urllib.parse.urlencode(params)}"
for attempt in range(5):
try:
request = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
with urllib.request.urlopen(request, timeout=60) as response:
data = json.load(response)
if data.get("error", {}).get("code") == "maxlag":
time.sleep(5 * (attempt + 1))
continue
return data
except (urllib.error.URLError, TimeoutError):
time.sleep(5 * (attempt + 1))
raise RuntimeError(f"Commons API unreachable: {url}")
def plain(text):
"""extmetadata values are HTML fragments; credits want plain text."""
return html.unescape(re.sub(r"<[^>]+>", "", text or "")).strip()
def candidates(lat, lon, radius_m):
"""
File pages around (lat, lon). geosearch returns at most 500 files per query, all packed around the point
asked for, and most of them fail the filters below — so it is asked around the centre and around eight
points on a ring half the radius out, merged without duplicates.
"""
ring_km = radius_m / 2000
dlat = ring_km / 111.0
dlon = ring_km / (111.0 * max(math.cos(math.radians(lat)), 0.1))
points = [(lat, lon)] + [(lat + dy * dlat, lon + dx * dlon)
for dx, dy in [(1, 0), (-1, 0), (0, 1), (0, -1), (0.7, 0.7), (-0.7, 0.7), (0.7, -0.7), (-0.7, -0.7)]]
titles, seen = [], set()
for plat, plon in points:
data = api(action="query", list="geosearch", gscoord=f"{plat}|{plon}", gsradius=str(radius_m),
gsnamespace="6", gslimit="500")
for hit in data.get("query", {}).get("geosearch", []):
if hit["title"] not in seen:
seen.add(hit["title"])
titles.append(hit["title"])
return titles
def details(titles):
"""imageinfo for up to 50 titles: original URL, size, mime, license and the file's own EXIF."""
data = api(action="query", titles="|".join(titles), prop="imageinfo|categories",
iiprop="url|size|mime|extmetadata|commonmetadata", iiextmetadatafilter="LicenseShortName|Artist|LicenseUrl|ImageDescription",
clcategories="|".join(QUALITY_CATEGORIES), cllimit="max")
return data.get("query", {}).get("pages", [])
def distance_m(lat1, lon1, lat2, lon2):
"""Great-circle distance, haversine."""
p1, p2 = math.radians(lat1), math.radians(lat2)
a = math.sin((p2 - p1) / 2) ** 2 + math.cos(p1) * math.cos(p2) * math.sin(math.radians(lon2 - lon1) / 2) ** 2
return 2 * 6371000 * math.asin(math.sqrt(a))
def exif_coordinate(exif, key):
"""commonmetadata gives decimal degrees, signed or with a separate N/S/E/W reference."""
try:
value = float(exif[key])
except (KeyError, TypeError, ValueError):
return None
if str(exif.get(key + "Ref", "")).upper().startswith(("S", "W")) and value > 0:
value = -value
return value
def keepable(page, min_width, max_bytes, center, radius_m, skip_center_m, quality_only):
if quality_only and not page.get("categories"):
return None
info = (page.get("imageinfo") or [None])[0]
if not info or info.get("mime") != "image/jpeg":
return None
if info.get("width", 0) < min_width or info.get("size", 0) > max_bytes:
return None
exif = {entry["name"]: entry["value"] for entry in info.get("commonmetadata", [])}
if "DateTimeOriginal" not in exif:
return None
# The Commons page's own coordinates (what geosearch matched) are sometimes wrong — a photo of Osaka
# filed at Kyoto's centre. What Pholio will read is the file's EXIF, so that is what has to be near.
lat, lon = exif_coordinate(exif, "GPSLatitude"), exif_coordinate(exif, "GPSLongitude")
if lat is None or lon is None:
return None
# A photo geotagged "at the city" rather than where it was taken gets the city's default point — the
# same centre coordinates the places below use (photos of Osaka came back tagged at Kyoto's exact
# centre). Real and placeholder positions can't be told apart there, so that small disc is skipped.
distance = distance_m(lat, lon, *center)
if distance > 1.5 * radius_m or distance < skip_center_m:
return None
meta = info.get("extmetadata", {})
license = plain(meta.get("LicenseShortName", {}).get("value"))
if not ALLOWED_LICENSE.match(license):
return None
return {
"title": page["title"],
"url": info["url"],
"page": info.get("descriptionurl", ""),
"author": plain(meta.get("Artist", {}).get("value")) or "unknown",
"license": license,
"license_url": plain(meta.get("LicenseUrl", {}).get("value")),
"captured": exif["DateTimeOriginal"],
"size": info.get("size", 0),
}
def safe_name(title):
name = title.split(":", 1)[-1]
return re.sub(r'[\\/:*?"<>|]+', "_", name)
def download(url, target):
tmp = target.with_suffix(target.suffix + ".part")
request = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
with urllib.request.urlopen(request, timeout=120) as response, open(tmp, "wb") as out:
while chunk := response.read(1 << 16):
out.write(chunk)
tmp.rename(target)
def main():
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
parser.add_argument("output", nargs="?", type=Path, help="destination folder (one sub-folder per place)")
parser.add_argument("--places", nargs="+", default=list(PLACES), help="places to use (default: all)")
parser.add_argument("--per-place", type=int, default=40, help="photos per place (default: 40)")
parser.add_argument("--radius", type=int, default=10000, help="search radius in metres, max 10000 (default: 10000)")
parser.add_argument("--min-width", type=int, default=2000, help="minimum width in pixels (default: 2000)")
parser.add_argument("--max-mb", type=float, default=15, help="skip originals above this size (default: 15 MB)")
parser.add_argument("--skip-center", type=int, default=200,
help="skip photos geotagged within this many metres of the place's centre, where "
"placeholder 'somewhere in the city' positions land (default: 200, 0 to keep them)")
parser.add_argument("--max-per-author", type=int, default=2,
help="at most this many photos by the same author per place, so one photo series does "
"not fill a whole place (default: 2)")
parser.add_argument("--quality", action="store_true",
help="keep only Commons 'Quality images' and 'Featured pictures' — nicer, but fewer")
parser.add_argument("--list-places", action="store_true", help="print the known places and exit")
args = parser.parse_args()
if args.list_places:
for name, (lat, lon) in PLACES.items():
print(f"{name:12} {lat:9.4f} {lon:9.4f}")
return
if args.output is None:
parser.error("output folder is required")
unknown = [p for p in args.places if p not in PLACES]
if unknown:
parser.error(f"unknown place(s): {', '.join(unknown)} — see --list-places")
args.output.mkdir(parents=True, exist_ok=True)
credits, total_bytes = [], 0
for place in args.places:
lat, lon = PLACES[place]
folder = args.output / place
folder.mkdir(exist_ok=True)
titles = candidates(lat, lon, min(args.radius, 10000))
# Shuffled, seeded by the place name: spread over the whole city rather than the files nearest the
# centre, and the same selection on every run, so an interrupted download resumes where it stopped.
random.Random(place).shuffle(titles)
kept, per_author = 0, {}
print(f"\n{place}: {len(titles)} candidate(s) nearby", flush=True)
for start in range(0, len(titles), 50):
if kept >= args.per_place:
break
for page in details(titles[start:start + 50]):
if kept >= args.per_place:
break
photo = keepable(page, args.min_width, int(args.max_mb * 1024 * 1024), (lat, lon), min(args.radius, 10000), args.skip_center, args.quality)
if photo is None or per_author.get(photo["author"], 0) >= args.max_per_author:
continue
target = folder / safe_name(photo["title"])
if not target.exists():
try:
download(photo["url"], target)
time.sleep(0.5) # be gentle with upload.wikimedia.org
except (urllib.error.URLError, TimeoutError) as error:
print(f" ! {target.name}: {error}", file=sys.stderr)
continue
kept += 1
per_author[photo["author"]] = per_author.get(photo["author"], 0) + 1
total_bytes += target.stat().st_size
credits.append({"place": place, "file": f"{place}/{target.name}", **photo})
print(f" [{kept:3}/{args.per_place}] {target.name} ({photo['captured']}, {photo['license']})", flush=True)
if kept < args.per_place:
print(f" only {kept} photo(s) met the criteria around {place}")
with open(args.output / "credits.csv", "w", newline="", encoding="utf-8") as out:
writer = csv.DictWriter(out, fieldnames=["file", "title", "author", "license", "license_url", "page", "captured"],
extrasaction="ignore")
writer.writeheader()
writer.writerows(credits)
with open(args.output / "CREDITS.md", "w", encoding="utf-8") as out:
out.write("# Photo credits\n\nAll photos from [Wikimedia Commons](https://commons.wikimedia.org).\n\n")
out.write("| File | Author | License | Source |\n|---|---|---|---|\n")
for c in credits:
license = f"[{c['license']}]({c['license_url']})" if c["license_url"] else c["license"]
out.write(f"| {c['file']} | {c['author'].replace('|', '/')} | {license} | [page]({c['page']}) |\n")
print(f"\n{len(credits)} photo(s), {total_bytes / 1048576:.0f} MB in {args.output}")
print(f"Credits: {args.output / 'CREDITS.md'} and credits.csv")
if __name__ == "__main__":
main()