feat(tools): add a Wikimedia Commons demo library downloader
tools/demo/fetch-commons-photos.py builds a screenshot library from Wikimedia Commons: geosearch around ten cities (a ring of nine points, since each query returns at most 500 files), then only freely licensed JPEG originals whose own EXIF has a capture date and GPS within range, skipping placeholder "city centre" positions, at most two per author per city, with an optional filter on Commons' Quality/Featured pictures. Writes CREDITS.md and credits.csv, resumes interrupted runs, standard library only. Listed in the README's useful scripts. Co-Authored-By: Claude Opus 5.5 <noreply@anthropic.com> Claude-Session: https://claude.ai/code/session_011xpLSeYKKHX6o16jzYLgZv
This commit is contained in:
@@ -90,6 +90,11 @@ To re-measure: `unzip -v target/pholio.jar | sort -k3 -n -r | head -20` (third c
|
||||
|
||||
- `python3 tools/icons/make-macos-icon.py` — regenerates the macOS icon, `icon.png` and `pholio.icns`
|
||||
from `src/main/resources/images/spo-1024x1024.png`.
|
||||
- `python3 tools/demo/fetch-commons-photos.py <dir> [--quality] [--places Paris Kyoto …] [--per-place N]` —
|
||||
downloads a demo library for screenshots from Wikimedia Commons: freely licensed JPEG originals whose own
|
||||
EXIF has a capture date and GPS, one folder per city, with a `CREDITS.md` (CC BY / BY-SA require the
|
||||
credit wherever the photos are shown). `--quality` keeps only Commons' Quality/Featured pictures — nicer,
|
||||
fewer, slower. `--help` for every option.
|
||||
|
||||
## Release
|
||||
|
||||
|
||||
@@ -0,0 +1,262 @@
|
||||
#!/usr/bin/env python3
|
||||
"""
|
||||
Downloads a demo photo library from Wikimedia Commons, for Pholio screenshots.
|
||||
|
||||
Only photos that carry their own metadata are kept — the point is a library whose timeline, map, places and
|
||||
camera details look real: every file must be a JPEG whose EXIF (in the file itself, not just on the Commons
|
||||
page) has a capture date and GPS coordinates, under a free license (public domain, CC0, CC BY, CC BY-SA).
|
||||
|
||||
Photos are found by geographic search around a handful of places, so they are geotagged by construction, and
|
||||
downloaded as originals: Commons thumbnails lose their EXIF. A CREDITS.md and a credits.csv list author,
|
||||
license and source page of every file — CC BY / BY-SA require that credit wherever the photos are shown.
|
||||
|
||||
No dependency beyond Python 3's standard library. Resumable: files already downloaded are skipped.
|
||||
|
||||
Usage:
|
||||
python3 tools/demo/fetch-commons-photos.py ~/Pictures/pholio-demo # defaults: 10 places x 40 photos
|
||||
python3 tools/demo/fetch-commons-photos.py ~/Pictures/pholio-demo --per-place 10 --places Paris Kyoto
|
||||
python3 tools/demo/fetch-commons-photos.py --list-places
|
||||
"""
|
||||
import argparse
|
||||
import csv
|
||||
import html
|
||||
import json
|
||||
import math
|
||||
import random
|
||||
import re
|
||||
import sys
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
from pathlib import Path
|
||||
|
||||
API = "https://commons.wikimedia.org/w/api.php"
|
||||
# Wikimedia's API policy asks every client to identify itself with a descriptive User-Agent.
|
||||
USER_AGENT = "PholioDemoLibraryFetcher/1.0 (https://github.com/Imag-In/Pholio; demo screenshots)"
|
||||
|
||||
PLACES = {
|
||||
"Paris": (48.8566, 2.3522),
|
||||
"Rome": (41.9028, 12.4964),
|
||||
"Lisbon": (38.7223, -9.1393),
|
||||
"Reykjavik": (64.1466, -21.9426),
|
||||
"Kyoto": (35.0116, 135.7681),
|
||||
"New York": (40.7580, -73.9855),
|
||||
"Montreal": (45.5019, -73.5674),
|
||||
"Marrakech": (31.6295, -7.9811),
|
||||
"Cape Town": (-33.9249, 18.4241),
|
||||
"Sydney": (-33.8568, 151.2153),
|
||||
}
|
||||
|
||||
# Commons' own curation labels, awarded by its community after review: technically good, well-composed photos.
|
||||
QUALITY_CATEGORIES = ["Category:Quality images", "Category:Featured pictures on Wikimedia Commons"]
|
||||
|
||||
ALLOWED_LICENSE = re.compile(r"^(public domain|pd|cc0|cc[ -]by([ -]sa)?)", re.IGNORECASE)
|
||||
|
||||
|
||||
def api(**params):
|
||||
params.update(format="json", formatversion="2", maxlag="5")
|
||||
url = f"{API}?{urllib.parse.urlencode(params)}"
|
||||
for attempt in range(5):
|
||||
try:
|
||||
request = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
|
||||
with urllib.request.urlopen(request, timeout=60) as response:
|
||||
data = json.load(response)
|
||||
if data.get("error", {}).get("code") == "maxlag":
|
||||
time.sleep(5 * (attempt + 1))
|
||||
continue
|
||||
return data
|
||||
except (urllib.error.URLError, TimeoutError):
|
||||
time.sleep(5 * (attempt + 1))
|
||||
raise RuntimeError(f"Commons API unreachable: {url}")
|
||||
|
||||
|
||||
def plain(text):
|
||||
"""extmetadata values are HTML fragments; credits want plain text."""
|
||||
return html.unescape(re.sub(r"<[^>]+>", "", text or "")).strip()
|
||||
|
||||
|
||||
def candidates(lat, lon, radius_m):
|
||||
"""
|
||||
File pages around (lat, lon). geosearch returns at most 500 files per query, all packed around the point
|
||||
asked for, and most of them fail the filters below — so it is asked around the centre and around eight
|
||||
points on a ring half the radius out, merged without duplicates.
|
||||
"""
|
||||
ring_km = radius_m / 2000
|
||||
dlat = ring_km / 111.0
|
||||
dlon = ring_km / (111.0 * max(math.cos(math.radians(lat)), 0.1))
|
||||
points = [(lat, lon)] + [(lat + dy * dlat, lon + dx * dlon)
|
||||
for dx, dy in [(1, 0), (-1, 0), (0, 1), (0, -1), (0.7, 0.7), (-0.7, 0.7), (0.7, -0.7), (-0.7, -0.7)]]
|
||||
titles, seen = [], set()
|
||||
for plat, plon in points:
|
||||
data = api(action="query", list="geosearch", gscoord=f"{plat}|{plon}", gsradius=str(radius_m),
|
||||
gsnamespace="6", gslimit="500")
|
||||
for hit in data.get("query", {}).get("geosearch", []):
|
||||
if hit["title"] not in seen:
|
||||
seen.add(hit["title"])
|
||||
titles.append(hit["title"])
|
||||
return titles
|
||||
|
||||
|
||||
def details(titles):
|
||||
"""imageinfo for up to 50 titles: original URL, size, mime, license and the file's own EXIF."""
|
||||
data = api(action="query", titles="|".join(titles), prop="imageinfo|categories",
|
||||
iiprop="url|size|mime|extmetadata|commonmetadata", iiextmetadatafilter="LicenseShortName|Artist|LicenseUrl|ImageDescription",
|
||||
clcategories="|".join(QUALITY_CATEGORIES), cllimit="max")
|
||||
return data.get("query", {}).get("pages", [])
|
||||
|
||||
|
||||
def distance_m(lat1, lon1, lat2, lon2):
|
||||
"""Great-circle distance, haversine."""
|
||||
p1, p2 = math.radians(lat1), math.radians(lat2)
|
||||
a = math.sin((p2 - p1) / 2) ** 2 + math.cos(p1) * math.cos(p2) * math.sin(math.radians(lon2 - lon1) / 2) ** 2
|
||||
return 2 * 6371000 * math.asin(math.sqrt(a))
|
||||
|
||||
|
||||
def exif_coordinate(exif, key):
|
||||
"""commonmetadata gives decimal degrees, signed or with a separate N/S/E/W reference."""
|
||||
try:
|
||||
value = float(exif[key])
|
||||
except (KeyError, TypeError, ValueError):
|
||||
return None
|
||||
if str(exif.get(key + "Ref", "")).upper().startswith(("S", "W")) and value > 0:
|
||||
value = -value
|
||||
return value
|
||||
|
||||
|
||||
def keepable(page, min_width, max_bytes, center, radius_m, skip_center_m, quality_only):
|
||||
if quality_only and not page.get("categories"):
|
||||
return None
|
||||
info = (page.get("imageinfo") or [None])[0]
|
||||
if not info or info.get("mime") != "image/jpeg":
|
||||
return None
|
||||
if info.get("width", 0) < min_width or info.get("size", 0) > max_bytes:
|
||||
return None
|
||||
exif = {entry["name"]: entry["value"] for entry in info.get("commonmetadata", [])}
|
||||
if "DateTimeOriginal" not in exif:
|
||||
return None
|
||||
# The Commons page's own coordinates (what geosearch matched) are sometimes wrong — a photo of Osaka
|
||||
# filed at Kyoto's centre. What Pholio will read is the file's EXIF, so that is what has to be near.
|
||||
lat, lon = exif_coordinate(exif, "GPSLatitude"), exif_coordinate(exif, "GPSLongitude")
|
||||
if lat is None or lon is None:
|
||||
return None
|
||||
# A photo geotagged "at the city" rather than where it was taken gets the city's default point — the
|
||||
# same centre coordinates the places below use (photos of Osaka came back tagged at Kyoto's exact
|
||||
# centre). Real and placeholder positions can't be told apart there, so that small disc is skipped.
|
||||
distance = distance_m(lat, lon, *center)
|
||||
if distance > 1.5 * radius_m or distance < skip_center_m:
|
||||
return None
|
||||
meta = info.get("extmetadata", {})
|
||||
license = plain(meta.get("LicenseShortName", {}).get("value"))
|
||||
if not ALLOWED_LICENSE.match(license):
|
||||
return None
|
||||
return {
|
||||
"title": page["title"],
|
||||
"url": info["url"],
|
||||
"page": info.get("descriptionurl", ""),
|
||||
"author": plain(meta.get("Artist", {}).get("value")) or "unknown",
|
||||
"license": license,
|
||||
"license_url": plain(meta.get("LicenseUrl", {}).get("value")),
|
||||
"captured": exif["DateTimeOriginal"],
|
||||
"size": info.get("size", 0),
|
||||
}
|
||||
|
||||
|
||||
def safe_name(title):
|
||||
name = title.split(":", 1)[-1]
|
||||
return re.sub(r'[\\/:*?"<>|]+', "_", name)
|
||||
|
||||
|
||||
def download(url, target):
|
||||
tmp = target.with_suffix(target.suffix + ".part")
|
||||
request = urllib.request.Request(url, headers={"User-Agent": USER_AGENT})
|
||||
with urllib.request.urlopen(request, timeout=120) as response, open(tmp, "wb") as out:
|
||||
while chunk := response.read(1 << 16):
|
||||
out.write(chunk)
|
||||
tmp.rename(target)
|
||||
|
||||
|
||||
def main():
|
||||
parser = argparse.ArgumentParser(description=__doc__, formatter_class=argparse.RawDescriptionHelpFormatter)
|
||||
parser.add_argument("output", nargs="?", type=Path, help="destination folder (one sub-folder per place)")
|
||||
parser.add_argument("--places", nargs="+", default=list(PLACES), help="places to use (default: all)")
|
||||
parser.add_argument("--per-place", type=int, default=40, help="photos per place (default: 40)")
|
||||
parser.add_argument("--radius", type=int, default=10000, help="search radius in metres, max 10000 (default: 10000)")
|
||||
parser.add_argument("--min-width", type=int, default=2000, help="minimum width in pixels (default: 2000)")
|
||||
parser.add_argument("--max-mb", type=float, default=15, help="skip originals above this size (default: 15 MB)")
|
||||
parser.add_argument("--skip-center", type=int, default=200,
|
||||
help="skip photos geotagged within this many metres of the place's centre, where "
|
||||
"placeholder 'somewhere in the city' positions land (default: 200, 0 to keep them)")
|
||||
parser.add_argument("--max-per-author", type=int, default=2,
|
||||
help="at most this many photos by the same author per place, so one photo series does "
|
||||
"not fill a whole place (default: 2)")
|
||||
parser.add_argument("--quality", action="store_true",
|
||||
help="keep only Commons 'Quality images' and 'Featured pictures' — nicer, but fewer")
|
||||
parser.add_argument("--list-places", action="store_true", help="print the known places and exit")
|
||||
args = parser.parse_args()
|
||||
|
||||
if args.list_places:
|
||||
for name, (lat, lon) in PLACES.items():
|
||||
print(f"{name:12} {lat:9.4f} {lon:9.4f}")
|
||||
return
|
||||
if args.output is None:
|
||||
parser.error("output folder is required")
|
||||
unknown = [p for p in args.places if p not in PLACES]
|
||||
if unknown:
|
||||
parser.error(f"unknown place(s): {', '.join(unknown)} — see --list-places")
|
||||
|
||||
args.output.mkdir(parents=True, exist_ok=True)
|
||||
credits, total_bytes = [], 0
|
||||
for place in args.places:
|
||||
lat, lon = PLACES[place]
|
||||
folder = args.output / place
|
||||
folder.mkdir(exist_ok=True)
|
||||
titles = candidates(lat, lon, min(args.radius, 10000))
|
||||
# Shuffled, seeded by the place name: spread over the whole city rather than the files nearest the
|
||||
# centre, and the same selection on every run, so an interrupted download resumes where it stopped.
|
||||
random.Random(place).shuffle(titles)
|
||||
kept, per_author = 0, {}
|
||||
print(f"\n{place}: {len(titles)} candidate(s) nearby", flush=True)
|
||||
for start in range(0, len(titles), 50):
|
||||
if kept >= args.per_place:
|
||||
break
|
||||
for page in details(titles[start:start + 50]):
|
||||
if kept >= args.per_place:
|
||||
break
|
||||
photo = keepable(page, args.min_width, int(args.max_mb * 1024 * 1024), (lat, lon), min(args.radius, 10000), args.skip_center, args.quality)
|
||||
if photo is None or per_author.get(photo["author"], 0) >= args.max_per_author:
|
||||
continue
|
||||
target = folder / safe_name(photo["title"])
|
||||
if not target.exists():
|
||||
try:
|
||||
download(photo["url"], target)
|
||||
time.sleep(0.5) # be gentle with upload.wikimedia.org
|
||||
except (urllib.error.URLError, TimeoutError) as error:
|
||||
print(f" ! {target.name}: {error}", file=sys.stderr)
|
||||
continue
|
||||
kept += 1
|
||||
per_author[photo["author"]] = per_author.get(photo["author"], 0) + 1
|
||||
total_bytes += target.stat().st_size
|
||||
credits.append({"place": place, "file": f"{place}/{target.name}", **photo})
|
||||
print(f" [{kept:3}/{args.per_place}] {target.name} ({photo['captured']}, {photo['license']})", flush=True)
|
||||
if kept < args.per_place:
|
||||
print(f" only {kept} photo(s) met the criteria around {place}")
|
||||
|
||||
with open(args.output / "credits.csv", "w", newline="", encoding="utf-8") as out:
|
||||
writer = csv.DictWriter(out, fieldnames=["file", "title", "author", "license", "license_url", "page", "captured"],
|
||||
extrasaction="ignore")
|
||||
writer.writeheader()
|
||||
writer.writerows(credits)
|
||||
with open(args.output / "CREDITS.md", "w", encoding="utf-8") as out:
|
||||
out.write("# Photo credits\n\nAll photos from [Wikimedia Commons](https://commons.wikimedia.org).\n\n")
|
||||
out.write("| File | Author | License | Source |\n|---|---|---|---|\n")
|
||||
for c in credits:
|
||||
license = f"[{c['license']}]({c['license_url']})" if c["license_url"] else c["license"]
|
||||
out.write(f"| {c['file']} | {c['author'].replace('|', '/')} | {license} | [page]({c['page']}) |\n")
|
||||
|
||||
print(f"\n{len(credits)} photo(s), {total_bytes / 1048576:.0f} MB in {args.output}")
|
||||
print(f"Credits: {args.output / 'CREDITS.md'} and credits.csv")
|
||||
|
||||
|
||||
if __name__ == "__main__":
|
||||
main()
|
||||
Reference in New Issue
Block a user