diff --git a/tools/anki-deck-periodic.py b/tools/anki-deck-periodic.py index 25ccdae..c4216e7 100644 --- a/tools/anki-deck-periodic.py +++ b/tools/anki-deck-periodic.py @@ -38,13 +38,22 @@ the planet, for one) — TITLE_OVERRIDES below is the fix-up list; if a generated card shows an obviously wrong photo for some element, that's almost certainly another one of these collisions — add it there. -Caveat: this fetch logic was written and dry-run tested without live -access to Wikipedia (sandboxed here with no route to en.wikipedia.org), -so it's exercised structurally (see --dry-run-tts) but not against the -real API response shape or real image content. Skips are per-element and -non-fatal — one bad/missing photo won't abort the rest of the deck — and -every fetch attempt prints what it did, so check that output the first -time you actually run this deck for real. +Batched and rate-limited on purpose: metadata lookups (which element has +which photo) go out up to 50 titles per request, not one request per +element, and every HTTP call retries with backoff on a 429 (honoring +Wikipedia's own Retry-After header when it sends one). The first version +of this fetched one element at a time with no delay between requests and +got rate-limited by Wikimedia on a real run — this replaced it. + +Caveat: this was written and tested without live access to Wikipedia +(sandboxed here with no route to en.wikipedia.org) — exercised +structurally (see --dry-run-tts), the redirect/normalization-chain +resolution logic and the 429-retry path both unit-tested against +simulated responses, but never against the real API's actual response +shape or real image content. Skips are per-element and non-fatal — one +bad/missing photo won't abort the rest of the deck — and every fetch +attempt prints what it did, so check that output the first time you +actually run this deck for real. Element data: Bowserinator/Periodic-Table-JSON (a widely used, actively maintained public dataset), fetched and spot-checked against known facts @@ -69,6 +78,7 @@ import json import os import random import subprocess +import time import urllib.error import urllib.parse import urllib.request @@ -97,49 +107,139 @@ NO_PHOTO_NUMBERS = set(range(100, 119)) TITLE_OVERRIDES = {"Hg": "Mercury (element)"} -def fetch_element_image(symbol, name): - """Downloads (once, cached under periodic_images/) the real sample - photo Wikipedia itself shows for this element — its page's - 'pageimage', the same image used in the infobox — via the standard - MediaWiki query API. Returns the cached local path, or None if no - image exists on the page (normal for some elements, not a bug).""" - cache_path = os.path.join(IMAGE_CACHE, f"{symbol}.jpg") - if os.path.isfile(cache_path): - return cache_path +USER_AGENT = "anki-deck-periodic.py/1.0 (personal Anki deck generator, run locally by its owner)" + + +def http_get_with_retry(url, max_retries=5): + """GET with retry-on-429: honors a numeric Retry-After header if + Wikipedia sends one, otherwise backs off 5s * attempt. Returns the raw + response bytes, or None if every attempt failed — never raises, since + one element's fetch failing must not abort the whole deck build.""" + headers = {"User-Agent": USER_AGENT} + for attempt in range(1, max_retries + 1): + try: + req = urllib.request.Request(url, headers=headers) + with urllib.request.urlopen(req, timeout=20) as resp: + return resp.read() + except urllib.error.HTTPError as e: + if e.code == 429 and attempt < max_retries: + wait = 5 * attempt + retry_after = e.headers.get("Retry-After") if e.headers else None + if retry_after and retry_after.isdigit(): + wait = int(retry_after) + print(f" (rate limited, waiting {wait}s before retry {attempt}/{max_retries})") + time.sleep(wait) + continue + print(f" (request failed: {e})") + return None + except (urllib.error.URLError, OSError, ValueError) as e: + print(f" (request failed: {e})") + return None + return None + + +def resolve_pageimage_urls(pairs): + """pairs: list of (symbol, wikipedia_title). Batches lookups — up to + 50 titles per MediaWiki query, the documented anonymous-access limit — + instead of one API call per element; a tight one-request-per-element + loop is exactly what triggered Wikipedia's rate limiting on a real + run. Requests thumbnails (piprop=thumbnail), not full-resolution + originals, per Wikimedia's own guidance on their 429 response. Returns + {symbol: thumbnail_url} for whichever elements actually have one.""" + result = {} + CHUNK = 50 + for i in range(0, len(pairs), CHUNK): + chunk = pairs[i:i + CHUNK] + symbol_by_title = {title: symbol for symbol, title in chunk} + titles_param = "|".join(title for _, title in chunk) + api_url = ("https://en.wikipedia.org/w/api.php?action=query&format=json" + "&prop=pageimages&piprop=thumbnail&pithumbsize=300&redirects=1&titles=" + + urllib.parse.quote(titles_param)) + raw = http_get_with_retry(api_url) + if raw is None: + continue + data = json.loads(raw) + query = data.get("query", {}) + + # "pages" below is keyed by pageid with only the *final* resolved + # title on it, so build a title-at-this-point -> original-input- + # title map and walk it forward through each normalized/redirect + # step, re-keying as the title changes, to reach the same result. + input_of_title = {title: title for _, title in chunk} + for step in query.get("normalized", []) + query.get("redirects", []): + frm, to = step["from"], step["to"] + if frm in input_of_title: + input_of_title[to] = input_of_title.pop(frm) + + for page in query.get("pages", {}).values(): + final_title = page.get("title") + input_title = input_of_title.get(final_title, final_title) + symbol = symbol_by_title.get(input_title) + if symbol is None: + continue + thumb_url = page.get("thumbnail", {}).get("source") + if thumb_url: + result[symbol] = thumb_url + if i + CHUNK < len(pairs): + time.sleep(1) # be polite between batches + return result + + +def download_element_photos(url_map): + """url_map: {symbol: thumbnail_url}. Downloads each into + periodic_images/, one request at a time with a short gap between — + the metadata lookups above are batched, but the actual image bytes + still need one HTTP request per element, and Wikimedia's upload + servers rate-limit that too if hit back-to-back with no gap.""" + for symbol, url in url_map.items(): + cache_path = os.path.join(IMAGE_CACHE, f"{symbol}.jpg") + if os.path.isfile(cache_path): + continue + raw = http_get_with_retry(url) + if raw is None: + print(f" (photo download failed for {symbol})") + continue + try: + img = Image.open(io.BytesIO(raw)).convert("RGB") + img.thumbnail((300, 300)) + img.save(cache_path, "JPEG", quality=85) + print(f" fetched photo for {symbol}") + except OSError as e: + print(f" (photo decode failed for {symbol}: {e})") + time.sleep(0.5) + + +def ensure_element_photos(elements): + """Call once per deck, before building any notes: makes sure every + eligible element (not --no-images, not NO_PHOTO_NUMBERS, not already + cached) has its photo downloaded into periodic_images/ up front — + batched and rate-limited, rather than the old one-request-per-card + approach that got 429'd on a real run. element_image_html() below then + only ever reads the cache; it makes no network calls itself.""" + if args.no_images: + return + needed = [(symbol, name) for number, symbol, name, _cat in elements + if number not in NO_PHOTO_NUMBERS + and not os.path.isfile(os.path.join(IMAGE_CACHE, f"{symbol}.jpg"))] + if not needed: + return if args.dry_run_tts: # No real network call in dry-run mode — a flat placeholder lets the # rest of the pipeline (HTML wiring, media_files list, .apkg # packaging) still be exercised end-to-end without it. - Image.new("RGB", (100, 100), (190, 190, 190)).save(cache_path, "JPEG") - return cache_path + for symbol, _name in needed: + cache_path = os.path.join(IMAGE_CACHE, f"{symbol}.jpg") + Image.new("RGB", (100, 100), (190, 190, 190)).save(cache_path, "JPEG") + return - title = TITLE_OVERRIDES.get(symbol, name) - api_url = ("https://en.wikipedia.org/w/api.php?action=query&format=json" - "&prop=pageimages&piprop=original&redirects=1&titles=" - + urllib.parse.quote(title)) - headers = {"User-Agent": "anki-deck-periodic.py (personal Anki deck generator)"} - try: - req = urllib.request.Request(api_url, headers=headers) - with urllib.request.urlopen(req, timeout=15) as resp: - data = json.load(resp) - pages = data.get("query", {}).get("pages", {}) - page = next(iter(pages.values()), {}) - image_url = page.get("original", {}).get("source") - if not image_url: - print(f" (no photo found for {name})") - return None - img_req = urllib.request.Request(image_url, headers=headers) - with urllib.request.urlopen(img_req, timeout=15) as resp: - raw = resp.read() - img = Image.open(io.BytesIO(raw)).convert("RGB") - img.thumbnail((300, 300)) - img.save(cache_path, "JPEG", quality=85) - print(f" fetched photo for {name} ({title})") - return cache_path - except (urllib.error.URLError, urllib.error.HTTPError, OSError, ValueError) as e: - print(f" (photo fetch failed for {name}: {e})") - return None + pairs = [(symbol, TITLE_OVERRIDES.get(symbol, name)) for symbol, name in needed] + print(f"Fetching {len(pairs)} element photo(s) from Wikipedia (batched, rate-limited)...") + url_map = resolve_pageimage_urls(pairs) + for symbol, _title in pairs: + if symbol not in url_map: + print(f" (no photo found for {symbol})") + download_element_photos(url_map) _CANDIDATES = [ args.model_path, @@ -365,18 +465,20 @@ def build_deck(deck_key, deck_title): return genanki.Deck(deck_id, deck_title) -def element_image_html(number, symbol, name, media_files): - """Returns an tag for this element's cached/fetched photo, or "" - if --no-images was passed, the element is in NO_PHOTO_NUMBERS, or no - photo could be found/downloaded for it.""" +def element_image_html(number, symbol, media_files): + """Returns an tag for this element's cached photo, or "" if + --no-images was passed, the element is in NO_PHOTO_NUMBERS, or no + photo was found/downloaded for it. Reads the cache only — every + network call happens up front in ensure_element_photos(), once per + deck, not per card.""" if args.no_images or number in NO_PHOTO_NUMBERS: return "" - local_path = fetch_element_image(symbol, name) - if local_path is None: + cache_path = os.path.join(IMAGE_CACHE, f"{symbol}.jpg") + if not os.path.isfile(cache_path): return "" - if local_path not in media_files: - media_files.append(local_path) - return f'' + if cache_path not in media_files: + media_files.append(cache_path) + return f'' def add_note(deck, model, prompt, answer, qtext, atext, media_files, tag): @@ -399,6 +501,7 @@ def gen_prehs(): deck = build_deck(deck_key, "Periodic Table: Symbols & Names (1-36)") media_files = [] subset = [e for e in ELEMENTS if e[0] <= 36] + ensure_element_photos(subset) cards = [] for number, symbol, name, category in subset: cards.append(("symbol_to_name", number, symbol, name)) @@ -406,7 +509,7 @@ def gen_prehs(): random.seed(50) random.shuffle(cards) for kind, number, symbol, name in cards: - img = element_image_html(number, symbol, name, media_files) + img = element_image_html(number, symbol, media_files) if kind == "symbol_to_name": add_note(deck, model, img + symbol, name, f"What element has the symbol {symbol}?", name, @@ -423,6 +526,7 @@ def gen_hs(): model_id, model = build_model(deck_key) deck = build_deck(deck_key, "Periodic Table: Symbols, Names & Numbers (1-118)") media_files = [] + ensure_element_photos(ELEMENTS) cards = [] for number, symbol, name, category in ELEMENTS: cards.append(("symbol_to_name", number, symbol, name)) @@ -431,7 +535,7 @@ def gen_hs(): random.seed(51) random.shuffle(cards) for kind, number, symbol, name in cards: - img = element_image_html(number, symbol, name, media_files) + img = element_image_html(number, symbol, media_files) if kind == "symbol_to_name": add_note(deck, model, img + symbol, name, f"What element has the symbol {symbol}?", name,