Merge pull request #456 from outis1one/claude/pensive-hopper-4c9e7i
anki-deck-periodic.py: fix element-photo fetching hitting Wikipedia's…
This commit is contained in:
+159
-55
@@ -38,13 +38,22 @@ the planet, for one) — TITLE_OVERRIDES below is the fix-up list; if a
|
||||
generated card shows an obviously wrong photo for some element, that's
|
||||
almost certainly another one of these collisions — add it there.
|
||||
|
||||
Caveat: this fetch logic was written and dry-run tested without live
|
||||
access to Wikipedia (sandboxed here with no route to en.wikipedia.org),
|
||||
so it's exercised structurally (see --dry-run-tts) but not against the
|
||||
real API response shape or real image content. Skips are per-element and
|
||||
non-fatal — one bad/missing photo won't abort the rest of the deck — and
|
||||
every fetch attempt prints what it did, so check that output the first
|
||||
time you actually run this deck for real.
|
||||
Batched and rate-limited on purpose: metadata lookups (which element has
|
||||
which photo) go out up to 50 titles per request, not one request per
|
||||
element, and every HTTP call retries with backoff on a 429 (honoring
|
||||
Wikipedia's own Retry-After header when it sends one). The first version
|
||||
of this fetched one element at a time with no delay between requests and
|
||||
got rate-limited by Wikimedia on a real run — this replaced it.
|
||||
|
||||
Caveat: this was written and tested without live access to Wikipedia
|
||||
(sandboxed here with no route to en.wikipedia.org) — exercised
|
||||
structurally (see --dry-run-tts), the redirect/normalization-chain
|
||||
resolution logic and the 429-retry path both unit-tested against
|
||||
simulated responses, but never against the real API's actual response
|
||||
shape or real image content. Skips are per-element and non-fatal — one
|
||||
bad/missing photo won't abort the rest of the deck — and every fetch
|
||||
attempt prints what it did, so check that output the first time you
|
||||
actually run this deck for real.
|
||||
|
||||
Element data: Bowserinator/Periodic-Table-JSON (a widely used, actively
|
||||
maintained public dataset), fetched and spot-checked against known facts
|
||||
@@ -69,6 +78,7 @@ import json
|
||||
import os
|
||||
import random
|
||||
import subprocess
|
||||
import time
|
||||
import urllib.error
|
||||
import urllib.parse
|
||||
import urllib.request
|
||||
@@ -97,49 +107,139 @@ NO_PHOTO_NUMBERS = set(range(100, 119))
|
||||
TITLE_OVERRIDES = {"Hg": "Mercury (element)"}
|
||||
|
||||
|
||||
def fetch_element_image(symbol, name):
|
||||
"""Downloads (once, cached under periodic_images/) the real sample
|
||||
photo Wikipedia itself shows for this element — its page's
|
||||
'pageimage', the same image used in the infobox — via the standard
|
||||
MediaWiki query API. Returns the cached local path, or None if no
|
||||
image exists on the page (normal for some elements, not a bug)."""
|
||||
cache_path = os.path.join(IMAGE_CACHE, f"{symbol}.jpg")
|
||||
if os.path.isfile(cache_path):
|
||||
return cache_path
|
||||
USER_AGENT = "anki-deck-periodic.py/1.0 (personal Anki deck generator, run locally by its owner)"
|
||||
|
||||
|
||||
def http_get_with_retry(url, max_retries=5):
|
||||
"""GET with retry-on-429: honors a numeric Retry-After header if
|
||||
Wikipedia sends one, otherwise backs off 5s * attempt. Returns the raw
|
||||
response bytes, or None if every attempt failed — never raises, since
|
||||
one element's fetch failing must not abort the whole deck build."""
|
||||
headers = {"User-Agent": USER_AGENT}
|
||||
for attempt in range(1, max_retries + 1):
|
||||
try:
|
||||
req = urllib.request.Request(url, headers=headers)
|
||||
with urllib.request.urlopen(req, timeout=20) as resp:
|
||||
return resp.read()
|
||||
except urllib.error.HTTPError as e:
|
||||
if e.code == 429 and attempt < max_retries:
|
||||
wait = 5 * attempt
|
||||
retry_after = e.headers.get("Retry-After") if e.headers else None
|
||||
if retry_after and retry_after.isdigit():
|
||||
wait = int(retry_after)
|
||||
print(f" (rate limited, waiting {wait}s before retry {attempt}/{max_retries})")
|
||||
time.sleep(wait)
|
||||
continue
|
||||
print(f" (request failed: {e})")
|
||||
return None
|
||||
except (urllib.error.URLError, OSError, ValueError) as e:
|
||||
print(f" (request failed: {e})")
|
||||
return None
|
||||
return None
|
||||
|
||||
|
||||
def resolve_pageimage_urls(pairs):
|
||||
"""pairs: list of (symbol, wikipedia_title). Batches lookups — up to
|
||||
50 titles per MediaWiki query, the documented anonymous-access limit —
|
||||
instead of one API call per element; a tight one-request-per-element
|
||||
loop is exactly what triggered Wikipedia's rate limiting on a real
|
||||
run. Requests thumbnails (piprop=thumbnail), not full-resolution
|
||||
originals, per Wikimedia's own guidance on their 429 response. Returns
|
||||
{symbol: thumbnail_url} for whichever elements actually have one."""
|
||||
result = {}
|
||||
CHUNK = 50
|
||||
for i in range(0, len(pairs), CHUNK):
|
||||
chunk = pairs[i:i + CHUNK]
|
||||
symbol_by_title = {title: symbol for symbol, title in chunk}
|
||||
titles_param = "|".join(title for _, title in chunk)
|
||||
api_url = ("https://en.wikipedia.org/w/api.php?action=query&format=json"
|
||||
"&prop=pageimages&piprop=thumbnail&pithumbsize=300&redirects=1&titles="
|
||||
+ urllib.parse.quote(titles_param))
|
||||
raw = http_get_with_retry(api_url)
|
||||
if raw is None:
|
||||
continue
|
||||
data = json.loads(raw)
|
||||
query = data.get("query", {})
|
||||
|
||||
# "pages" below is keyed by pageid with only the *final* resolved
|
||||
# title on it, so build a title-at-this-point -> original-input-
|
||||
# title map and walk it forward through each normalized/redirect
|
||||
# step, re-keying as the title changes, to reach the same result.
|
||||
input_of_title = {title: title for _, title in chunk}
|
||||
for step in query.get("normalized", []) + query.get("redirects", []):
|
||||
frm, to = step["from"], step["to"]
|
||||
if frm in input_of_title:
|
||||
input_of_title[to] = input_of_title.pop(frm)
|
||||
|
||||
for page in query.get("pages", {}).values():
|
||||
final_title = page.get("title")
|
||||
input_title = input_of_title.get(final_title, final_title)
|
||||
symbol = symbol_by_title.get(input_title)
|
||||
if symbol is None:
|
||||
continue
|
||||
thumb_url = page.get("thumbnail", {}).get("source")
|
||||
if thumb_url:
|
||||
result[symbol] = thumb_url
|
||||
if i + CHUNK < len(pairs):
|
||||
time.sleep(1) # be polite between batches
|
||||
return result
|
||||
|
||||
|
||||
def download_element_photos(url_map):
|
||||
"""url_map: {symbol: thumbnail_url}. Downloads each into
|
||||
periodic_images/, one request at a time with a short gap between —
|
||||
the metadata lookups above are batched, but the actual image bytes
|
||||
still need one HTTP request per element, and Wikimedia's upload
|
||||
servers rate-limit that too if hit back-to-back with no gap."""
|
||||
for symbol, url in url_map.items():
|
||||
cache_path = os.path.join(IMAGE_CACHE, f"{symbol}.jpg")
|
||||
if os.path.isfile(cache_path):
|
||||
continue
|
||||
raw = http_get_with_retry(url)
|
||||
if raw is None:
|
||||
print(f" (photo download failed for {symbol})")
|
||||
continue
|
||||
try:
|
||||
img = Image.open(io.BytesIO(raw)).convert("RGB")
|
||||
img.thumbnail((300, 300))
|
||||
img.save(cache_path, "JPEG", quality=85)
|
||||
print(f" fetched photo for {symbol}")
|
||||
except OSError as e:
|
||||
print(f" (photo decode failed for {symbol}: {e})")
|
||||
time.sleep(0.5)
|
||||
|
||||
|
||||
def ensure_element_photos(elements):
|
||||
"""Call once per deck, before building any notes: makes sure every
|
||||
eligible element (not --no-images, not NO_PHOTO_NUMBERS, not already
|
||||
cached) has its photo downloaded into periodic_images/ up front —
|
||||
batched and rate-limited, rather than the old one-request-per-card
|
||||
approach that got 429'd on a real run. element_image_html() below then
|
||||
only ever reads the cache; it makes no network calls itself."""
|
||||
if args.no_images:
|
||||
return
|
||||
needed = [(symbol, name) for number, symbol, name, _cat in elements
|
||||
if number not in NO_PHOTO_NUMBERS
|
||||
and not os.path.isfile(os.path.join(IMAGE_CACHE, f"{symbol}.jpg"))]
|
||||
if not needed:
|
||||
return
|
||||
|
||||
if args.dry_run_tts:
|
||||
# No real network call in dry-run mode — a flat placeholder lets the
|
||||
# rest of the pipeline (HTML wiring, media_files list, .apkg
|
||||
# packaging) still be exercised end-to-end without it.
|
||||
Image.new("RGB", (100, 100), (190, 190, 190)).save(cache_path, "JPEG")
|
||||
return cache_path
|
||||
for symbol, _name in needed:
|
||||
cache_path = os.path.join(IMAGE_CACHE, f"{symbol}.jpg")
|
||||
Image.new("RGB", (100, 100), (190, 190, 190)).save(cache_path, "JPEG")
|
||||
return
|
||||
|
||||
title = TITLE_OVERRIDES.get(symbol, name)
|
||||
api_url = ("https://en.wikipedia.org/w/api.php?action=query&format=json"
|
||||
"&prop=pageimages&piprop=original&redirects=1&titles="
|
||||
+ urllib.parse.quote(title))
|
||||
headers = {"User-Agent": "anki-deck-periodic.py (personal Anki deck generator)"}
|
||||
try:
|
||||
req = urllib.request.Request(api_url, headers=headers)
|
||||
with urllib.request.urlopen(req, timeout=15) as resp:
|
||||
data = json.load(resp)
|
||||
pages = data.get("query", {}).get("pages", {})
|
||||
page = next(iter(pages.values()), {})
|
||||
image_url = page.get("original", {}).get("source")
|
||||
if not image_url:
|
||||
print(f" (no photo found for {name})")
|
||||
return None
|
||||
img_req = urllib.request.Request(image_url, headers=headers)
|
||||
with urllib.request.urlopen(img_req, timeout=15) as resp:
|
||||
raw = resp.read()
|
||||
img = Image.open(io.BytesIO(raw)).convert("RGB")
|
||||
img.thumbnail((300, 300))
|
||||
img.save(cache_path, "JPEG", quality=85)
|
||||
print(f" fetched photo for {name} ({title})")
|
||||
return cache_path
|
||||
except (urllib.error.URLError, urllib.error.HTTPError, OSError, ValueError) as e:
|
||||
print(f" (photo fetch failed for {name}: {e})")
|
||||
return None
|
||||
pairs = [(symbol, TITLE_OVERRIDES.get(symbol, name)) for symbol, name in needed]
|
||||
print(f"Fetching {len(pairs)} element photo(s) from Wikipedia (batched, rate-limited)...")
|
||||
url_map = resolve_pageimage_urls(pairs)
|
||||
for symbol, _title in pairs:
|
||||
if symbol not in url_map:
|
||||
print(f" (no photo found for {symbol})")
|
||||
download_element_photos(url_map)
|
||||
|
||||
_CANDIDATES = [
|
||||
args.model_path,
|
||||
@@ -365,18 +465,20 @@ def build_deck(deck_key, deck_title):
|
||||
return genanki.Deck(deck_id, deck_title)
|
||||
|
||||
|
||||
def element_image_html(number, symbol, name, media_files):
|
||||
"""Returns an <img> tag for this element's cached/fetched photo, or ""
|
||||
if --no-images was passed, the element is in NO_PHOTO_NUMBERS, or no
|
||||
photo could be found/downloaded for it."""
|
||||
def element_image_html(number, symbol, media_files):
|
||||
"""Returns an <img> tag for this element's cached photo, or "" if
|
||||
--no-images was passed, the element is in NO_PHOTO_NUMBERS, or no
|
||||
photo was found/downloaded for it. Reads the cache only — every
|
||||
network call happens up front in ensure_element_photos(), once per
|
||||
deck, not per card."""
|
||||
if args.no_images or number in NO_PHOTO_NUMBERS:
|
||||
return ""
|
||||
local_path = fetch_element_image(symbol, name)
|
||||
if local_path is None:
|
||||
cache_path = os.path.join(IMAGE_CACHE, f"{symbol}.jpg")
|
||||
if not os.path.isfile(cache_path):
|
||||
return ""
|
||||
if local_path not in media_files:
|
||||
media_files.append(local_path)
|
||||
return f'<img class="elem-img" src="{os.path.basename(local_path)}">'
|
||||
if cache_path not in media_files:
|
||||
media_files.append(cache_path)
|
||||
return f'<img class="elem-img" src="{os.path.basename(cache_path)}">'
|
||||
|
||||
|
||||
def add_note(deck, model, prompt, answer, qtext, atext, media_files, tag):
|
||||
@@ -399,6 +501,7 @@ def gen_prehs():
|
||||
deck = build_deck(deck_key, "Periodic Table: Symbols & Names (1-36)")
|
||||
media_files = []
|
||||
subset = [e for e in ELEMENTS if e[0] <= 36]
|
||||
ensure_element_photos(subset)
|
||||
cards = []
|
||||
for number, symbol, name, category in subset:
|
||||
cards.append(("symbol_to_name", number, symbol, name))
|
||||
@@ -406,7 +509,7 @@ def gen_prehs():
|
||||
random.seed(50)
|
||||
random.shuffle(cards)
|
||||
for kind, number, symbol, name in cards:
|
||||
img = element_image_html(number, symbol, name, media_files)
|
||||
img = element_image_html(number, symbol, media_files)
|
||||
if kind == "symbol_to_name":
|
||||
add_note(deck, model, img + symbol, name,
|
||||
f"What element has the symbol {symbol}?", name,
|
||||
@@ -423,6 +526,7 @@ def gen_hs():
|
||||
model_id, model = build_model(deck_key)
|
||||
deck = build_deck(deck_key, "Periodic Table: Symbols, Names & Numbers (1-118)")
|
||||
media_files = []
|
||||
ensure_element_photos(ELEMENTS)
|
||||
cards = []
|
||||
for number, symbol, name, category in ELEMENTS:
|
||||
cards.append(("symbol_to_name", number, symbol, name))
|
||||
@@ -431,7 +535,7 @@ def gen_hs():
|
||||
random.seed(51)
|
||||
random.shuffle(cards)
|
||||
for kind, number, symbol, name in cards:
|
||||
img = element_image_html(number, symbol, name, media_files)
|
||||
img = element_image_html(number, symbol, media_files)
|
||||
if kind == "symbol_to_name":
|
||||
add_note(deck, model, img + symbol, name,
|
||||
f"What element has the symbol {symbol}?", name,
|
||||
|
||||
Reference in New Issue
Block a user