|
| 1 | +"""Backfill freely licensed Commons photos from already cited Wikipedia articles. |
| 2 | +
|
| 3 | +Run a dry sample first, then use --apply --offset/--limit for sequential batches. |
| 4 | +The append-only decision cache is shared between dry runs and apply runs. |
| 5 | +""" |
| 6 | + |
| 7 | +from __future__ import annotations |
| 8 | + |
| 9 | +import argparse |
| 10 | +import json |
| 11 | +import re |
| 12 | +from collections import Counter |
| 13 | +from pathlib import Path |
| 14 | +from typing import Any |
| 15 | +from urllib.parse import unquote, urlparse |
| 16 | + |
| 17 | +import httpx |
| 18 | +from bs4 import BeautifulSoup |
| 19 | + |
| 20 | +from app.verify.wikipedia_smartphone_backfill import ( |
| 21 | + USER_AGENT, |
| 22 | + PoliteWiki, |
| 23 | + append_cache, |
| 24 | + variant_conflict, |
| 25 | +) |
| 26 | + |
| 27 | +VERSION = 7 |
| 28 | +BAD_IMAGE = re.compile( |
| 29 | + r"(?<![A-Za-z0-9])(?:logo|logotype|wordmark|icon|emblem|flag|symbol|diagram|chart|screenshot|placeholder|render|advertisement|battery|headquarters?|building|campus|series|lineup|packaging|시리즈)(?![A-Za-z0-9])", |
| 30 | + re.I, |
| 31 | +) |
| 32 | +GROUP_IMAGE = re.compile(r"_and_.*(?:Xiaomi|Samsung|Huawei|OnePlus|Oppo|Vivo)", re.I) |
| 33 | +NON_PHONE_MODEL = re.compile(r"^(?:Surface \d+|Palm TX)$", re.I) |
| 34 | +BAD_LICENSE = re.compile(r"non.free|fair.use|all.rights.reserved|unknown|unclear|copyrighted", re.I) |
| 35 | +FREE_LICENSE = re.compile( |
| 36 | + r"^(?:CC[- ]?BY(?:[- ]?SA)?[- ]?[1-4](?:\.0)?|CC0(?:[- ]?1\.0)?|PUBLIC DOMAIN)$", re.I |
| 37 | +) |
| 38 | + |
| 39 | + |
| 40 | +def article_url(record: dict[str, Any]) -> str | None: |
| 41 | + for url in record.get("source_urls") or []: |
| 42 | + if not isinstance(url, str): |
| 43 | + continue |
| 44 | + parsed = urlparse(url) |
| 45 | + if ( |
| 46 | + parsed.scheme == "https" |
| 47 | + and parsed.hostname |
| 48 | + and (parsed.hostname == "wikipedia.org" or parsed.hostname.endswith(".wikipedia.org")) |
| 49 | + and parsed.path.startswith("/wiki/") |
| 50 | + ): |
| 51 | + return url |
| 52 | + return None |
| 53 | + |
| 54 | + |
| 55 | +def eligible(root: Path) -> list[tuple[Path, dict[str, Any], str]]: |
| 56 | + rows = [] |
| 57 | + for path in sorted((root / "data" / "smartphone").rglob("*.json")): |
| 58 | + try: |
| 59 | + record = json.loads(path.read_text(encoding="utf-8-sig")) |
| 60 | + except (ValueError, OSError): |
| 61 | + continue |
| 62 | + if isinstance(record, dict) and "image_url" in record and record["image_url"] is None: |
| 63 | + url = article_url(record) |
| 64 | + if url: |
| 65 | + rows.append((path, record, url)) |
| 66 | + return rows |
| 67 | + |
| 68 | + |
| 69 | +def plain(value: object) -> str: |
| 70 | + return BeautifulSoup(str(value or ""), "html.parser").get_text(" ", strip=True) |
| 71 | + |
| 72 | + |
| 73 | +def metadata_value(metadata: dict[str, Any], key: str) -> str: |
| 74 | + item = metadata.get(key) |
| 75 | + return plain(item.get("value")) if isinstance(item, dict) else "" |
| 76 | + |
| 77 | + |
| 78 | +def filename_matches_model(name: str, filename: str) -> bool: |
| 79 | + """Require a distinctive model token in the file name; generic brand photos fail.""" |
| 80 | + file_words = re.findall( |
| 81 | + r"[a-z]+\d+[a-z]*|\d+[a-z]+|[a-z]+|\d+", |
| 82 | + re.sub(r"([a-z])([A-Z])", r"\1 \2", filename.rsplit(".", 1)[0]).lower(), |
| 83 | + ) |
| 84 | + file_tokens = set(file_words) |
| 85 | + file_tokens.update( |
| 86 | + left + right |
| 87 | + for left, right in zip(file_words, file_words[1:], strict=False) |
| 88 | + if left.isalpha() and right.isdigit() |
| 89 | + ) |
| 90 | + name_tokens = re.findall(r"[a-z]+\d+[a-z]*|\d+[a-z]+|[a-z]+|\d+", name.lower()) |
| 91 | + codes = [ |
| 92 | + token |
| 93 | + for token in name_tokens |
| 94 | + if any(char.isdigit() for char in token) and not token.endswith("gb") |
| 95 | + ] |
| 96 | + if codes: |
| 97 | + return any(code in file_tokens for code in codes) |
| 98 | + generic = { |
| 99 | + "apple", |
| 100 | + "samsung", |
| 101 | + "google", |
| 102 | + "honor", |
| 103 | + "huawei", |
| 104 | + "motorola", |
| 105 | + "nokia", |
| 106 | + "xiaomi", |
| 107 | + "oppo", |
| 108 | + "vivo", |
| 109 | + "realme", |
| 110 | + "blackberry", |
| 111 | + "casio", |
| 112 | + "amazon", |
| 113 | + "nothing", |
| 114 | + "jolla", |
| 115 | + "itel", |
| 116 | + "oneplus", |
| 117 | + "microsoft", |
| 118 | + "sony", |
| 119 | + "palm", |
| 120 | + "phone", |
| 121 | + "smartphone", |
| 122 | + "mobile", |
| 123 | + "edition", |
| 124 | + "generation", |
| 125 | + "plus", |
| 126 | + "ultra", |
| 127 | + "pro", |
| 128 | + "gb", |
| 129 | + "htc", |
| 130 | + "galaxy", |
| 131 | + "xperia", |
| 132 | + "lumia", |
| 133 | + } |
| 134 | + distinctive = [token for token in name_tokens if token not in generic and len(token) >= 2] |
| 135 | + return any(token in file_tokens for token in distinctive) |
| 136 | + |
| 137 | + |
| 138 | +def license_name(metadata: dict[str, Any]) -> str | None: |
| 139 | + short = metadata_value(metadata, "LicenseShortName").upper().replace(" ", "-") |
| 140 | + terms = " ".join( |
| 141 | + metadata_value(metadata, key) for key in ("UsageTerms", "Restrictions", "License") |
| 142 | + ) |
| 143 | + if BAD_LICENSE.search(short + " " + terms) or not FREE_LICENSE.fullmatch( |
| 144 | + short.replace("PUBLIC-DOMAIN", "PUBLIC DOMAIN") |
| 145 | + ): |
| 146 | + return None |
| 147 | + if short.startswith("CC0"): |
| 148 | + return "CC0-1.0" |
| 149 | + if short == "PUBLIC-DOMAIN": |
| 150 | + return "Public Domain" |
| 151 | + return short.replace("CC-BY-SA-", "CC-BY-SA-").replace("CC-BY-", "CC-BY-") |
| 152 | + |
| 153 | + |
| 154 | +def load_decisions(path: Path) -> dict[str, dict[str, Any]]: |
| 155 | + decisions = {} |
| 156 | + if path.exists(): |
| 157 | + for line in path.read_text(encoding="utf-8").splitlines(): |
| 158 | + try: |
| 159 | + item = json.loads(line) |
| 160 | + except ValueError: |
| 161 | + continue |
| 162 | + if item.get("version") in {6, VERSION} and isinstance(item.get("path"), str): |
| 163 | + decisions[item["path"]] = item |
| 164 | + return decisions |
| 165 | + |
| 166 | + |
| 167 | +class CommonsFetcher(PoliteWiki): |
| 168 | + def __init__(self, sleep_s: float = 1.0) -> None: |
| 169 | + super().__init__(sleep_s=max(1.0, sleep_s), timeout=20.0) |
| 170 | + |
| 171 | + def query(self, host: str, params: dict[str, str]) -> dict[str, Any]: |
| 172 | + self._pause() |
| 173 | + if self._client is None: |
| 174 | + self._client = httpx.Client( |
| 175 | + timeout=self.timeout, headers={"User-Agent": USER_AGENT}, follow_redirects=True |
| 176 | + ) |
| 177 | + response = self._client.get( |
| 178 | + f"https://{host}/w/api.php", params={"action": "query", "format": "json", **params} |
| 179 | + ) |
| 180 | + response.raise_for_status() |
| 181 | + return response.json() |
| 182 | + |
| 183 | + |
| 184 | +def inspect(url: str, fetcher: CommonsFetcher, name: str = "") -> dict[str, str]: |
| 185 | + parsed = urlparse(url) |
| 186 | + title = unquote(parsed.path.removeprefix("/wiki/")).replace("_", " ") |
| 187 | + try: |
| 188 | + result = fetcher.query( |
| 189 | + parsed.hostname or "en.wikipedia.org", |
| 190 | + {"titles": title, "prop": "pageimages", "piprop": "original", "redirects": "1"}, |
| 191 | + ) |
| 192 | + pages = result.get("query", {}).get("pages", {}) |
| 193 | + page = next(iter(pages.values())) |
| 194 | + original = page.get("original") or {} |
| 195 | + source = original.get("source") if isinstance(original, dict) else None |
| 196 | + filename = page.get("pageimage") |
| 197 | + if not filename and isinstance(source, str): |
| 198 | + filename = unquote(urlparse(source).path.rsplit("/", 1)[-1]) |
| 199 | + if not isinstance(filename, str) or not filename: |
| 200 | + return {"reason": "no_image"} |
| 201 | + if ( |
| 202 | + BAD_IMAGE.search(filename) |
| 203 | + or GROUP_IMAGE.search(filename) |
| 204 | + or NON_PHONE_MODEL.search(name) |
| 205 | + or not filename.lower().endswith((".jpg", ".jpeg", ".webp")) |
| 206 | + or ( |
| 207 | + name |
| 208 | + and ( |
| 209 | + not filename_matches_model(name, filename) |
| 210 | + or variant_conflict(name, filename.replace("_", " ")) |
| 211 | + ) |
| 212 | + ) |
| 213 | + ): |
| 214 | + return {"reason": "logo_like", "file": filename} |
| 215 | + commons = fetcher.query( |
| 216 | + "commons.wikimedia.org", |
| 217 | + { |
| 218 | + "titles": f"File:{filename}", |
| 219 | + "prop": "imageinfo", |
| 220 | + "iiprop": "url|mime|extmetadata", |
| 221 | + "iiextmetadatafilter": ( |
| 222 | + "LicenseShortName|License|UsageTerms|Restrictions|Artist|Credit|" |
| 223 | + "ImageDescription|ObjectName" |
| 224 | + ), |
| 225 | + }, |
| 226 | + ) |
| 227 | + file_page = next(iter(commons.get("query", {}).get("pages", {}).values())) |
| 228 | + info = (file_page.get("imageinfo") or [None])[0] |
| 229 | + if not isinstance(info, dict): |
| 230 | + return {"reason": "bad_license", "file": filename} |
| 231 | + image_url = info.get("url") |
| 232 | + metadata = info.get("extmetadata") or {} |
| 233 | + license_id = license_name(metadata) |
| 234 | + if ( |
| 235 | + not license_id |
| 236 | + or not isinstance(image_url, str) |
| 237 | + or urlparse(image_url).hostname not in {"upload.wikimedia.org", "commons.wikimedia.org"} |
| 238 | + ): |
| 239 | + return { |
| 240 | + "reason": "bad_license", |
| 241 | + "file": filename, |
| 242 | + "raw_license": metadata_value(metadata, "LicenseShortName"), |
| 243 | + } |
| 244 | + if info.get("mime") not in {"image/jpeg", "image/webp"}: |
| 245 | + return {"reason": "logo_like", "file": filename} |
| 246 | + description = " ".join( |
| 247 | + metadata_value(metadata, key) for key in ("ObjectName", "ImageDescription") |
| 248 | + ) |
| 249 | + if BAD_IMAGE.search(description): |
| 250 | + return {"reason": "logo_like", "file": filename} |
| 251 | + attribution = metadata_value(metadata, "Artist") or metadata_value(metadata, "Credit") |
| 252 | + if not attribution: |
| 253 | + return {"reason": "bad_license", "file": filename, "raw_license": license_id} |
| 254 | + return { |
| 255 | + "reason": "accepted", |
| 256 | + "file": filename, |
| 257 | + "image_url": image_url.split("?", 1)[0], |
| 258 | + "image_license": license_id, |
| 259 | + "image_attribution": attribution, |
| 260 | + } |
| 261 | + except (httpx.HTTPError, ValueError, KeyError, StopIteration, TypeError) as exc: |
| 262 | + return {"reason": "error", "error": str(exc)[:200]} |
| 263 | + |
| 264 | + |
| 265 | +def write_image(path: Path, result: dict[str, str]) -> None: |
| 266 | + text = path.read_text(encoding="utf-8") |
| 267 | + record = json.loads(text) |
| 268 | + if record.get("image_url") is not None: |
| 269 | + return |
| 270 | + replacement = ( |
| 271 | + '"image_url": ' + json.dumps(result["image_url"], ensure_ascii=False) + ",\n" |
| 272 | + ' "image_license": ' + json.dumps(result["image_license"], ensure_ascii=False) + ",\n" |
| 273 | + ' "image_attribution": ' + json.dumps(result["image_attribution"], ensure_ascii=False) |
| 274 | + ) |
| 275 | + updated, count = re.subn(r'"image_url"\s*:\s*null', lambda _match: replacement, text, count=1) |
| 276 | + if count != 1: |
| 277 | + raise ValueError(f"missing null image_url in {path}") |
| 278 | + path.write_text(updated, encoding="utf-8") |
| 279 | + |
| 280 | + |
| 281 | +def run( |
| 282 | + root: Path, |
| 283 | + *, |
| 284 | + offset: int = 0, |
| 285 | + limit: int | None = None, |
| 286 | + apply: bool = False, |
| 287 | + sleep_s: float = 1.0, |
| 288 | + cache_path: Path | None = None, |
| 289 | +) -> list[dict[str, Any]]: |
| 290 | + cache_path = cache_path or root / "data" / "_verify" / "state" / "wikipedia_image_cache.jsonl" |
| 291 | + cache = load_decisions(cache_path) |
| 292 | + rows = eligible(root)[offset : None if limit is None else offset + limit] |
| 293 | + fetcher = CommonsFetcher(sleep_s) |
| 294 | + results = [] |
| 295 | + for index, (path, record, article) in enumerate(rows, 1): |
| 296 | + rel = path.relative_to(root).as_posix() |
| 297 | + decision = cache.get(rel) |
| 298 | + if ( |
| 299 | + decision is not None |
| 300 | + and decision.get("version") == 6 |
| 301 | + and decision.get("article") == article |
| 302 | + ): |
| 303 | + decision = dict(decision, version=VERSION) |
| 304 | + if decision.get("reason") == "accepted" and ( |
| 305 | + BAD_IMAGE.search(str(decision.get("file") or "")) |
| 306 | + or GROUP_IMAGE.search(str(decision.get("file") or "")) |
| 307 | + or NON_PHONE_MODEL.search(str(record.get("name") or "")) |
| 308 | + or not filename_matches_model( |
| 309 | + str(record.get("name") or ""), str(decision.get("file") or "") |
| 310 | + ) |
| 311 | + or variant_conflict( |
| 312 | + str(record.get("name") or ""), str(decision.get("file") or "").replace("_", " ") |
| 313 | + ) |
| 314 | + ): |
| 315 | + decision["reason"] = "logo_like" |
| 316 | + for key in ("image_url", "image_license", "image_attribution"): |
| 317 | + decision.pop(key, None) |
| 318 | + append_cache(decision, cache_path) |
| 319 | + if ( |
| 320 | + decision is None |
| 321 | + or decision.get("article") != article |
| 322 | + or decision.get("reason") == "error" |
| 323 | + ): |
| 324 | + decision = { |
| 325 | + "version": VERSION, |
| 326 | + "path": rel, |
| 327 | + "name": record.get("name"), |
| 328 | + "article": article, |
| 329 | + **inspect(article, fetcher, str(record.get("name") or "")), |
| 330 | + } |
| 331 | + if decision["reason"] != "error": |
| 332 | + append_cache(decision, cache_path) |
| 333 | + if apply and decision["reason"] == "accepted": |
| 334 | + write_image(path, decision) |
| 335 | + results.append(decision) |
| 336 | + message = ( |
| 337 | + f"[{index}/{len(rows)}] {decision['reason']}: " |
| 338 | + f"{record.get('name')} ({decision.get('file', '')})" |
| 339 | + ) |
| 340 | + print(message.encode("ascii", "backslashreplace").decode("ascii"), flush=True) |
| 341 | + return results |
| 342 | + |
| 343 | + |
| 344 | +def main() -> None: |
| 345 | + parser = argparse.ArgumentParser(description=__doc__) |
| 346 | + parser.add_argument("--data-root", type=Path, required=True) |
| 347 | + parser.add_argument("--offset", type=int, default=0) |
| 348 | + parser.add_argument("--limit", type=int) |
| 349 | + parser.add_argument("--sleep", type=float, default=1.0) |
| 350 | + parser.add_argument("--apply", action="store_true") |
| 351 | + args = parser.parse_args() |
| 352 | + results = run( |
| 353 | + args.data_root, offset=args.offset, limit=args.limit, apply=args.apply, sleep_s=args.sleep |
| 354 | + ) |
| 355 | + print( |
| 356 | + json.dumps( |
| 357 | + { |
| 358 | + "checked": len(results), |
| 359 | + "reasons": Counter(row["reason"] for row in results), |
| 360 | + "licenses": Counter( |
| 361 | + row.get("image_license") for row in results if row["reason"] == "accepted" |
| 362 | + ), |
| 363 | + }, |
| 364 | + indent=2, |
| 365 | + ) |
| 366 | + ) |
| 367 | + |
| 368 | + |
| 369 | +if __name__ == "__main__": |
| 370 | + main() |
0 commit comments