Skip to content

Commit bc198ad

Browse files
committed
Add Wikipedia Commons smartphone image backfill
1 parent fe545dd commit bc198ad

2 files changed

Lines changed: 515 additions & 0 deletions

File tree

Lines changed: 370 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -0,0 +1,370 @@
1+
"""Backfill freely licensed Commons photos from already cited Wikipedia articles.
2+
3+
Run a dry sample first, then use --apply --offset/--limit for sequential batches.
4+
The append-only decision cache is shared between dry runs and apply runs.
5+
"""
6+
7+
from __future__ import annotations
8+
9+
import argparse
10+
import json
11+
import re
12+
from collections import Counter
13+
from pathlib import Path
14+
from typing import Any
15+
from urllib.parse import unquote, urlparse
16+
17+
import httpx
18+
from bs4 import BeautifulSoup
19+
20+
from app.verify.wikipedia_smartphone_backfill import (
21+
USER_AGENT,
22+
PoliteWiki,
23+
append_cache,
24+
variant_conflict,
25+
)
26+
27+
VERSION = 7
28+
BAD_IMAGE = re.compile(
29+
r"(?<![A-Za-z0-9])(?:logo|logotype|wordmark|icon|emblem|flag|symbol|diagram|chart|screenshot|placeholder|render|advertisement|battery|headquarters?|building|campus|series|lineup|packaging|시리즈)(?![A-Za-z0-9])",
30+
re.I,
31+
)
32+
GROUP_IMAGE = re.compile(r"_and_.*(?:Xiaomi|Samsung|Huawei|OnePlus|Oppo|Vivo)", re.I)
33+
NON_PHONE_MODEL = re.compile(r"^(?:Surface \d+|Palm TX)$", re.I)
34+
BAD_LICENSE = re.compile(r"non.free|fair.use|all.rights.reserved|unknown|unclear|copyrighted", re.I)
35+
FREE_LICENSE = re.compile(
36+
r"^(?:CC[- ]?BY(?:[- ]?SA)?[- ]?[1-4](?:\.0)?|CC0(?:[- ]?1\.0)?|PUBLIC DOMAIN)$", re.I
37+
)
38+
39+
40+
def article_url(record: dict[str, Any]) -> str | None:
41+
for url in record.get("source_urls") or []:
42+
if not isinstance(url, str):
43+
continue
44+
parsed = urlparse(url)
45+
if (
46+
parsed.scheme == "https"
47+
and parsed.hostname
48+
and (parsed.hostname == "wikipedia.org" or parsed.hostname.endswith(".wikipedia.org"))
49+
and parsed.path.startswith("/wiki/")
50+
):
51+
return url
52+
return None
53+
54+
55+
def eligible(root: Path) -> list[tuple[Path, dict[str, Any], str]]:
56+
rows = []
57+
for path in sorted((root / "data" / "smartphone").rglob("*.json")):
58+
try:
59+
record = json.loads(path.read_text(encoding="utf-8-sig"))
60+
except (ValueError, OSError):
61+
continue
62+
if isinstance(record, dict) and "image_url" in record and record["image_url"] is None:
63+
url = article_url(record)
64+
if url:
65+
rows.append((path, record, url))
66+
return rows
67+
68+
69+
def plain(value: object) -> str:
70+
return BeautifulSoup(str(value or ""), "html.parser").get_text(" ", strip=True)
71+
72+
73+
def metadata_value(metadata: dict[str, Any], key: str) -> str:
74+
item = metadata.get(key)
75+
return plain(item.get("value")) if isinstance(item, dict) else ""
76+
77+
78+
def filename_matches_model(name: str, filename: str) -> bool:
79+
"""Require a distinctive model token in the file name; generic brand photos fail."""
80+
file_words = re.findall(
81+
r"[a-z]+\d+[a-z]*|\d+[a-z]+|[a-z]+|\d+",
82+
re.sub(r"([a-z])([A-Z])", r"\1 \2", filename.rsplit(".", 1)[0]).lower(),
83+
)
84+
file_tokens = set(file_words)
85+
file_tokens.update(
86+
left + right
87+
for left, right in zip(file_words, file_words[1:], strict=False)
88+
if left.isalpha() and right.isdigit()
89+
)
90+
name_tokens = re.findall(r"[a-z]+\d+[a-z]*|\d+[a-z]+|[a-z]+|\d+", name.lower())
91+
codes = [
92+
token
93+
for token in name_tokens
94+
if any(char.isdigit() for char in token) and not token.endswith("gb")
95+
]
96+
if codes:
97+
return any(code in file_tokens for code in codes)
98+
generic = {
99+
"apple",
100+
"samsung",
101+
"google",
102+
"honor",
103+
"huawei",
104+
"motorola",
105+
"nokia",
106+
"xiaomi",
107+
"oppo",
108+
"vivo",
109+
"realme",
110+
"blackberry",
111+
"casio",
112+
"amazon",
113+
"nothing",
114+
"jolla",
115+
"itel",
116+
"oneplus",
117+
"microsoft",
118+
"sony",
119+
"palm",
120+
"phone",
121+
"smartphone",
122+
"mobile",
123+
"edition",
124+
"generation",
125+
"plus",
126+
"ultra",
127+
"pro",
128+
"gb",
129+
"htc",
130+
"galaxy",
131+
"xperia",
132+
"lumia",
133+
}
134+
distinctive = [token for token in name_tokens if token not in generic and len(token) >= 2]
135+
return any(token in file_tokens for token in distinctive)
136+
137+
138+
def license_name(metadata: dict[str, Any]) -> str | None:
139+
short = metadata_value(metadata, "LicenseShortName").upper().replace(" ", "-")
140+
terms = " ".join(
141+
metadata_value(metadata, key) for key in ("UsageTerms", "Restrictions", "License")
142+
)
143+
if BAD_LICENSE.search(short + " " + terms) or not FREE_LICENSE.fullmatch(
144+
short.replace("PUBLIC-DOMAIN", "PUBLIC DOMAIN")
145+
):
146+
return None
147+
if short.startswith("CC0"):
148+
return "CC0-1.0"
149+
if short == "PUBLIC-DOMAIN":
150+
return "Public Domain"
151+
return short.replace("CC-BY-SA-", "CC-BY-SA-").replace("CC-BY-", "CC-BY-")
152+
153+
154+
def load_decisions(path: Path) -> dict[str, dict[str, Any]]:
155+
decisions = {}
156+
if path.exists():
157+
for line in path.read_text(encoding="utf-8").splitlines():
158+
try:
159+
item = json.loads(line)
160+
except ValueError:
161+
continue
162+
if item.get("version") in {6, VERSION} and isinstance(item.get("path"), str):
163+
decisions[item["path"]] = item
164+
return decisions
165+
166+
167+
class CommonsFetcher(PoliteWiki):
168+
def __init__(self, sleep_s: float = 1.0) -> None:
169+
super().__init__(sleep_s=max(1.0, sleep_s), timeout=20.0)
170+
171+
def query(self, host: str, params: dict[str, str]) -> dict[str, Any]:
172+
self._pause()
173+
if self._client is None:
174+
self._client = httpx.Client(
175+
timeout=self.timeout, headers={"User-Agent": USER_AGENT}, follow_redirects=True
176+
)
177+
response = self._client.get(
178+
f"https://{host}/w/api.php", params={"action": "query", "format": "json", **params}
179+
)
180+
response.raise_for_status()
181+
return response.json()
182+
183+
184+
def inspect(url: str, fetcher: CommonsFetcher, name: str = "") -> dict[str, str]:
185+
parsed = urlparse(url)
186+
title = unquote(parsed.path.removeprefix("/wiki/")).replace("_", " ")
187+
try:
188+
result = fetcher.query(
189+
parsed.hostname or "en.wikipedia.org",
190+
{"titles": title, "prop": "pageimages", "piprop": "original", "redirects": "1"},
191+
)
192+
pages = result.get("query", {}).get("pages", {})
193+
page = next(iter(pages.values()))
194+
original = page.get("original") or {}
195+
source = original.get("source") if isinstance(original, dict) else None
196+
filename = page.get("pageimage")
197+
if not filename and isinstance(source, str):
198+
filename = unquote(urlparse(source).path.rsplit("/", 1)[-1])
199+
if not isinstance(filename, str) or not filename:
200+
return {"reason": "no_image"}
201+
if (
202+
BAD_IMAGE.search(filename)
203+
or GROUP_IMAGE.search(filename)
204+
or NON_PHONE_MODEL.search(name)
205+
or not filename.lower().endswith((".jpg", ".jpeg", ".webp"))
206+
or (
207+
name
208+
and (
209+
not filename_matches_model(name, filename)
210+
or variant_conflict(name, filename.replace("_", " "))
211+
)
212+
)
213+
):
214+
return {"reason": "logo_like", "file": filename}
215+
commons = fetcher.query(
216+
"commons.wikimedia.org",
217+
{
218+
"titles": f"File:{filename}",
219+
"prop": "imageinfo",
220+
"iiprop": "url|mime|extmetadata",
221+
"iiextmetadatafilter": (
222+
"LicenseShortName|License|UsageTerms|Restrictions|Artist|Credit|"
223+
"ImageDescription|ObjectName"
224+
),
225+
},
226+
)
227+
file_page = next(iter(commons.get("query", {}).get("pages", {}).values()))
228+
info = (file_page.get("imageinfo") or [None])[0]
229+
if not isinstance(info, dict):
230+
return {"reason": "bad_license", "file": filename}
231+
image_url = info.get("url")
232+
metadata = info.get("extmetadata") or {}
233+
license_id = license_name(metadata)
234+
if (
235+
not license_id
236+
or not isinstance(image_url, str)
237+
or urlparse(image_url).hostname not in {"upload.wikimedia.org", "commons.wikimedia.org"}
238+
):
239+
return {
240+
"reason": "bad_license",
241+
"file": filename,
242+
"raw_license": metadata_value(metadata, "LicenseShortName"),
243+
}
244+
if info.get("mime") not in {"image/jpeg", "image/webp"}:
245+
return {"reason": "logo_like", "file": filename}
246+
description = " ".join(
247+
metadata_value(metadata, key) for key in ("ObjectName", "ImageDescription")
248+
)
249+
if BAD_IMAGE.search(description):
250+
return {"reason": "logo_like", "file": filename}
251+
attribution = metadata_value(metadata, "Artist") or metadata_value(metadata, "Credit")
252+
if not attribution:
253+
return {"reason": "bad_license", "file": filename, "raw_license": license_id}
254+
return {
255+
"reason": "accepted",
256+
"file": filename,
257+
"image_url": image_url.split("?", 1)[0],
258+
"image_license": license_id,
259+
"image_attribution": attribution,
260+
}
261+
except (httpx.HTTPError, ValueError, KeyError, StopIteration, TypeError) as exc:
262+
return {"reason": "error", "error": str(exc)[:200]}
263+
264+
265+
def write_image(path: Path, result: dict[str, str]) -> None:
266+
text = path.read_text(encoding="utf-8")
267+
record = json.loads(text)
268+
if record.get("image_url") is not None:
269+
return
270+
replacement = (
271+
'"image_url": ' + json.dumps(result["image_url"], ensure_ascii=False) + ",\n"
272+
' "image_license": ' + json.dumps(result["image_license"], ensure_ascii=False) + ",\n"
273+
' "image_attribution": ' + json.dumps(result["image_attribution"], ensure_ascii=False)
274+
)
275+
updated, count = re.subn(r'"image_url"\s*:\s*null', lambda _match: replacement, text, count=1)
276+
if count != 1:
277+
raise ValueError(f"missing null image_url in {path}")
278+
path.write_text(updated, encoding="utf-8")
279+
280+
281+
def run(
282+
root: Path,
283+
*,
284+
offset: int = 0,
285+
limit: int | None = None,
286+
apply: bool = False,
287+
sleep_s: float = 1.0,
288+
cache_path: Path | None = None,
289+
) -> list[dict[str, Any]]:
290+
cache_path = cache_path or root / "data" / "_verify" / "state" / "wikipedia_image_cache.jsonl"
291+
cache = load_decisions(cache_path)
292+
rows = eligible(root)[offset : None if limit is None else offset + limit]
293+
fetcher = CommonsFetcher(sleep_s)
294+
results = []
295+
for index, (path, record, article) in enumerate(rows, 1):
296+
rel = path.relative_to(root).as_posix()
297+
decision = cache.get(rel)
298+
if (
299+
decision is not None
300+
and decision.get("version") == 6
301+
and decision.get("article") == article
302+
):
303+
decision = dict(decision, version=VERSION)
304+
if decision.get("reason") == "accepted" and (
305+
BAD_IMAGE.search(str(decision.get("file") or ""))
306+
or GROUP_IMAGE.search(str(decision.get("file") or ""))
307+
or NON_PHONE_MODEL.search(str(record.get("name") or ""))
308+
or not filename_matches_model(
309+
str(record.get("name") or ""), str(decision.get("file") or "")
310+
)
311+
or variant_conflict(
312+
str(record.get("name") or ""), str(decision.get("file") or "").replace("_", " ")
313+
)
314+
):
315+
decision["reason"] = "logo_like"
316+
for key in ("image_url", "image_license", "image_attribution"):
317+
decision.pop(key, None)
318+
append_cache(decision, cache_path)
319+
if (
320+
decision is None
321+
or decision.get("article") != article
322+
or decision.get("reason") == "error"
323+
):
324+
decision = {
325+
"version": VERSION,
326+
"path": rel,
327+
"name": record.get("name"),
328+
"article": article,
329+
**inspect(article, fetcher, str(record.get("name") or "")),
330+
}
331+
if decision["reason"] != "error":
332+
append_cache(decision, cache_path)
333+
if apply and decision["reason"] == "accepted":
334+
write_image(path, decision)
335+
results.append(decision)
336+
message = (
337+
f"[{index}/{len(rows)}] {decision['reason']}: "
338+
f"{record.get('name')} ({decision.get('file', '')})"
339+
)
340+
print(message.encode("ascii", "backslashreplace").decode("ascii"), flush=True)
341+
return results
342+
343+
344+
def main() -> None:
345+
parser = argparse.ArgumentParser(description=__doc__)
346+
parser.add_argument("--data-root", type=Path, required=True)
347+
parser.add_argument("--offset", type=int, default=0)
348+
parser.add_argument("--limit", type=int)
349+
parser.add_argument("--sleep", type=float, default=1.0)
350+
parser.add_argument("--apply", action="store_true")
351+
args = parser.parse_args()
352+
results = run(
353+
args.data_root, offset=args.offset, limit=args.limit, apply=args.apply, sleep_s=args.sleep
354+
)
355+
print(
356+
json.dumps(
357+
{
358+
"checked": len(results),
359+
"reasons": Counter(row["reason"] for row in results),
360+
"licenses": Counter(
361+
row.get("image_license") for row in results if row["reason"] == "accepted"
362+
),
363+
},
364+
indent=2,
365+
)
366+
)
367+
368+
369+
if __name__ == "__main__":
370+
main()

0 commit comments

Comments
 (0)