Skip to content

Commit 439e30d

Browse files
committed
Allow opt-in Commons backfill for missing image URL keys
1 parent fbf1619 commit 439e30d

2 files changed

Lines changed: 67 additions & 9 deletions

File tree

‎app/verify/wikipedia_image_backfill.py‎

Lines changed: 37 additions & 9 deletions
Original file line numberDiff line numberDiff line change
@@ -52,14 +52,19 @@ def article_url(record: dict[str, Any]) -> str | None:
5252
return None
5353

5454

55-
def eligible(root: Path) -> list[tuple[Path, dict[str, Any], str]]:
55+
def eligible(
56+
root: Path, *, include_missing_key: bool = False
57+
) -> list[tuple[Path, dict[str, Any], str]]:
5658
rows = []
5759
for path in sorted((root / "data" / "smartphone").rglob("*.json")):
5860
try:
5961
record = json.loads(path.read_text(encoding="utf-8-sig"))
6062
except (ValueError, OSError):
6163
continue
62-
if isinstance(record, dict) and "image_url" in record and record["image_url"] is None:
64+
if isinstance(record, dict) and (
65+
("image_url" in record and record["image_url"] is None)
66+
or (include_missing_key and "image_url" not in record)
67+
):
6368
url = article_url(record)
6469
if url:
6570
rows.append((path, record, url))
@@ -264,19 +269,33 @@ def inspect(url: str, fetcher: CommonsFetcher, name: str = "") -> dict[str, str]
264269

265270

266271
def write_image(path: Path, result: dict[str, str]) -> None:
267-
text = path.read_text(encoding="utf-8")
272+
text = path.read_bytes().decode("utf-8")
268273
record = json.loads(text)
269274
if record.get("image_url") is not None:
270275
return
276+
newline = "\r\n" if "\r\n" in text else "\n"
271277
replacement = (
272278
'"image_url": ' + json.dumps(result["image_url"], ensure_ascii=False) + ",\n"
273279
' "image_license": ' + json.dumps(result["image_license"], ensure_ascii=False) + ",\n"
274280
' "image_attribution": ' + json.dumps(result["image_attribution"], ensure_ascii=False)
275281
)
276-
updated, count = re.subn(r'"image_url"\s*:\s*null', lambda _match: replacement, text, count=1)
277-
if count != 1:
278-
raise ValueError(f"missing null image_url in {path}")
279-
path.write_text(updated, encoding="utf-8")
282+
replacement = replacement.replace("\n", newline)
283+
if "image_url" in record:
284+
updated, count = re.subn(
285+
r'"image_url"\s*:\s*null', lambda _match: replacement, text, count=1
286+
)
287+
if count != 1:
288+
raise ValueError(f"missing null image_url in {path}")
289+
else:
290+
if "image_license" in record or "image_attribution" in record:
291+
raise ValueError(f"existing image metadata in {path}")
292+
match = re.match(r'\{(?P<newline>\r?\n)(?P<indent>[ \t]+)(?=")', text)
293+
if match is None:
294+
raise ValueError(f"cannot insert image fields in {path}")
295+
indent = match.group("indent")
296+
fields = replacement.replace(newline + " ", newline + indent)
297+
updated = text[: match.end()] + fields + "," + newline + indent + text[match.end() :]
298+
path.write_bytes(updated.encode("utf-8"))
280299

281300

282301
def run(
@@ -287,10 +306,13 @@ def run(
287306
apply: bool = False,
288307
sleep_s: float = 1.0,
289308
cache_path: Path | None = None,
309+
include_missing_key: bool = False,
290310
) -> list[dict[str, Any]]:
291311
cache_path = cache_path or root / "data" / "_verify" / "state" / "wikipedia_image_cache.jsonl"
292312
cache = load_decisions(cache_path)
293-
rows = eligible(root)[offset : None if limit is None else offset + limit]
313+
rows = eligible(root, include_missing_key=include_missing_key)[
314+
offset : None if limit is None else offset + limit
315+
]
294316
fetcher = CommonsFetcher(sleep_s)
295317
results = []
296318
for index, (path, record, article) in enumerate(rows, 1):
@@ -349,9 +371,15 @@ def main() -> None:
349371
parser.add_argument("--limit", type=int)
350372
parser.add_argument("--sleep", type=float, default=1.0)
351373
parser.add_argument("--apply", action="store_true")
374+
parser.add_argument("--include-missing-key", action="store_true")
352375
args = parser.parse_args()
353376
results = run(
354-
args.data_root, offset=args.offset, limit=args.limit, apply=args.apply, sleep_s=args.sleep
377+
args.data_root,
378+
offset=args.offset,
379+
limit=args.limit,
380+
apply=args.apply,
381+
sleep_s=args.sleep,
382+
include_missing_key=args.include_missing_key,
355383
)
356384
print(
357385
json.dumps(

‎tests/verify/test_wikipedia_image_backfill.py‎

Lines changed: 30 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -143,3 +143,33 @@ def test_eligible_requires_explicit_null_image_url() -> None:
143143
json.dumps({"image_url": None, "source_urls": source}), encoding="utf-8"
144144
)
145145
assert [path.name for path, _, _ in eligible(root)] == ["null.json"]
146+
assert [path.name for path, _, _ in eligible(root, include_missing_key=True)] == [
147+
"missing.json",
148+
"null.json",
149+
]
150+
151+
152+
def test_write_image_inserts_missing_fields_without_other_changes(tmp_path: Path) -> None:
153+
path = tmp_path / "missing.json"
154+
before = (
155+
'{\r\n "name": "Example",\r\n "source_urls": '
156+
'["https://en.wikipedia.org/wiki/Example"]\r\n}\r\n'
157+
)
158+
path.write_bytes(before.encode("utf-8"))
159+
write_image(
160+
path,
161+
{
162+
"image_url": "https://upload.wikimedia.org/a.jpg",
163+
"image_license": "CC-BY-4.0",
164+
"image_attribution": "Alice",
165+
},
166+
)
167+
after = path.read_bytes().decode("utf-8")
168+
assert after == before.replace(
169+
' "name":',
170+
' "image_url": "https://upload.wikimedia.org/a.jpg",\r\n'
171+
' "image_license": "CC-BY-4.0",\r\n'
172+
' "image_attribution": "Alice",\r\n'
173+
' "name":',
174+
1,
175+
)

0 commit comments

Comments
 (0)