Skip to content

Commit 294a495

Browse files
committed
Add Wikipedia watch and PDA backfill categories
1 parent 5e8a17c commit 294a495

2 files changed

Lines changed: 118 additions & 11 deletions

File tree

‎app/verify/wikipedia_smartphone_backfill.py‎

Lines changed: 44 additions & 11 deletions
Original file line numberDiff line numberDiff line change
@@ -96,6 +96,35 @@
9696
("google", "Google_Pixel_Tablet", "Google Pixel Tablet"),
9797
)
9898

99+
WATCH_CROSSREF_PAGES: tuple[tuple[str, str, str], ...] = (
100+
("apple", "Apple_Watch", "Apple Watch"),
101+
("samsung", "Samsung_Galaxy_Watch_series", "Samsung Galaxy Watch series"),
102+
("samsung", "Samsung_Gear", "Samsung Gear"),
103+
("google", "Pixel_Watch", "Pixel Watch"),
104+
("fitbit", "List_of_Fitbit_products", "List of Fitbit products"),
105+
("garmin", "Garmin_Forerunner", "Garmin Forerunner"),
106+
("huawei", "Huawei_Watch", "Huawei Watch"),
107+
("pebble", "Pebble_(watch)", "Pebble (watch)"),
108+
)
109+
110+
PDA_CROSSREF_PAGES: tuple[tuple[str, str, str], ...] = (
111+
("palm", "Palm_(companion)", "Palm (companion)"),
112+
("palm", "Palm_Treo", "Palm Treo"),
113+
("hp", "HP_iPAQ", "HP iPAQ"),
114+
("htc", "List_of_HTC_devices", "List of HTC devices"),
115+
("blackberry", "BlackBerry", "BlackBerry"),
116+
("sony", "CLIÉ", "CLIÉ"),
117+
("casio", "Casio_Cassiopeia", "Casio Cassiopeia"),
118+
("dell", "Dell_Axim", "Dell Axim"),
119+
)
120+
121+
CATEGORY_CROSSREF_PAGES = {
122+
"smartphone": CROSSREF_PAGES,
123+
"tablet": TABLET_CROSSREF_PAGES,
124+
"watch": WATCH_CROSSREF_PAGES,
125+
"pda": PDA_CROSSREF_PAGES,
126+
}
127+
99128
_BRAND_TOKENS = (
100129
"samsung",
101130
"apple",
@@ -121,6 +150,13 @@
121150
"meizu",
122151
"infinix",
123152
"tecno",
153+
"palm",
154+
"garmin",
155+
"fitbit",
156+
"pebble",
157+
"hp",
158+
"dell",
159+
"casio",
124160
)
125161

126162
# Header rules for parsing smartphone tables in list articles.
@@ -1182,10 +1218,10 @@ def backfill(
11821218
cache_path: Path | None = None,
11831219
category: str = "smartphone",
11841220
) -> RunResult:
1185-
if category not in {"smartphone", "tablet"}:
1221+
if category not in CATEGORY_CROSSREF_PAGES:
11861222
raise ValueError(f"unsupported category: {category}")
11871223
writing = apply and not dry_run
1188-
html_cache = data_root / "data" / "_verify" / "cache" / "wikipedia_html"
1224+
html_cache = None if dry_run else data_root / "data" / "_verify" / "cache" / "wikipedia_html"
11891225
polite = PoliteWiki(sleep_s=sleep_s, cache_dir=html_cache) if fetch_page is None else None
11901226
fetch = fetch_page or (polite.fetch if polite is not None else None)
11911227
assert fetch is not None
@@ -1198,9 +1234,7 @@ def backfill(
11981234
parsed_pages: set[str] = set()
11991235

12001236
# Pre-parse list pages
1201-
target_pages = pages if pages is not None else (
1202-
TABLET_CROSSREF_PAGES if category == "tablet" else CROSSREF_PAGES
1203-
)
1237+
target_pages = pages if pages is not None else CATEGORY_CROSSREF_PAGES[category]
12041238
for _mfg, page, _title in target_pages:
12051239
status, final, html = fetch(page)
12061240
_alive, reason = classify(f"https://en.wikipedia.org/wiki/{page}", status, final or None)
@@ -1259,7 +1293,7 @@ def consider_fallback(base: str, record_brand: str = "") -> None:
12591293
if records is None:
12601294
phone_dir, repo_root = smartphone_scan_root(data_root, category)
12611295
chosen = sample_diverse_records(phone_dir, repo_root, limit, exclude_paths=cached_paths)
1262-
if category == "tablet" and limit is not None and len(chosen) < limit:
1296+
if category != "smartphone" and limit is not None and len(chosen) < limit:
12631297
remaining = sample_diverse_records(
12641298
phone_dir,
12651299
repo_root,
@@ -1325,12 +1359,11 @@ def maybe_write(rel: str, decision: object, url: object) -> None:
13251359
consider_fallback(phone.base, record_brand=rec_b)
13261360
hits = matching_rows(name, fetcher.rows, record_brand=rec_b)
13271361

1328-
if category == "tablet":
1362+
if category != "smartphone":
13291363
# A shared substring or a brand-omitted article is insufficient for
1330-
# the tablet batch: the article heading must name this exact model.
1364+
# these category batches: the article heading must name this exact model.
13311365
hits = [
1332-
row for row in hits
1333-
if normalize_heading(phone.base) == normalize_heading(row.model)
1366+
row for row in hits if normalize_heading(phone.base) == normalize_heading(row.model)
13341367
]
13351368

13361369
live = _liveness_for(hits[0].url if hits else None, page_liveness)
@@ -1415,7 +1448,7 @@ def render_summary(
14151448
def main(argv: list[str] | None = None) -> int:
14161449
parser = argparse.ArgumentParser(description=__doc__)
14171450
parser.add_argument("--data-root", type=Path, default=Path("."), help="TechAPI repository root")
1418-
parser.add_argument("--category", choices=("smartphone", "tablet"), default="smartphone")
1451+
parser.add_argument("--category", choices=tuple(CATEGORY_CROSSREF_PAGES), default="smartphone")
14191452
parser.add_argument("--limit", type=int, default=300, help="Max records to process")
14201453
parser.add_argument(
14211454
"--sleep", type=float, default=MIN_SLEEP_S, help="Sleep between Wikipedia calls"

‎tests/verify/test_wikipedia_smartphone_backfill.py‎

Lines changed: 74 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -5,8 +5,13 @@
55
import json
66
from pathlib import Path
77

8+
import pytest
9+
10+
from app.verify import wikipedia_smartphone_backfill as wiki_backfill
811
from app.verify.crossref import _heading_matches
912
from app.verify.wikipedia_smartphone_backfill import (
13+
PDA_CROSSREF_PAGES,
14+
WATCH_CROSSREF_PAGES,
1015
WikiRow,
1116
backfill,
1217
decide,
@@ -241,3 +246,72 @@ def test_apply_writes_only_confirmed_records(tmp_path: Path) -> None:
241246
"https://en.wikipedia.org/wiki/List_of_Samsung_Galaxy_smartphones#Galaxy_S_series"
242247
in updated["source_urls"]
243248
)
249+
250+
251+
@pytest.mark.parametrize(
252+
("category", "pages", "brand", "model"),
253+
[
254+
("watch", WATCH_CROSSREF_PAGES, "apple", "Apple Watch Series 6"),
255+
("pda", PDA_CROSSREF_PAGES, "dell", "Dell Axim X5"),
256+
],
257+
)
258+
def test_category_pages_and_exact_heading(
259+
tmp_path: Path,
260+
category: str,
261+
pages: tuple[tuple[str, str, str], ...],
262+
brand: str,
263+
model: str,
264+
) -> None:
265+
page = next(page for page_brand, page, _ in pages if page_brand == brand)
266+
html = f"""<table class="wikitable"><tr><th>Model</th><th>Released</th>
267+
<th>RAM</th><th>Battery</th></tr><tr><td>{model}</td><td>2020</td>
268+
<td>8 GB</td><td>4000 mAh</td></tr></table>"""
269+
fetched: list[str] = []
270+
271+
def fetch(candidate: str) -> tuple[int, str, str]:
272+
fetched.append(candidate)
273+
return 200, f"https://en.wikipedia.org/wiki/{candidate}", html if candidate == page else ""
274+
275+
record = _sample_rec(name=model, brand=brand)
276+
result = backfill(
277+
tmp_path,
278+
category=category,
279+
records=[(f"data/{category}/{brand}/model.json", record)],
280+
fetch_page=fetch,
281+
search_fn=lambda _name: [],
282+
cache_path=tmp_path / "cache.jsonl",
283+
)
284+
assert fetched == [item[1] for item in pages]
285+
assert result.counts()["confirm"] == 1
286+
assert result.written == 0
287+
288+
near_match = {**record, "name": f"{model} Pro"}
289+
rejected = backfill(
290+
tmp_path,
291+
category=category,
292+
records=[(f"data/{category}/{brand}/near.json", near_match)],
293+
pages=[(brand, page, model)],
294+
fetch_page=fetch,
295+
search_fn=lambda _name: [],
296+
)
297+
assert rejected.counts()["confirm"] == 0
298+
299+
300+
@pytest.mark.parametrize("category", ["watch", "pda"])
301+
def test_category_cli_uses_category_cache(
302+
tmp_path: Path, monkeypatch: pytest.MonkeyPatch, category: str
303+
) -> None:
304+
captured: dict[str, object] = {}
305+
306+
def fake_backfill(data_root: Path, **kwargs: object) -> wiki_backfill.RunResult:
307+
captured.update(kwargs)
308+
return wiki_backfill.RunResult()
309+
310+
monkeypatch.setattr(wiki_backfill, "backfill", fake_backfill)
311+
assert wiki_backfill.main(["--data-root", str(tmp_path), "--category", category]) == 0
312+
assert captured["category"] == category
313+
assert (
314+
captured["cache_path"]
315+
== tmp_path / "data" / "_verify" / "state" / f"wikipedia_{category}_cache.jsonl"
316+
)
317+
assert captured["apply"] is False

0 commit comments

Comments
 (0)