Skip to content

Commit 9f6b98a

Browse files
committed
fix(verify): keep the query in URL dedupe and cache every citation of a checked page
dedupe_urls keyed on host and path only, so cpubenchmark.net/cpu.php?cpu=X pages collapsed into one representative and 762 green CPUs never had their own citation checked. Pages cited under different #fragments were checked once, but the verdict was cached only under the representative URL, which promotion looks up by exact URL. The key now includes the query, and each result is stored for every citation of the same resource. Refs #98 Refs GetTechAPI/TechAPI#297
1 parent ccb89ee commit 9f6b98a

3 files changed

Lines changed: 38 additions & 9 deletions

File tree

‎app/verify/cli.py‎

Lines changed: 11 additions & 1 deletion
Original file line numberDiff line numberDiff line change
@@ -462,9 +462,19 @@ def _fresh(u: str) -> bool:
462462
# A rate-limited answer is not a verdict — leave it out so the next run asks
463463
# again instead of parking the URL as dead for the whole TTL.
464464
throttled = sum(1 for r in results if r.transient)
465+
# Promotion looks citations up by exact URL, so every citation of a checked
466+
# resource (e.g. other #fragments of the same page) gets the verdict.
467+
by_key: dict[tuple[str, str, str], list[str]] = defaultdict(list)
468+
for u in urls:
469+
key = http_check.url_key(u)
470+
if key is not None:
471+
by_key[key].append(u)
465472
for r in results:
466473
if not r.transient:
467-
cache[r.url] = http_check.result_to_entry(r, ts)
474+
entry = http_check.result_to_entry(r, ts)
475+
key = http_check.url_key(r.url)
476+
for u in (by_key.get(key, []) if key is not None else []) or [r.url]:
477+
cache[u] = entry
468478
http_check.save_cache(cache)
469479
if throttled:
470480
print(

‎app/verify/http_check.py‎

Lines changed: 17 additions & 8 deletions
Original file line numberDiff line numberDiff line change
@@ -244,16 +244,25 @@ def wait(self, host: str) -> None:
244244
# --- batch driver ----------------------------------------------------------------
245245

246246

247+
def url_key(u: str) -> tuple[str, str, str] | None:
248+
"""Identity of the fetched resource: host, path and query; the fragment is ignored."""
249+
try:
250+
p = urlparse(u)
251+
except ValueError:
252+
return None
253+
return (p.netloc.lower(), p.path.rstrip("/"), p.query)
254+
255+
247256
def dedupe_urls(urls: Iterable[str]) -> list[str]:
248-
"""Collapse to one representative per (host, path) — kaggle dumps share a URL."""
249-
seen: dict[tuple[str, str], str] = {}
257+
"""Collapse to one representative per fetched resource — kaggle dumps share a URL.
258+
259+
The query stays in the key (``cpu.php?cpu=X`` pages are distinct resources).
260+
"""
261+
seen: dict[tuple[str, str, str], str] = {}
250262
for u in urls:
251-
try:
252-
p = urlparse(u)
253-
except Exception:
254-
continue
255-
key = (p.netloc.lower(), p.path.rstrip("/"))
256-
seen.setdefault(key, u)
263+
key = url_key(u)
264+
if key is not None:
265+
seen.setdefault(key, u)
257266
return list(seen.values())
258267

259268

‎tests/verify/test_http_check.py‎

Lines changed: 10 additions & 0 deletions
Original file line numberDiff line numberDiff line change
@@ -76,6 +76,16 @@ def test_dedupe_by_host_and_path():
7676
assert len(http_check.dedupe_urls(urls)) == 2
7777

7878

79+
def test_dedupe_keeps_query_and_ignores_fragment():
80+
urls = [
81+
"https://www.cpubenchmark.net/cpu.php?cpu=A",
82+
"https://www.cpubenchmark.net/cpu.php?cpu=B",
83+
"https://en.wikipedia.org/wiki/List#One",
84+
"https://en.wikipedia.org/wiki/List#Two",
85+
]
86+
assert http_check.dedupe_urls(urls) == urls[:3]
87+
88+
7989
def test_cache_freshness():
8090
from datetime import datetime
8191
now = datetime(2026, 6, 22, tzinfo=UTC)

0 commit comments

Comments
 (0)