diff --git a/app/verify/cli.py b/app/verify/cli.py index 522c4c6..fb39657 100644 --- a/app/verify/cli.py +++ b/app/verify/cli.py @@ -462,9 +462,19 @@ def _fresh(u: str) -> bool: # A rate-limited answer is not a verdict — leave it out so the next run asks # again instead of parking the URL as dead for the whole TTL. throttled = sum(1 for r in results if r.transient) + # Promotion looks citations up by exact URL, so every citation of a checked + # resource (e.g. other #fragments of the same page) gets the verdict. + by_key: dict[tuple[str, str, str], list[str]] = defaultdict(list) + for u in urls: + key = http_check.url_key(u) + if key is not None: + by_key[key].append(u) for r in results: if not r.transient: - cache[r.url] = http_check.result_to_entry(r, ts) + entry = http_check.result_to_entry(r, ts) + key = http_check.url_key(r.url) + for u in (by_key.get(key, []) if key is not None else []) or [r.url]: + cache[u] = entry http_check.save_cache(cache) if throttled: print( diff --git a/app/verify/http_check.py b/app/verify/http_check.py index 271a98a..b2050d7 100644 --- a/app/verify/http_check.py +++ b/app/verify/http_check.py @@ -244,16 +244,25 @@ def wait(self, host: str) -> None: # --- batch driver ---------------------------------------------------------------- +def url_key(u: str) -> tuple[str, str, str] | None: + """Identity of the fetched resource: host, path and query; the fragment is ignored.""" + try: + p = urlparse(u) + except ValueError: + return None + return (p.netloc.lower(), p.path.rstrip("/"), p.query) + + def dedupe_urls(urls: Iterable[str]) -> list[str]: - """Collapse to one representative per (host, path) — kaggle dumps share a URL.""" - seen: dict[tuple[str, str], str] = {} + """Collapse to one representative per fetched resource — kaggle dumps share a URL. + + The query stays in the key (``cpu.php?cpu=X`` pages are distinct resources). + """ + seen: dict[tuple[str, str, str], str] = {} for u in urls: - try: - p = urlparse(u) - except Exception: - continue - key = (p.netloc.lower(), p.path.rstrip("/")) - seen.setdefault(key, u) + key = url_key(u) + if key is not None: + seen.setdefault(key, u) return list(seen.values()) diff --git a/tests/verify/test_http_check.py b/tests/verify/test_http_check.py index 04ecb15..3659c90 100644 --- a/tests/verify/test_http_check.py +++ b/tests/verify/test_http_check.py @@ -76,6 +76,16 @@ def test_dedupe_by_host_and_path(): assert len(http_check.dedupe_urls(urls)) == 2 +def test_dedupe_keeps_query_and_ignores_fragment(): + urls = [ + "https://www.cpubenchmark.net/cpu.php?cpu=A", + "https://www.cpubenchmark.net/cpu.php?cpu=B", + "https://en.wikipedia.org/wiki/List#One", + "https://en.wikipedia.org/wiki/List#Two", + ] + assert http_check.dedupe_urls(urls) == urls[:3] + + def test_cache_freshness(): from datetime import datetime now = datetime(2026, 6, 22, tzinfo=UTC)