From 0e6413265398d61ec16e5c37509f2bd6ba7f6458 Mon Sep 17 00:00:00 2001 From: Seungpyo1007 Date: Sun, 27 Sep 2026 16:43:08 +0900 Subject: [PATCH] fix(verify): keep the query in URL dedupe and cache every citation of a checked page dedupe_urls keyed on host and path only, so cpubenchmark.net/cpu.php?cpu=X pages collapsed into one representative and 762 green CPUs never had their own citation checked. Pages cited under different #fragments were checked once, but the verdict was cached only under the representative URL, which promotion looks up by exact URL. The key now includes the query, and each result is stored for every citation of the same resource. Refs #98 Refs GetTechAPI/TechAPI#297 --- app/verify/cli.py | 12 +++++++++++- app/verify/http_check.py | 25 +++++++++++++++++-------- tests/verify/test_http_check.py | 10 ++++++++++ 3 files changed, 38 insertions(+), 9 deletions(-) diff --git a/app/verify/cli.py b/app/verify/cli.py index 522c4c6..fb39657 100644 --- a/app/verify/cli.py +++ b/app/verify/cli.py @@ -462,9 +462,19 @@ def _fresh(u: str) -> bool: # A rate-limited answer is not a verdict — leave it out so the next run asks # again instead of parking the URL as dead for the whole TTL. throttled = sum(1 for r in results if r.transient) + # Promotion looks citations up by exact URL, so every citation of a checked + # resource (e.g. other #fragments of the same page) gets the verdict. + by_key: dict[tuple[str, str, str], list[str]] = defaultdict(list) + for u in urls: + key = http_check.url_key(u) + if key is not None: + by_key[key].append(u) for r in results: if not r.transient: - cache[r.url] = http_check.result_to_entry(r, ts) + entry = http_check.result_to_entry(r, ts) + key = http_check.url_key(r.url) + for u in (by_key.get(key, []) if key is not None else []) or [r.url]: + cache[u] = entry http_check.save_cache(cache) if throttled: print( diff --git a/app/verify/http_check.py b/app/verify/http_check.py index 271a98a..b2050d7 100644 --- a/app/verify/http_check.py +++ b/app/verify/http_check.py @@ -244,16 +244,25 @@ def wait(self, host: str) -> None: # --- batch driver ---------------------------------------------------------------- +def url_key(u: str) -> tuple[str, str, str] | None: + """Identity of the fetched resource: host, path and query; the fragment is ignored.""" + try: + p = urlparse(u) + except ValueError: + return None + return (p.netloc.lower(), p.path.rstrip("/"), p.query) + + def dedupe_urls(urls: Iterable[str]) -> list[str]: - """Collapse to one representative per (host, path) — kaggle dumps share a URL.""" - seen: dict[tuple[str, str], str] = {} + """Collapse to one representative per fetched resource — kaggle dumps share a URL. + + The query stays in the key (``cpu.php?cpu=X`` pages are distinct resources). + """ + seen: dict[tuple[str, str, str], str] = {} for u in urls: - try: - p = urlparse(u) - except Exception: - continue - key = (p.netloc.lower(), p.path.rstrip("/")) - seen.setdefault(key, u) + key = url_key(u) + if key is not None: + seen.setdefault(key, u) return list(seen.values()) diff --git a/tests/verify/test_http_check.py b/tests/verify/test_http_check.py index 04ecb15..3659c90 100644 --- a/tests/verify/test_http_check.py +++ b/tests/verify/test_http_check.py @@ -76,6 +76,16 @@ def test_dedupe_by_host_and_path(): assert len(http_check.dedupe_urls(urls)) == 2 +def test_dedupe_keeps_query_and_ignores_fragment(): + urls = [ + "https://www.cpubenchmark.net/cpu.php?cpu=A", + "https://www.cpubenchmark.net/cpu.php?cpu=B", + "https://en.wikipedia.org/wiki/List#One", + "https://en.wikipedia.org/wiki/List#Two", + ] + assert http_check.dedupe_urls(urls) == urls[:3] + + def test_cache_freshness(): from datetime import datetime now = datetime(2026, 6, 22, tzinfo=UTC)