Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
18 changes: 13 additions & 5 deletions app/verify/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -416,7 +416,19 @@ def cmd_check_urls(args: argparse.Namespace) -> int:
now_year = offline.now_year_today()
categories = tuple(args.category) if args.category else CATEGORIES

cache = http_check.load_cache()
now = datetime.now(UTC)

def _fresh(u: str) -> bool:
return u in cache and http_check.is_fresh(cache[u], now, args.ttl_days)

frontier = _ranked_unverified(records, soc_release, now_year, categories)
if not args.recheck:
# Records whose citations are all fresh would fill the quota every run and
# starve lower-ranked records that were never checked.
frontier = [rec for rec in frontier if not all(
_fresh(u) for u in rec.data.get("source_urls", []) if isinstance(u, str)
)]
if args.max is not None:
frontier = frontier[: args.max]

Expand All @@ -425,14 +437,10 @@ def cmd_check_urls(args: argparse.Namespace) -> int:
urls.extend(u for u in rec.data.get("source_urls", []) if isinstance(u, str))
targets = http_check.dedupe_urls(urls)

cache = http_check.load_cache()
now = datetime.now(UTC)
if args.recheck:
todo = targets
else:
todo = [u for u in targets if not (
u in cache and http_check.is_fresh(cache[u], now, args.ttl_days)
)]
todo = [u for u in targets if not _fresh(u)]

print(
f"check-urls: {len(frontier)} record(s) -> {len(targets)} unique URL(s); "
Expand Down
28 changes: 28 additions & 0 deletions tests/verify/test_http_check.py
Original file line number Diff line number Diff line change
Expand Up @@ -213,3 +213,31 @@ def test_host_backoff_is_capped():
for _ in range(20):
limiter.back_off("gsmarena.com")
assert limiter.interval_for("gsmarena.com") == http_check.MAX_HOST_INTERVAL_S


def test_check_urls_frontier_skips_fully_cached_records(monkeypatch, capsys):
"""Already-checked top records must not fill --max and starve unchecked ones."""
import argparse
from datetime import datetime

from app.verify import cli
from app.verify.common import Record

recs = [Record("cpu", f"cpu/{i}.json",
{"slug": f"c{i}", "verified": False,
"source_urls": [f"https://en.wikipedia.org/wiki/C{i}"]}) for i in range(3)]
now = datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%SZ")
cache = {"https://en.wikipedia.org/wiki/C0": {"alive": False, "status": 404, "checked_at": now}}
monkeypatch.setattr(cli, "load_all", lambda *a, **k: {"cpu": recs})
monkeypatch.setattr(cli, "foreign_key_sets", lambda r: (set(), set(), {}))
monkeypatch.setattr(cli, "_ranked_unverified", lambda *a: list(recs))
monkeypatch.setattr(http_check, "load_cache", lambda *a: dict(cache))
monkeypatch.setattr(http_check, "is_fresh", lambda e, now, ttl: True)
checked = []
monkeypatch.setattr(http_check, "check_urls",
lambda urls, **k: checked.extend(urls) or [])
monkeypatch.setattr(http_check, "save_cache", lambda *a, **k: None)
args = argparse.Namespace(category=["cpu"], max=1, recheck=False, ttl_days=30,
workers=1, min_interval=0.0)
cli.cmd_check_urls(args)
assert checked == ["https://en.wikipedia.org/wiki/C1"]
Loading