From d8a3e42722f70d4ed2bfc6b6538123f4131f7d51 Mon Sep 17 00:00:00 2001 From: Seungpyo1007 Date: Sun, 27 Sep 2026 14:28:44 +0900 Subject: [PATCH 1/2] feat(verify): score software and websites with batched Wikidata liveness Refs #98 Refs GetTechAPI/TechAPI#297 --- .github/workflows/verify-network.yml | 13 ++++ app/verify/cli.py | 44 ++++++++++++-- app/verify/offline.py | 3 + app/verify/signals.py | 58 ++++++++++++++++++ app/verify/wikidata.py | 90 ++++++++++++++++++++++++++++ tests/unit/test_bot_coverage.py | 2 +- tests/verify/test_digital_scoring.py | 61 +++++++++++++++++++ tests/verify/test_wikidata.py | 89 +++++++++++++++++++++++++++ 8 files changed, 355 insertions(+), 5 deletions(-) create mode 100644 app/verify/wikidata.py create mode 100644 tests/verify/test_digital_scoring.py create mode 100644 tests/verify/test_wikidata.py diff --git a/.github/workflows/verify-network.yml b/.github/workflows/verify-network.yml index 7389035..12c1df0 100644 --- a/.github/workflows/verify-network.yml +++ b/.github/workflows/verify-network.yml @@ -15,6 +15,9 @@ on: max_urls: description: "Frontier records to URL-check" default: "2000" + max_wikidata: + description: "Maximum Wikidata QIDs to batch-check" + default: "20000" max_crossref: description: "Records to cross-reference" default: "500" @@ -59,12 +62,22 @@ jobs: - name: Install TechEngine run: pip install -e . + - name: Restore source URL cache + uses: actions/cache@v4 + with: + path: TechAPI/data/_verify/state/url_cache.jsonl + key: verify-url-cache-${{ github.run_id }}-${{ github.run_attempt }} + restore-keys: verify-url-cache- + - name: Tier 0 score (recomputed; no committed cache) run: python -m app.verify score --no-cache - name: Tier 1 source-URL liveness run: python -m app.verify check-urls --max ${{ github.event.inputs.max_urls || '2000' }} + - name: Batched Wikidata source liveness + run: python -m app.verify check-wikidata --max ${{ github.event.inputs.max_wikidata || '20000' }} + - name: Tier 2 external cross-reference run: python -m app.verify crossref --max ${{ github.event.inputs.max_crossref || '500' }} diff --git a/app/verify/cli.py b/app/verify/cli.py index cdd224a..be32aee 100644 --- a/app/verify/cli.py +++ b/app/verify/cli.py @@ -22,7 +22,7 @@ from app.validate import DATA_DIR -from . import crossref, http_check, ledger, offline, promote +from . import crossref, http_check, ledger, offline, promote, wikidata from .common import ( CATEGORIES, SCORES_PATH, @@ -198,8 +198,6 @@ def _print_markdown(hist: dict[str, Counter[str]], scored: int, hard_flags: Coun ) gtot = sum(totals.values()) or 1 print(f"**{scored} record(s) assessed.**\n") - print("Software and website assess required fields and sources only; " - "domain consistency rules are unavailable and these categories cannot earn green.\n") # Overall distribution as a Mermaid pie (rendered by GitHub). Mermaid colors # slices pie1/pie2/pie3 in declaration order, so pin them to green/amber/red @@ -384,6 +382,34 @@ def _ranked_unverified( return [rec for _score, rec in scored] +def cmd_check_wikidata(args: argparse.Namespace) -> int: + records = load_all(args.category or ("software", "website")) + cache = http_check.load_cache() + now = datetime.now(UTC) + grouped: dict[str, list[str]] = {} + for rows in records.values(): + for rec in rows: + for url in rec.data.get("source_urls", []): + qid = wikidata.qid_of(url) + if qid and (args.recheck or url not in cache + or not str(cache[url].get("reason", "")).startswith("wikidata-") + or not http_check.is_fresh(cache[url], now, args.ttl_days)): + grouped.setdefault(qid, []).append(url) + selected = list(grouped)[:args.max] + urls = list(dict.fromkeys(u for qid in selected for u in grouped[qid])) + checked = alive = 0 + for results in wikidata.check_batches(urls): + for result in results: + cache[result.url] = http_check.result_to_entry(result, _now_iso()) + checked += 1 + alive += result.alive + if results: + http_check.save_cache(cache) + print(f"check-wikidata: {len(selected)} QIDs; {checked} URLs checked, {alive} alive; " + "indeterminate results remain uncached") + return 0 + + def cmd_check_urls(args: argparse.Namespace) -> int: records = load_all() _, _, soc_release = foreign_key_sets(records) @@ -418,10 +444,13 @@ def cmd_check_urls(args: argparse.Namespace) -> int: ts = _now_iso() results = http_check.check_urls( - todo, + [u for u in todo if not wikidata.qid_of(u)], max_workers=args.workers, min_interval=args.min_interval, ) + # A redirect entity must not become alive again through a generic HTTP 200. + for batch in wikidata.check_batches([u for u in todo if wikidata.qid_of(u)]): + results.extend(batch) # A rate-limited answer is not a verdict — leave it out so the next run asks # again instead of parking the URL as dead for the whole TTL. throttled = sum(1 for r in results if r.transient) @@ -740,6 +769,13 @@ def build_parser() -> argparse.ArgumentParser: cu.add_argument("--recheck", action="store_true", help="ignore cache freshness") cu.set_defaults(func=cmd_check_urls) + wd = sub.add_parser("check-wikidata", help="Batched Wikidata source liveness") + wd.add_argument("--category", nargs="*", choices=CATEGORIES) + wd.add_argument("--max", type=int, default=20000, help="maximum uncached QIDs") + wd.add_argument("--ttl-days", type=int, default=http_check.DEFAULT_TTL_DAYS) + wd.add_argument("--recheck", action="store_true") + wd.set_defaults(func=cmd_check_wikidata) + cr = sub.add_parser("crossref", help="Tier 2: external cross-reference (exact heading)") cr.add_argument("--category", nargs="*", choices=CATEGORIES, help="limit to categories") cr.add_argument("--max", type=int, default=200, help="number of yellow/red records to escalate") diff --git a/app/verify/offline.py b/app/verify/offline.py index 62385b8..a0e88d7 100644 --- a/app/verify/offline.py +++ b/app/verify/offline.py @@ -38,6 +38,9 @@ # "Rich" fields per category: presence (non-null) signals a fleshed-out record. # Dotted paths index into nested dicts (e.g. "display.ppi"). RICH_FIELDS: dict[str, tuple[str, ...]] = { + "software": ("release_date", "developers", "operating_systems", "licenses", "genres", + "programming_languages", "publishers"), + "website": ("homepage_url", "launch_date", "languages", "owners"), "cpu": ("architecture", "base_clock_ghz", "boost_clock_ghz", "l3_cache_mb", "socket", "tdp_w", "passmark_cpu_mark"), "gpu": ("architecture", "boost_clock_mhz", "memory_type", "memory_bandwidth_gbps", diff --git a/app/verify/signals.py b/app/verify/signals.py index ab2803f..5ead0f7 100644 --- a/app/verify/signals.py +++ b/app/verify/signals.py @@ -16,7 +16,11 @@ import math import re +from datetime import date from typing import Any, NamedTuple +from urllib.parse import urlparse + +from .wikidata import qid_of # Range table mirrored from app.validate's _check_range call sites, keyed by # (category, field) -> (lo, hi). A parity smoke test asserts this stays in sync. @@ -342,9 +346,63 @@ def monitor_signals(rec: dict[str, Any], now_year: int) -> list[Signal]: ] +def _digital_date(rec: dict[str, Any], field: str, now_year: int, earliest: int) -> Signal: + value = rec.get(field) + name = f"{field}_plausible" + if value in (None, ""): + return Signal(name, "na") + try: + parsed = date.fromisoformat(value) if isinstance(value, str) else None + except ValueError: + parsed = None + if parsed is None: + return Signal(name, "fail", hard=True) + # Imported dates can describe the publisher's founding (290 websites predate + # the Web), or a planned release. These are ambiguous, not impossibilities. + return Signal(name, "pass" if earliest <= parsed.year <= now_year else "fail") + + +def digital_signals(category: str, rec: dict[str, Any], now_year: int) -> list[Signal]: + urls = rec.get("source_urls") + has_qid = isinstance(urls, list) and any(qid_of(u) for u in urls) + out = [Signal("wikidata_qid_source", "pass" if has_qid else "fail")] + fields = ("release_date",) if category == "software" else ( + "launch_date", "release_date", "founded_date", + ) + for field in fields: + earliest = 1950 if category == "software" else (1800 if field == "founded_date" else 1989) + out.append(_digital_date(rec, field, now_year, earliest)) + if category == "software": + for field in ("developers", "operating_systems", "licenses", "genres"): + value = rec.get(field) + valid = isinstance(value, list) and bool(value) and all( + isinstance(v, str) and bool(v.strip()) for v in value + ) + out.append(Signal(f"{field}_string_list", "na" if value is None else ( + "pass" if valid else "fail" + ))) + else: + value = rec.get("homepage_url") + try: + parsed = urlparse(value) if isinstance(value, str) else None + valid = isinstance(value, str) and parsed is not None and parsed.scheme in { + "http", "https", + } and bool( + parsed.hostname + ) and not any(c.isspace() for c in value) + except ValueError: + valid = False + out.append(Signal("homepage_http_url", "na" if value is None else ( + "pass" if valid else "fail" + ))) + return out + + def signals_for( category: str, rec: dict[str, Any], now_year: int, soc_release: dict[str, str] ) -> list[Signal]: + if category in {"software", "website"}: + return digital_signals(category, rec, now_year) if category == "laptop": return laptop_signals(rec, now_year) if category == "monitor": diff --git a/app/verify/wikidata.py b/app/verify/wikidata.py new file mode 100644 index 0000000..2a70c34 --- /dev/null +++ b/app/verify/wikidata.py @@ -0,0 +1,90 @@ +"""Batched entity existence checks, using the promotion URL cache. + +Identity/property cross-reference is deliberately deferred: labels may be aliases +or translations, and source dates can differ in precision or release semantics. +Existence is source liveness, not independent confirmation of a record's claims. +""" + +from __future__ import annotations + +import json +import re +import time +from collections.abc import Callable, Iterator +from typing import Any +from urllib.parse import urlencode, urlparse +from urllib.request import Request, build_opener + +from .http_check import CheckResult + +USER_AGENT = "TechEngine/0.1 (https://github.com/GetTechAPI/TechEngine; source verification)" + + +def qid_of(url: Any) -> str | None: + if not isinstance(url, str): + return None + try: + parsed = urlparse(url) + if (parsed.scheme not in {"http", "https"} + or parsed.netloc.lower() not in {"wikidata.org", "www.wikidata.org"} + or parsed.query or parsed.fragment): + return None + match = re.fullmatch(r"/wiki/(Q[1-9][0-9]*)/?", parsed.path) + return match[1] if match else None + except ValueError: + return None + + +def check_batches( + urls: list[str], *, opener: Any = None, + sleep: Callable[[float], None] = time.sleep, +) -> Iterator[list[CheckResult]]: + """Yield completed batches for incremental persistence; errors stay uncached. + + No redirect resolution is requested: redirect entities are dead citations. + maxlag/API/transport failures are indeterminate and retried next run. + """ + grouped: dict[str, list[str]] = {} + for url in urls: + qid = qid_of(url) + if qid: + grouped.setdefault(qid, []).append(url) + ids = list(grouped) + opener = opener or build_opener() + for start in range(0, len(ids), 50): + if start: + sleep(1.0) + batch = ids[start:start + 50] + params = urlencode({ + "action": "wbgetentities", "ids": "|".join(batch), "props": "info", + "format": "json", "maxlag": "5", + }) + request = Request( + "https://www.wikidata.org/w/api.php?" + params, + headers={"User-Agent": USER_AGENT}, + ) + try: + with opener.open(request, timeout=30) as response: + payload = json.load(response) + if "error" in payload: + sleep(5.0) + yield [] + continue + entities = payload.get("entities", {}) + results: list[CheckResult] = [] + for qid in batch: + entity = entities.get(qid) + if not isinstance(entity, dict): + continue # incomplete/malformed response is not a dead verdict + missing = "missing" in entity + redirected = "redirect" in entity or entity.get("id") != qid + if not missing and not redirected and "lastrevid" not in entity: + continue + alive = not missing and not redirected + reason = "wikidata-entity" if alive else ( + "wikidata-missing" if missing else "wikidata-redirect" + ) + results.extend(CheckResult(url, 200, url, alive, reason) for url in grouped[qid]) + yield results + except (OSError, ValueError, TypeError, AttributeError): + yield [] diff --git a/tests/unit/test_bot_coverage.py b/tests/unit/test_bot_coverage.py index d4415fe..8037347 100644 --- a/tests/unit/test_bot_coverage.py +++ b/tests/unit/test_bot_coverage.py @@ -21,7 +21,7 @@ def test_all_categories_share_registry(): } -@pytest.mark.parametrize("category", ["software", "website", "future"]) +@pytest.mark.parametrize("category", ["future"]) def test_missing_domain_rules_never_earn_green(category): score = offline.score_record(Record(category, "example.json", { "slug": "example", "name": "Example", "source_urls": ["https://intel.com/example"], diff --git a/tests/verify/test_digital_scoring.py b/tests/verify/test_digital_scoring.py new file mode 100644 index 0000000..28dd461 --- /dev/null +++ b/tests/verify/test_digital_scoring.py @@ -0,0 +1,61 @@ +import pytest + +from app.verify.common import Record +from app.verify.offline import score_record +from app.verify.signals import signals_for + + +@pytest.mark.parametrize("category,fields", [ + ("software", {"release_date": "2020-01-01", "developers": ["Developer"], + "operating_systems": ["Linux"], "licenses": ["MIT"], "genres": ["Editor"], + "programming_languages": ["C"], "publishers": ["Publisher"]}), + ("website", {"homepage_url": "https://example.org", "launch_date": "2000-01-01", + "languages": ["English"], "owners": ["Owner"]}), +]) +def test_rich_digital_record_is_green(category, fields): + data = {**fields, "source_urls": ["https://www.wikidata.org/wiki/Q1"]} + score = score_record(Record(category, "example.json", data), 2026, {}) + assert score.band == "green" + assert score.flags == [] + + +@pytest.mark.parametrize("category,field,value", [ + ("software", "release_date", "2020-02-30"), + ("website", "launch_date", "0000-01-01"), +]) +def test_impossible_dates_force_red(category, field, value): + data = {field: value, "source_urls": ["https://www.wikidata.org/wiki/Q1"]} + score = score_record(Record(category, "example.json", data), 2026, {}) + assert score.band == "red" + assert f"!{field}_plausible" in score.flags + + +@pytest.mark.parametrize("category", ["software", "website"]) +def test_missing_qid_is_soft_failure(category): + sigs = signals_for(category, {}, 2026, {}) + assert sigs[0].failed and not sigs[0].hard + assert all(s.result == "na" for s in sigs[1:]) + + +@pytest.mark.parametrize("value", [[], "MIT", [""], [42]]) +def test_invalid_software_lists_are_soft(value): + sig = next(s for s in signals_for("software", {"licenses": value}, 2026, {}) + if s.name == "licenses_string_list") + assert sig.failed and not sig.hard + + +def test_future_dates_and_bad_homepage_are_soft(): + sigs = signals_for("website", {"launch_date": "2099-01-01", + "homepage_url": "https:///broken"}, 2026, {}) + assert all(not s.hard for s in sigs) + assert sum(s.failed for s in sigs) == 3 + + +@pytest.mark.parametrize("category,field,value", [ + ("software", "release_date", "1949-01-01"), + ("website", "launch_date", "1962-01-01"), +]) +def test_early_dates_are_ambiguous_not_impossible(category, field, value): + sig = next(s for s in signals_for(category, {field: value}, 2026, {}) + if s.name == f"{field}_plausible") + assert sig.failed and not sig.hard diff --git a/tests/verify/test_wikidata.py b/tests/verify/test_wikidata.py new file mode 100644 index 0000000..6f66566 --- /dev/null +++ b/tests/verify/test_wikidata.py @@ -0,0 +1,89 @@ +import io +import json +from urllib.parse import parse_qs, urlparse + +import pytest + +from app.verify import http_check, promote, wikidata + + +@pytest.mark.parametrize("url", ["https://evil.org/wiki/Q1", "https://wikidata.org/wiki/Q0", + "https://wikidata.org/wiki/Q1?x=1", "not a URL", None]) +def test_malformed_qid_urls(url): + assert wikidata.qid_of(url) is None + + +class Opener: + def __init__(self, error=False): + self.calls = [] + self.error = error + + def open(self, request, timeout): + params = parse_qs(urlparse(request.full_url).query) + self.calls.append((request, params)) + assert timeout == 30 + assert params["maxlag"] == ["5"] + assert "github.com/GetTechAPI/TechEngine" in request.get_header("User-agent") + ids = params["ids"][0].split("|") + entities = {qid: {"id": qid, "lastrevid": 1} for qid in ids} + if "Q2" in ids: + entities["Q2"] = {"id": "Q2", "missing": ""} + if "Q3" in ids: + entities["Q3"] = {"id": "Q3", "redirect": "Q4"} + payload = {"error": {"code": "maxlag"}} if self.error else {"entities": entities} + return io.StringIO(json.dumps(payload)) + + +def test_batches_dedupe_and_cache_promotion(tmp_path): + urls = [f"https://www.wikidata.org/wiki/Q{i}" for i in range(1, 52)] + urls.append("https://wikidata.org/wiki/Q1") + op = Opener() + sleeps = [] + results = [r for batch in wikidata.check_batches(urls, opener=op, sleep=sleeps.append) + for r in batch] + assert [len(call[1]["ids"][0].split("|")) for call in op.calls] == [50, 1] + assert sleeps == [1.0] + assert len(results) == 52 + assert sum(r.alive for r in results) == 50 + cache = {r.url: http_check.result_to_entry(r, "2026-09-27T00:00:00Z") for r in results} + path = tmp_path / "url_cache.jsonl" + http_check.save_cache(cache, path) + loaded = http_check.load_cache(path) + for qid, expected in [(1, True), (2, False), (3, False)]: + decision = promote.decide(band="green", source_urls=[urls[qid - 1]], url_cache=loaded, + crossref_decision=None) + assert decision.promote is expected + + +def test_api_errors_are_not_cached(): + assert list(wikidata.check_batches(["https://wikidata.org/wiki/Q1"], + opener=Opener(error=True), sleep=lambda _: None)) == [[]] + + +def test_transport_failure_is_not_dead(): + class BrokenOpener: + def open(self, request, timeout): + raise OSError("offline") + + assert list(wikidata.check_batches(["https://wikidata.org/wiki/Q1"], + opener=BrokenOpener(), sleep=lambda _: None)) == [[]] + + +def test_cli_skips_fresh_ids_before_cap(monkeypatch): + from argparse import Namespace + from datetime import UTC, datetime + + from app.verify import cli + from app.verify.common import Record + + urls = [f"https://wikidata.org/wiki/Q{i}" for i in range(1, 4)] + records = [Record("software", f"{i}.json", {"source_urls": [u]}) + for i, u in enumerate(urls)] + cache = {urls[0]: {"checked_at": datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%SZ"), + "reason": "wikidata-entity"}} + monkeypatch.setattr(cli, "load_all", lambda _: {"software": records}) + monkeypatch.setattr(http_check, "load_cache", lambda: cache) + seen = [] + monkeypatch.setattr(wikidata, "check_batches", lambda targets: seen.append(targets) or []) + assert cli.cmd_check_wikidata(Namespace(category=None, recheck=False, ttl_days=30, max=1)) == 0 + assert seen == [[urls[1]]] From 1d75fdda74a070213ef759eec6470a77083d651e Mon Sep 17 00:00:00 2001 From: Seungpyo1007 Date: Sun, 27 Sep 2026 15:40:39 +0900 Subject: [PATCH 2/2] fix(verify): skip fully cached records before capping the check-urls frontier The frontier took the top --max unverified records by score before dropping fresh cached URLs, so the same already-checked records filled the quota on every run and lower-ranked green records were never checked. 848 green records (763 CPUs) had never had a citation checked and could not be promoted. Refs #98 Refs GetTechAPI/TechAPI#297 --- app/verify/cli.py | 18 +++++++++++++----- tests/verify/test_http_check.py | 28 ++++++++++++++++++++++++++++ 2 files changed, 41 insertions(+), 5 deletions(-) diff --git a/app/verify/cli.py b/app/verify/cli.py index be32aee..522c4c6 100644 --- a/app/verify/cli.py +++ b/app/verify/cli.py @@ -416,7 +416,19 @@ def cmd_check_urls(args: argparse.Namespace) -> int: now_year = offline.now_year_today() categories = tuple(args.category) if args.category else CATEGORIES + cache = http_check.load_cache() + now = datetime.now(UTC) + + def _fresh(u: str) -> bool: + return u in cache and http_check.is_fresh(cache[u], now, args.ttl_days) + frontier = _ranked_unverified(records, soc_release, now_year, categories) + if not args.recheck: + # Records whose citations are all fresh would fill the quota every run and + # starve lower-ranked records that were never checked. + frontier = [rec for rec in frontier if not all( + _fresh(u) for u in rec.data.get("source_urls", []) if isinstance(u, str) + )] if args.max is not None: frontier = frontier[: args.max] @@ -425,14 +437,10 @@ def cmd_check_urls(args: argparse.Namespace) -> int: urls.extend(u for u in rec.data.get("source_urls", []) if isinstance(u, str)) targets = http_check.dedupe_urls(urls) - cache = http_check.load_cache() - now = datetime.now(UTC) if args.recheck: todo = targets else: - todo = [u for u in targets if not ( - u in cache and http_check.is_fresh(cache[u], now, args.ttl_days) - )] + todo = [u for u in targets if not _fresh(u)] print( f"check-urls: {len(frontier)} record(s) -> {len(targets)} unique URL(s); " diff --git a/tests/verify/test_http_check.py b/tests/verify/test_http_check.py index 6f3debb..04ecb15 100644 --- a/tests/verify/test_http_check.py +++ b/tests/verify/test_http_check.py @@ -213,3 +213,31 @@ def test_host_backoff_is_capped(): for _ in range(20): limiter.back_off("gsmarena.com") assert limiter.interval_for("gsmarena.com") == http_check.MAX_HOST_INTERVAL_S + + +def test_check_urls_frontier_skips_fully_cached_records(monkeypatch, capsys): + """Already-checked top records must not fill --max and starve unchecked ones.""" + import argparse + from datetime import datetime + + from app.verify import cli + from app.verify.common import Record + + recs = [Record("cpu", f"cpu/{i}.json", + {"slug": f"c{i}", "verified": False, + "source_urls": [f"https://en.wikipedia.org/wiki/C{i}"]}) for i in range(3)] + now = datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%SZ") + cache = {"https://en.wikipedia.org/wiki/C0": {"alive": False, "status": 404, "checked_at": now}} + monkeypatch.setattr(cli, "load_all", lambda *a, **k: {"cpu": recs}) + monkeypatch.setattr(cli, "foreign_key_sets", lambda r: (set(), set(), {})) + monkeypatch.setattr(cli, "_ranked_unverified", lambda *a: list(recs)) + monkeypatch.setattr(http_check, "load_cache", lambda *a: dict(cache)) + monkeypatch.setattr(http_check, "is_fresh", lambda e, now, ttl: True) + checked = [] + monkeypatch.setattr(http_check, "check_urls", + lambda urls, **k: checked.extend(urls) or []) + monkeypatch.setattr(http_check, "save_cache", lambda *a, **k: None) + args = argparse.Namespace(category=["cpu"], max=1, recheck=False, ttl_days=30, + workers=1, min_interval=0.0) + cli.cmd_check_urls(args) + assert checked == ["https://en.wikipedia.org/wiki/C1"]