diff --git a/.github/workflows/verify-network.yml b/.github/workflows/verify-network.yml index 7389035..12c1df0 100644 --- a/.github/workflows/verify-network.yml +++ b/.github/workflows/verify-network.yml @@ -15,6 +15,9 @@ on: max_urls: description: "Frontier records to URL-check" default: "2000" + max_wikidata: + description: "Maximum Wikidata QIDs to batch-check" + default: "20000" max_crossref: description: "Records to cross-reference" default: "500" @@ -59,12 +62,22 @@ jobs: - name: Install TechEngine run: pip install -e . + - name: Restore source URL cache + uses: actions/cache@v4 + with: + path: TechAPI/data/_verify/state/url_cache.jsonl + key: verify-url-cache-${{ github.run_id }}-${{ github.run_attempt }} + restore-keys: verify-url-cache- + - name: Tier 0 score (recomputed; no committed cache) run: python -m app.verify score --no-cache - name: Tier 1 source-URL liveness run: python -m app.verify check-urls --max ${{ github.event.inputs.max_urls || '2000' }} + - name: Batched Wikidata source liveness + run: python -m app.verify check-wikidata --max ${{ github.event.inputs.max_wikidata || '20000' }} + - name: Tier 2 external cross-reference run: python -m app.verify crossref --max ${{ github.event.inputs.max_crossref || '500' }} diff --git a/app/verify/cli.py b/app/verify/cli.py index cdd224a..be32aee 100644 --- a/app/verify/cli.py +++ b/app/verify/cli.py @@ -22,7 +22,7 @@ from app.validate import DATA_DIR -from . import crossref, http_check, ledger, offline, promote +from . import crossref, http_check, ledger, offline, promote, wikidata from .common import ( CATEGORIES, SCORES_PATH, @@ -198,8 +198,6 @@ def _print_markdown(hist: dict[str, Counter[str]], scored: int, hard_flags: Coun ) gtot = sum(totals.values()) or 1 print(f"**{scored} record(s) assessed.**\n") - print("Software and website assess required fields and sources only; " - "domain consistency rules are unavailable and these categories cannot earn green.\n") # Overall distribution as a Mermaid pie (rendered by GitHub). Mermaid colors # slices pie1/pie2/pie3 in declaration order, so pin them to green/amber/red @@ -384,6 +382,34 @@ def _ranked_unverified( return [rec for _score, rec in scored] +def cmd_check_wikidata(args: argparse.Namespace) -> int: + records = load_all(args.category or ("software", "website")) + cache = http_check.load_cache() + now = datetime.now(UTC) + grouped: dict[str, list[str]] = {} + for rows in records.values(): + for rec in rows: + for url in rec.data.get("source_urls", []): + qid = wikidata.qid_of(url) + if qid and (args.recheck or url not in cache + or not str(cache[url].get("reason", "")).startswith("wikidata-") + or not http_check.is_fresh(cache[url], now, args.ttl_days)): + grouped.setdefault(qid, []).append(url) + selected = list(grouped)[:args.max] + urls = list(dict.fromkeys(u for qid in selected for u in grouped[qid])) + checked = alive = 0 + for results in wikidata.check_batches(urls): + for result in results: + cache[result.url] = http_check.result_to_entry(result, _now_iso()) + checked += 1 + alive += result.alive + if results: + http_check.save_cache(cache) + print(f"check-wikidata: {len(selected)} QIDs; {checked} URLs checked, {alive} alive; " + "indeterminate results remain uncached") + return 0 + + def cmd_check_urls(args: argparse.Namespace) -> int: records = load_all() _, _, soc_release = foreign_key_sets(records) @@ -418,10 +444,13 @@ def cmd_check_urls(args: argparse.Namespace) -> int: ts = _now_iso() results = http_check.check_urls( - todo, + [u for u in todo if not wikidata.qid_of(u)], max_workers=args.workers, min_interval=args.min_interval, ) + # A redirect entity must not become alive again through a generic HTTP 200. + for batch in wikidata.check_batches([u for u in todo if wikidata.qid_of(u)]): + results.extend(batch) # A rate-limited answer is not a verdict — leave it out so the next run asks # again instead of parking the URL as dead for the whole TTL. throttled = sum(1 for r in results if r.transient) @@ -740,6 +769,13 @@ def build_parser() -> argparse.ArgumentParser: cu.add_argument("--recheck", action="store_true", help="ignore cache freshness") cu.set_defaults(func=cmd_check_urls) + wd = sub.add_parser("check-wikidata", help="Batched Wikidata source liveness") + wd.add_argument("--category", nargs="*", choices=CATEGORIES) + wd.add_argument("--max", type=int, default=20000, help="maximum uncached QIDs") + wd.add_argument("--ttl-days", type=int, default=http_check.DEFAULT_TTL_DAYS) + wd.add_argument("--recheck", action="store_true") + wd.set_defaults(func=cmd_check_wikidata) + cr = sub.add_parser("crossref", help="Tier 2: external cross-reference (exact heading)") cr.add_argument("--category", nargs="*", choices=CATEGORIES, help="limit to categories") cr.add_argument("--max", type=int, default=200, help="number of yellow/red records to escalate") diff --git a/app/verify/offline.py b/app/verify/offline.py index 62385b8..a0e88d7 100644 --- a/app/verify/offline.py +++ b/app/verify/offline.py @@ -38,6 +38,9 @@ # "Rich" fields per category: presence (non-null) signals a fleshed-out record. # Dotted paths index into nested dicts (e.g. "display.ppi"). RICH_FIELDS: dict[str, tuple[str, ...]] = { + "software": ("release_date", "developers", "operating_systems", "licenses", "genres", + "programming_languages", "publishers"), + "website": ("homepage_url", "launch_date", "languages", "owners"), "cpu": ("architecture", "base_clock_ghz", "boost_clock_ghz", "l3_cache_mb", "socket", "tdp_w", "passmark_cpu_mark"), "gpu": ("architecture", "boost_clock_mhz", "memory_type", "memory_bandwidth_gbps", diff --git a/app/verify/signals.py b/app/verify/signals.py index ab2803f..5ead0f7 100644 --- a/app/verify/signals.py +++ b/app/verify/signals.py @@ -16,7 +16,11 @@ import math import re +from datetime import date from typing import Any, NamedTuple +from urllib.parse import urlparse + +from .wikidata import qid_of # Range table mirrored from app.validate's _check_range call sites, keyed by # (category, field) -> (lo, hi). A parity smoke test asserts this stays in sync. @@ -342,9 +346,63 @@ def monitor_signals(rec: dict[str, Any], now_year: int) -> list[Signal]: ] +def _digital_date(rec: dict[str, Any], field: str, now_year: int, earliest: int) -> Signal: + value = rec.get(field) + name = f"{field}_plausible" + if value in (None, ""): + return Signal(name, "na") + try: + parsed = date.fromisoformat(value) if isinstance(value, str) else None + except ValueError: + parsed = None + if parsed is None: + return Signal(name, "fail", hard=True) + # Imported dates can describe the publisher's founding (290 websites predate + # the Web), or a planned release. These are ambiguous, not impossibilities. + return Signal(name, "pass" if earliest <= parsed.year <= now_year else "fail") + + +def digital_signals(category: str, rec: dict[str, Any], now_year: int) -> list[Signal]: + urls = rec.get("source_urls") + has_qid = isinstance(urls, list) and any(qid_of(u) for u in urls) + out = [Signal("wikidata_qid_source", "pass" if has_qid else "fail")] + fields = ("release_date",) if category == "software" else ( + "launch_date", "release_date", "founded_date", + ) + for field in fields: + earliest = 1950 if category == "software" else (1800 if field == "founded_date" else 1989) + out.append(_digital_date(rec, field, now_year, earliest)) + if category == "software": + for field in ("developers", "operating_systems", "licenses", "genres"): + value = rec.get(field) + valid = isinstance(value, list) and bool(value) and all( + isinstance(v, str) and bool(v.strip()) for v in value + ) + out.append(Signal(f"{field}_string_list", "na" if value is None else ( + "pass" if valid else "fail" + ))) + else: + value = rec.get("homepage_url") + try: + parsed = urlparse(value) if isinstance(value, str) else None + valid = isinstance(value, str) and parsed is not None and parsed.scheme in { + "http", "https", + } and bool( + parsed.hostname + ) and not any(c.isspace() for c in value) + except ValueError: + valid = False + out.append(Signal("homepage_http_url", "na" if value is None else ( + "pass" if valid else "fail" + ))) + return out + + def signals_for( category: str, rec: dict[str, Any], now_year: int, soc_release: dict[str, str] ) -> list[Signal]: + if category in {"software", "website"}: + return digital_signals(category, rec, now_year) if category == "laptop": return laptop_signals(rec, now_year) if category == "monitor": diff --git a/app/verify/wikidata.py b/app/verify/wikidata.py new file mode 100644 index 0000000..2a70c34 --- /dev/null +++ b/app/verify/wikidata.py @@ -0,0 +1,90 @@ +"""Batched entity existence checks, using the promotion URL cache. + +Identity/property cross-reference is deliberately deferred: labels may be aliases +or translations, and source dates can differ in precision or release semantics. +Existence is source liveness, not independent confirmation of a record's claims. +""" + +from __future__ import annotations + +import json +import re +import time +from collections.abc import Callable, Iterator +from typing import Any +from urllib.parse import urlencode, urlparse +from urllib.request import Request, build_opener + +from .http_check import CheckResult + +USER_AGENT = "TechEngine/0.1 (https://github.com/GetTechAPI/TechEngine; source verification)" + + +def qid_of(url: Any) -> str | None: + if not isinstance(url, str): + return None + try: + parsed = urlparse(url) + if (parsed.scheme not in {"http", "https"} + or parsed.netloc.lower() not in {"wikidata.org", "www.wikidata.org"} + or parsed.query or parsed.fragment): + return None + match = re.fullmatch(r"/wiki/(Q[1-9][0-9]*)/?", parsed.path) + return match[1] if match else None + except ValueError: + return None + + +def check_batches( + urls: list[str], *, opener: Any = None, + sleep: Callable[[float], None] = time.sleep, +) -> Iterator[list[CheckResult]]: + """Yield completed batches for incremental persistence; errors stay uncached. + + No redirect resolution is requested: redirect entities are dead citations. + maxlag/API/transport failures are indeterminate and retried next run. + """ + grouped: dict[str, list[str]] = {} + for url in urls: + qid = qid_of(url) + if qid: + grouped.setdefault(qid, []).append(url) + ids = list(grouped) + opener = opener or build_opener() + for start in range(0, len(ids), 50): + if start: + sleep(1.0) + batch = ids[start:start + 50] + params = urlencode({ + "action": "wbgetentities", "ids": "|".join(batch), "props": "info", + "format": "json", "maxlag": "5", + }) + request = Request( + "https://www.wikidata.org/w/api.php?" + params, + headers={"User-Agent": USER_AGENT}, + ) + try: + with opener.open(request, timeout=30) as response: + payload = json.load(response) + if "error" in payload: + sleep(5.0) + yield [] + continue + entities = payload.get("entities", {}) + results: list[CheckResult] = [] + for qid in batch: + entity = entities.get(qid) + if not isinstance(entity, dict): + continue # incomplete/malformed response is not a dead verdict + missing = "missing" in entity + redirected = "redirect" in entity or entity.get("id") != qid + if not missing and not redirected and "lastrevid" not in entity: + continue + alive = not missing and not redirected + reason = "wikidata-entity" if alive else ( + "wikidata-missing" if missing else "wikidata-redirect" + ) + results.extend(CheckResult(url, 200, url, alive, reason) for url in grouped[qid]) + yield results + except (OSError, ValueError, TypeError, AttributeError): + yield [] diff --git a/tests/unit/test_bot_coverage.py b/tests/unit/test_bot_coverage.py index d4415fe..8037347 100644 --- a/tests/unit/test_bot_coverage.py +++ b/tests/unit/test_bot_coverage.py @@ -21,7 +21,7 @@ def test_all_categories_share_registry(): } -@pytest.mark.parametrize("category", ["software", "website", "future"]) +@pytest.mark.parametrize("category", ["future"]) def test_missing_domain_rules_never_earn_green(category): score = offline.score_record(Record(category, "example.json", { "slug": "example", "name": "Example", "source_urls": ["https://intel.com/example"], diff --git a/tests/verify/test_digital_scoring.py b/tests/verify/test_digital_scoring.py new file mode 100644 index 0000000..28dd461 --- /dev/null +++ b/tests/verify/test_digital_scoring.py @@ -0,0 +1,61 @@ +import pytest + +from app.verify.common import Record +from app.verify.offline import score_record +from app.verify.signals import signals_for + + +@pytest.mark.parametrize("category,fields", [ + ("software", {"release_date": "2020-01-01", "developers": ["Developer"], + "operating_systems": ["Linux"], "licenses": ["MIT"], "genres": ["Editor"], + "programming_languages": ["C"], "publishers": ["Publisher"]}), + ("website", {"homepage_url": "https://example.org", "launch_date": "2000-01-01", + "languages": ["English"], "owners": ["Owner"]}), +]) +def test_rich_digital_record_is_green(category, fields): + data = {**fields, "source_urls": ["https://www.wikidata.org/wiki/Q1"]} + score = score_record(Record(category, "example.json", data), 2026, {}) + assert score.band == "green" + assert score.flags == [] + + +@pytest.mark.parametrize("category,field,value", [ + ("software", "release_date", "2020-02-30"), + ("website", "launch_date", "0000-01-01"), +]) +def test_impossible_dates_force_red(category, field, value): + data = {field: value, "source_urls": ["https://www.wikidata.org/wiki/Q1"]} + score = score_record(Record(category, "example.json", data), 2026, {}) + assert score.band == "red" + assert f"!{field}_plausible" in score.flags + + +@pytest.mark.parametrize("category", ["software", "website"]) +def test_missing_qid_is_soft_failure(category): + sigs = signals_for(category, {}, 2026, {}) + assert sigs[0].failed and not sigs[0].hard + assert all(s.result == "na" for s in sigs[1:]) + + +@pytest.mark.parametrize("value", [[], "MIT", [""], [42]]) +def test_invalid_software_lists_are_soft(value): + sig = next(s for s in signals_for("software", {"licenses": value}, 2026, {}) + if s.name == "licenses_string_list") + assert sig.failed and not sig.hard + + +def test_future_dates_and_bad_homepage_are_soft(): + sigs = signals_for("website", {"launch_date": "2099-01-01", + "homepage_url": "https:///broken"}, 2026, {}) + assert all(not s.hard for s in sigs) + assert sum(s.failed for s in sigs) == 3 + + +@pytest.mark.parametrize("category,field,value", [ + ("software", "release_date", "1949-01-01"), + ("website", "launch_date", "1962-01-01"), +]) +def test_early_dates_are_ambiguous_not_impossible(category, field, value): + sig = next(s for s in signals_for(category, {field: value}, 2026, {}) + if s.name == f"{field}_plausible") + assert sig.failed and not sig.hard diff --git a/tests/verify/test_wikidata.py b/tests/verify/test_wikidata.py new file mode 100644 index 0000000..6f66566 --- /dev/null +++ b/tests/verify/test_wikidata.py @@ -0,0 +1,89 @@ +import io +import json +from urllib.parse import parse_qs, urlparse + +import pytest + +from app.verify import http_check, promote, wikidata + + +@pytest.mark.parametrize("url", ["https://evil.org/wiki/Q1", "https://wikidata.org/wiki/Q0", + "https://wikidata.org/wiki/Q1?x=1", "not a URL", None]) +def test_malformed_qid_urls(url): + assert wikidata.qid_of(url) is None + + +class Opener: + def __init__(self, error=False): + self.calls = [] + self.error = error + + def open(self, request, timeout): + params = parse_qs(urlparse(request.full_url).query) + self.calls.append((request, params)) + assert timeout == 30 + assert params["maxlag"] == ["5"] + assert "github.com/GetTechAPI/TechEngine" in request.get_header("User-agent") + ids = params["ids"][0].split("|") + entities = {qid: {"id": qid, "lastrevid": 1} for qid in ids} + if "Q2" in ids: + entities["Q2"] = {"id": "Q2", "missing": ""} + if "Q3" in ids: + entities["Q3"] = {"id": "Q3", "redirect": "Q4"} + payload = {"error": {"code": "maxlag"}} if self.error else {"entities": entities} + return io.StringIO(json.dumps(payload)) + + +def test_batches_dedupe_and_cache_promotion(tmp_path): + urls = [f"https://www.wikidata.org/wiki/Q{i}" for i in range(1, 52)] + urls.append("https://wikidata.org/wiki/Q1") + op = Opener() + sleeps = [] + results = [r for batch in wikidata.check_batches(urls, opener=op, sleep=sleeps.append) + for r in batch] + assert [len(call[1]["ids"][0].split("|")) for call in op.calls] == [50, 1] + assert sleeps == [1.0] + assert len(results) == 52 + assert sum(r.alive for r in results) == 50 + cache = {r.url: http_check.result_to_entry(r, "2026-09-27T00:00:00Z") for r in results} + path = tmp_path / "url_cache.jsonl" + http_check.save_cache(cache, path) + loaded = http_check.load_cache(path) + for qid, expected in [(1, True), (2, False), (3, False)]: + decision = promote.decide(band="green", source_urls=[urls[qid - 1]], url_cache=loaded, + crossref_decision=None) + assert decision.promote is expected + + +def test_api_errors_are_not_cached(): + assert list(wikidata.check_batches(["https://wikidata.org/wiki/Q1"], + opener=Opener(error=True), sleep=lambda _: None)) == [[]] + + +def test_transport_failure_is_not_dead(): + class BrokenOpener: + def open(self, request, timeout): + raise OSError("offline") + + assert list(wikidata.check_batches(["https://wikidata.org/wiki/Q1"], + opener=BrokenOpener(), sleep=lambda _: None)) == [[]] + + +def test_cli_skips_fresh_ids_before_cap(monkeypatch): + from argparse import Namespace + from datetime import UTC, datetime + + from app.verify import cli + from app.verify.common import Record + + urls = [f"https://wikidata.org/wiki/Q{i}" for i in range(1, 4)] + records = [Record("software", f"{i}.json", {"source_urls": [u]}) + for i, u in enumerate(urls)] + cache = {urls[0]: {"checked_at": datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%SZ"), + "reason": "wikidata-entity"}} + monkeypatch.setattr(cli, "load_all", lambda _: {"software": records}) + monkeypatch.setattr(http_check, "load_cache", lambda: cache) + seen = [] + monkeypatch.setattr(wikidata, "check_batches", lambda targets: seen.append(targets) or []) + assert cli.cmd_check_wikidata(Namespace(category=None, recheck=False, ttl_days=30, max=1)) == 0 + assert seen == [[urls[1]]]