Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension


Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
13 changes: 13 additions & 0 deletions .github/workflows/verify-network.yml
Original file line number Diff line number Diff line change
Expand Up @@ -15,6 +15,9 @@ on:
max_urls:
description: "Frontier records to URL-check"
default: "2000"
max_wikidata:
description: "Maximum Wikidata QIDs to batch-check"
default: "20000"
max_crossref:
description: "Records to cross-reference"
default: "500"
Expand Down Expand Up @@ -59,12 +62,22 @@ jobs:
- name: Install TechEngine
run: pip install -e .

- name: Restore source URL cache
uses: actions/cache@v4
with:
path: TechAPI/data/_verify/state/url_cache.jsonl
key: verify-url-cache-${{ github.run_id }}-${{ github.run_attempt }}
restore-keys: verify-url-cache-

- name: Tier 0 score (recomputed; no committed cache)
run: python -m app.verify score --no-cache

- name: Tier 1 source-URL liveness
run: python -m app.verify check-urls --max ${{ github.event.inputs.max_urls || '2000' }}

- name: Batched Wikidata source liveness
run: python -m app.verify check-wikidata --max ${{ github.event.inputs.max_wikidata || '20000' }}

- name: Tier 2 external cross-reference
run: python -m app.verify crossref --max ${{ github.event.inputs.max_crossref || '500' }}

Expand Down
44 changes: 40 additions & 4 deletions app/verify/cli.py
Original file line number Diff line number Diff line change
Expand Up @@ -22,7 +22,7 @@

from app.validate import DATA_DIR

from . import crossref, http_check, ledger, offline, promote
from . import crossref, http_check, ledger, offline, promote, wikidata
from .common import (
CATEGORIES,
SCORES_PATH,
Expand Down Expand Up @@ -198,8 +198,6 @@ def _print_markdown(hist: dict[str, Counter[str]], scored: int, hard_flags: Coun
)
gtot = sum(totals.values()) or 1
print(f"**{scored} record(s) assessed.**\n")
print("Software and website assess required fields and sources only; "
"domain consistency rules are unavailable and these categories cannot earn green.\n")

# Overall distribution as a Mermaid pie (rendered by GitHub). Mermaid colors
# slices pie1/pie2/pie3 in declaration order, so pin them to green/amber/red
Expand Down Expand Up @@ -384,6 +382,34 @@ def _ranked_unverified(
return [rec for _score, rec in scored]


def cmd_check_wikidata(args: argparse.Namespace) -> int:
records = load_all(args.category or ("software", "website"))
cache = http_check.load_cache()
now = datetime.now(UTC)
grouped: dict[str, list[str]] = {}
for rows in records.values():
for rec in rows:
for url in rec.data.get("source_urls", []):
qid = wikidata.qid_of(url)
if qid and (args.recheck or url not in cache
or not str(cache[url].get("reason", "")).startswith("wikidata-")
or not http_check.is_fresh(cache[url], now, args.ttl_days)):
grouped.setdefault(qid, []).append(url)
selected = list(grouped)[:args.max]
urls = list(dict.fromkeys(u for qid in selected for u in grouped[qid]))
checked = alive = 0
for results in wikidata.check_batches(urls):
for result in results:
cache[result.url] = http_check.result_to_entry(result, _now_iso())
checked += 1
alive += result.alive
if results:
http_check.save_cache(cache)
print(f"check-wikidata: {len(selected)} QIDs; {checked} URLs checked, {alive} alive; "
"indeterminate results remain uncached")
return 0


def cmd_check_urls(args: argparse.Namespace) -> int:
records = load_all()
_, _, soc_release = foreign_key_sets(records)
Expand Down Expand Up @@ -418,10 +444,13 @@ def cmd_check_urls(args: argparse.Namespace) -> int:

ts = _now_iso()
results = http_check.check_urls(
todo,
[u for u in todo if not wikidata.qid_of(u)],
max_workers=args.workers,
min_interval=args.min_interval,
)
# A redirect entity must not become alive again through a generic HTTP 200.
for batch in wikidata.check_batches([u for u in todo if wikidata.qid_of(u)]):
results.extend(batch)
# A rate-limited answer is not a verdict — leave it out so the next run asks
# again instead of parking the URL as dead for the whole TTL.
throttled = sum(1 for r in results if r.transient)
Expand Down Expand Up @@ -740,6 +769,13 @@ def build_parser() -> argparse.ArgumentParser:
cu.add_argument("--recheck", action="store_true", help="ignore cache freshness")
cu.set_defaults(func=cmd_check_urls)

wd = sub.add_parser("check-wikidata", help="Batched Wikidata source liveness")
wd.add_argument("--category", nargs="*", choices=CATEGORIES)
wd.add_argument("--max", type=int, default=20000, help="maximum uncached QIDs")
wd.add_argument("--ttl-days", type=int, default=http_check.DEFAULT_TTL_DAYS)
wd.add_argument("--recheck", action="store_true")
wd.set_defaults(func=cmd_check_wikidata)

cr = sub.add_parser("crossref", help="Tier 2: external cross-reference (exact heading)")
cr.add_argument("--category", nargs="*", choices=CATEGORIES, help="limit to categories")
cr.add_argument("--max", type=int, default=200, help="number of yellow/red records to escalate")
Expand Down
3 changes: 3 additions & 0 deletions app/verify/offline.py
Original file line number Diff line number Diff line change
Expand Up @@ -38,6 +38,9 @@
# "Rich" fields per category: presence (non-null) signals a fleshed-out record.
# Dotted paths index into nested dicts (e.g. "display.ppi").
RICH_FIELDS: dict[str, tuple[str, ...]] = {
"software": ("release_date", "developers", "operating_systems", "licenses", "genres",
"programming_languages", "publishers"),
"website": ("homepage_url", "launch_date", "languages", "owners"),
"cpu": ("architecture", "base_clock_ghz", "boost_clock_ghz", "l3_cache_mb",
"socket", "tdp_w", "passmark_cpu_mark"),
"gpu": ("architecture", "boost_clock_mhz", "memory_type", "memory_bandwidth_gbps",
Expand Down
58 changes: 58 additions & 0 deletions app/verify/signals.py
Original file line number Diff line number Diff line change
Expand Up @@ -16,7 +16,11 @@

import math
import re
from datetime import date
from typing import Any, NamedTuple
from urllib.parse import urlparse

from .wikidata import qid_of

# Range table mirrored from app.validate's _check_range call sites, keyed by
# (category, field) -> (lo, hi). A parity smoke test asserts this stays in sync.
Expand Down Expand Up @@ -342,9 +346,63 @@ def monitor_signals(rec: dict[str, Any], now_year: int) -> list[Signal]:
]


def _digital_date(rec: dict[str, Any], field: str, now_year: int, earliest: int) -> Signal:
value = rec.get(field)
name = f"{field}_plausible"
if value in (None, ""):
return Signal(name, "na")
try:
parsed = date.fromisoformat(value) if isinstance(value, str) else None
except ValueError:
parsed = None
if parsed is None:
return Signal(name, "fail", hard=True)
# Imported dates can describe the publisher's founding (290 websites predate
# the Web), or a planned release. These are ambiguous, not impossibilities.
return Signal(name, "pass" if earliest <= parsed.year <= now_year else "fail")


def digital_signals(category: str, rec: dict[str, Any], now_year: int) -> list[Signal]:
urls = rec.get("source_urls")
has_qid = isinstance(urls, list) and any(qid_of(u) for u in urls)
out = [Signal("wikidata_qid_source", "pass" if has_qid else "fail")]
fields = ("release_date",) if category == "software" else (
"launch_date", "release_date", "founded_date",
)
for field in fields:
earliest = 1950 if category == "software" else (1800 if field == "founded_date" else 1989)
out.append(_digital_date(rec, field, now_year, earliest))
if category == "software":
for field in ("developers", "operating_systems", "licenses", "genres"):
value = rec.get(field)
valid = isinstance(value, list) and bool(value) and all(
isinstance(v, str) and bool(v.strip()) for v in value
)
out.append(Signal(f"{field}_string_list", "na" if value is None else (
"pass" if valid else "fail"
)))
else:
value = rec.get("homepage_url")
try:
parsed = urlparse(value) if isinstance(value, str) else None
valid = isinstance(value, str) and parsed is not None and parsed.scheme in {
"http", "https",
} and bool(
parsed.hostname
) and not any(c.isspace() for c in value)
except ValueError:
valid = False
out.append(Signal("homepage_http_url", "na" if value is None else (
"pass" if valid else "fail"
)))
return out


def signals_for(
category: str, rec: dict[str, Any], now_year: int, soc_release: dict[str, str]
) -> list[Signal]:
if category in {"software", "website"}:
return digital_signals(category, rec, now_year)
if category == "laptop":
return laptop_signals(rec, now_year)
if category == "monitor":
Expand Down
90 changes: 90 additions & 0 deletions app/verify/wikidata.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,90 @@
"""Batched entity existence checks, using the promotion URL cache.

Identity/property cross-reference is deliberately deferred: labels may be aliases
or translations, and source dates can differ in precision or release semantics.
Existence is source liveness, not independent confirmation of a record's claims.
"""

from __future__ import annotations

import json
import re
import time
from collections.abc import Callable, Iterator
from typing import Any
from urllib.parse import urlencode, urlparse
from urllib.request import Request, build_opener

from .http_check import CheckResult

USER_AGENT = "TechEngine/0.1 (https://github.com/GetTechAPI/TechEngine; source verification)"


def qid_of(url: Any) -> str | None:
if not isinstance(url, str):
return None
try:
parsed = urlparse(url)
if (parsed.scheme not in {"http", "https"}
or parsed.netloc.lower() not in {"wikidata.org", "www.wikidata.org"}
or parsed.query or parsed.fragment):
return None
match = re.fullmatch(r"/wiki/(Q[1-9][0-9]*)/?", parsed.path)
return match[1] if match else None
except ValueError:
return None


def check_batches(
urls: list[str], *, opener: Any = None,
sleep: Callable[[float], None] = time.sleep,
) -> Iterator[list[CheckResult]]:
"""Yield completed batches for incremental persistence; errors stay uncached.

No redirect resolution is requested: redirect entities are dead citations.
maxlag/API/transport failures are indeterminate and retried next run.
"""
grouped: dict[str, list[str]] = {}
for url in urls:
qid = qid_of(url)
if qid:
grouped.setdefault(qid, []).append(url)
ids = list(grouped)
opener = opener or build_opener()
for start in range(0, len(ids), 50):
if start:
sleep(1.0)
batch = ids[start:start + 50]
params = urlencode({
"action": "wbgetentities", "ids": "|".join(batch), "props": "info",
"format": "json", "maxlag": "5",
})
request = Request(
"https://www.wikidata.org/w/api.php?" + params,
headers={"User-Agent": USER_AGENT},
)
try:
with opener.open(request, timeout=30) as response:
payload = json.load(response)
if "error" in payload:
sleep(5.0)
yield []
continue
entities = payload.get("entities", {})
results: list[CheckResult] = []
for qid in batch:
entity = entities.get(qid)
if not isinstance(entity, dict):
continue # incomplete/malformed response is not a dead verdict
missing = "missing" in entity
redirected = "redirect" in entity or entity.get("id") != qid
if not missing and not redirected and "lastrevid" not in entity:
continue
alive = not missing and not redirected
reason = "wikidata-entity" if alive else (
"wikidata-missing" if missing else "wikidata-redirect"
)
results.extend(CheckResult(url, 200, url, alive, reason) for url in grouped[qid])
yield results
except (OSError, ValueError, TypeError, AttributeError):
yield []
2 changes: 1 addition & 1 deletion tests/unit/test_bot_coverage.py
Original file line number Diff line number Diff line change
Expand Up @@ -21,7 +21,7 @@ def test_all_categories_share_registry():
}


@pytest.mark.parametrize("category", ["software", "website", "future"])
@pytest.mark.parametrize("category", ["future"])
def test_missing_domain_rules_never_earn_green(category):
score = offline.score_record(Record(category, "example.json", {
"slug": "example", "name": "Example", "source_urls": ["https://intel.com/example"],
Expand Down
61 changes: 61 additions & 0 deletions tests/verify/test_digital_scoring.py
Original file line number Diff line number Diff line change
@@ -0,0 +1,61 @@
import pytest

from app.verify.common import Record
from app.verify.offline import score_record
from app.verify.signals import signals_for


@pytest.mark.parametrize("category,fields", [
("software", {"release_date": "2020-01-01", "developers": ["Developer"],
"operating_systems": ["Linux"], "licenses": ["MIT"], "genres": ["Editor"],
"programming_languages": ["C"], "publishers": ["Publisher"]}),
("website", {"homepage_url": "https://example.org", "launch_date": "2000-01-01",
"languages": ["English"], "owners": ["Owner"]}),
])
def test_rich_digital_record_is_green(category, fields):
data = {**fields, "source_urls": ["https://www.wikidata.org/wiki/Q1"]}
score = score_record(Record(category, "example.json", data), 2026, {})
assert score.band == "green"
assert score.flags == []


@pytest.mark.parametrize("category,field,value", [
("software", "release_date", "2020-02-30"),
("website", "launch_date", "0000-01-01"),
])
def test_impossible_dates_force_red(category, field, value):
data = {field: value, "source_urls": ["https://www.wikidata.org/wiki/Q1"]}
score = score_record(Record(category, "example.json", data), 2026, {})
assert score.band == "red"
assert f"!{field}_plausible" in score.flags


@pytest.mark.parametrize("category", ["software", "website"])
def test_missing_qid_is_soft_failure(category):
sigs = signals_for(category, {}, 2026, {})
assert sigs[0].failed and not sigs[0].hard
assert all(s.result == "na" for s in sigs[1:])


@pytest.mark.parametrize("value", [[], "MIT", [""], [42]])
def test_invalid_software_lists_are_soft(value):
sig = next(s for s in signals_for("software", {"licenses": value}, 2026, {})
if s.name == "licenses_string_list")
assert sig.failed and not sig.hard


def test_future_dates_and_bad_homepage_are_soft():
sigs = signals_for("website", {"launch_date": "2099-01-01",
"homepage_url": "https:///broken"}, 2026, {})
assert all(not s.hard for s in sigs)
assert sum(s.failed for s in sigs) == 3


@pytest.mark.parametrize("category,field,value", [
("software", "release_date", "1949-01-01"),
("website", "launch_date", "1962-01-01"),
])
def test_early_dates_are_ambiguous_not_impossible(category, field, value):
sig = next(s for s in signals_for(category, {field: value}, 2026, {})
if s.name == f"{field}_plausible")
assert sig.failed and not sig.hard
Loading
Loading