From 52e632ec5331246e5189c8a562bca809ec216ca8 Mon Sep 17 00:00:00 2001 From: Seungpyo1007 Date: Sun, 27 Sep 2026 16:48:06 +0900 Subject: [PATCH 1/4] feat(ingest): fill absent software and website fields from Wikidata Refs #99 Refs GetTechAPI/TechAPI#295 --- app/ingest/wikidata_fill.py | 253 ++++++++++++++++++++++++++++++++++++ tests/test_wikidata_fill.py | 147 +++++++++++++++++++++ 2 files changed, 400 insertions(+) create mode 100644 app/ingest/wikidata_fill.py create mode 100644 tests/test_wikidata_fill.py diff --git a/app/ingest/wikidata_fill.py b/app/ingest/wikidata_fill.py new file mode 100644 index 0000000..2bb37c6 --- /dev/null +++ b/app/ingest/wikidata_fill.py @@ -0,0 +1,253 @@ +"""Fill absent software/website fields from their own cited Wikidata entities.""" +from __future__ import annotations + +import argparse +import difflib +import json +import re +import time +from collections import Counter +from collections.abc import Callable, Iterable +from datetime import date +from pathlib import Path +from typing import Any + +import httpx + +from app.data_root import get_data_root +from app.verify.common import Record, configure_stdout +from app.verify.offline import score_record +from app.verify.wikidata import USER_AGENT, qid_of + +MAPPINGS = { + "software": {"P577": "release_date", "P178": "developers", "P123": "publishers", + "P306": "operating_systems", "P275": "licenses", "P136": "genres", + "P277": "programming_languages"}, + "website": {"P856": "homepage_url", "P571": "launch_date", "P407": "languages", + "P127": "owners"}, +} +DATES = {"release_date", "launch_date"} +PROPERTIES = {prop for mapping in MAPPINGS.values() for prop in mapping} + + +def compact(entity: dict[str, Any]) -> dict[str, Any]: + """Discard unrelated claims, qualifiers and references from the fill cache.""" + result = {key: entity[key] for key in ("id", "missing", "redirect", "lastrevid", "labels") + if key in entity} + result["claims"] = { + prop: [{"rank": s.get("rank", "normal"), "mainsnak": s.get("mainsnak", {})} + for s in statements] + for prop, statements in entity.get("claims", {}).items() if prop in PROPERTIES + } + return result + + +def absent(value: Any) -> bool: + return value is None or value == "" or value == [] or value == {} + + +def usable(entity: dict[str, Any], qid: str) -> bool: + return not ("missing" in entity or "redirect" in entity) and entity.get("id") == qid + + +class EntityCache: + """Persistent per-entity JSON cache; failed responses are never cached.""" + + def __init__(self, directory: Path, client: httpx.Client, + sleep: Callable[[float], None] = time.sleep) -> None: + self.directory = directory + self.client = client + self.sleep = sleep + self.last_request: float | None = None + directory.mkdir(parents=True, exist_ok=True) + + def fetch(self, ids: Iterable[str]) -> dict[str, dict[str, Any]]: + result: dict[str, dict[str, Any]] = {} + pending = [] + for qid in dict.fromkeys(ids): + if not re.fullmatch(r"Q[1-9][0-9]*", qid): + raise ValueError(f"Invalid entity ID: {qid}") + path = self.directory / f"{qid}.json" + if path.exists(): + entity = json.loads(path.read_text(encoding="utf-8")) + if not isinstance(entity, dict): + raise ValueError(f"Invalid cache entry: {qid}") + result[qid] = compact(entity) + else: + pending.append(qid) + for start in range(0, len(pending), 50): + batch = pending[start:start + 50] + for attempt in range(3): + if self.last_request is not None: + self.sleep(max(0.0, 1.0 - (time.monotonic() - self.last_request))) + self.last_request = time.monotonic() + try: + response = self.client.get("https://www.wikidata.org/w/api.php", params={ + "action": "wbgetentities", "ids": "|".join(batch), + "props": "info|claims|labels", "languages": "en", + "format": "json", "maxlag": "5", + }, headers={"User-Agent": USER_AGENT}, timeout=60) + response.raise_for_status() + payload = response.json() + entities = payload.get("entities") + if "error" in payload or not isinstance(entities, dict) or not all( + isinstance(entities.get(qid), dict) for qid in batch + ): + raise ValueError("Wikidata error or incomplete entity batch") + break + except (httpx.HTTPError, ValueError): + if attempt == 2: + raise + self.sleep(5.0 * (attempt + 1)) + for qid in batch: + entity = compact(entities[qid]) + result[qid] = entity + path = self.directory / f"{qid}.json" + temporary = path.with_suffix(".tmp") + temporary.write_text(json.dumps(entity, ensure_ascii=False), encoding="utf-8") + temporary.replace(path) + print(f"Fetched {start + len(batch)}/{len(pending)} uncached entities", flush=True) + return result + + +def values(entity: dict[str, Any], prop: str) -> list[Any]: + statements = [s for s in entity.get("claims", {}).get(prop, []) + if s.get("rank") != "deprecated"] + preferred = [s for s in statements if s.get("rank") == "preferred"] + return [s["mainsnak"]["datavalue"]["value"] for s in preferred or statements + if s.get("mainsnak", {}).get("snaktype") == "value" + and "datavalue" in s["mainsnak"]] + + +def calendar_date(value: Any) -> str | None: + # Models/validate accept YYYY-MM-DD only: lesser precision cannot be represented. + if not isinstance(value, dict) or value.get("precision") != 11: + return None + if value.get("calendarmodel") != "http://www.wikidata.org/entity/Q1985727": + return None + match = re.fullmatch(r"\+(\d{4}-\d{2}-\d{2})T00:00:00Z", value.get("time", "")) + if match: + try: + return date.fromisoformat(match[1]).isoformat() + except ValueError: + pass + return None + + +def fill(record: dict[str, Any], category: str, entity: dict[str, Any], + labels: dict[str, dict[str, Any]]) -> dict[str, Any]: + updated = record.copy() + for prop, field in MAPPINGS[category].items(): + if not absent(record.get(field)): + continue + candidates = values(entity, prop) + if field in DATES: + dates = [parsed for v in candidates if (parsed := calendar_date(v))] + if dates: + updated[field] = min(dates) + elif field == "homepage_url": + urls = [v for v in candidates if isinstance(v, str) + and v.startswith(("https://", "http://"))] + if urls: + updated[field] = urls[0] + else: + names = [] + for value in candidates: + qid = value.get("id") if isinstance(value, dict) else None + target = labels.get(qid, {}) if isinstance(qid, str) else {} + label = target.get("labels", {}).get("en", {}).get("value") + if qid and usable(target, qid) and isinstance(label, str) and label: + names.append(label) + if names: + updated[field] = list(dict.fromkeys(names)) + return updated + + +def serialize(data: dict[str, Any], original: bytes) -> bytes: + newline = "\r\n" if b"\r\n" in original else "\n" + text = (json.dumps(data, ensure_ascii=False, indent=2) + "\n").replace("\n", newline) + return (b"\xef\xbb\xbf" if original.startswith(b"\xef\xbb\xbf") else b"") + text.encode("utf-8") + + +def run(category: str, cache: EntityCache, root: Path, *, apply: bool = False, + maximum: int | None = None) -> tuple[dict[str, Any], list[str]]: + paths = sorted((root / category).rglob("*.json")) + if not paths: + raise ValueError(f"No {category} records in data checkout") + if maximum is not None: + paths = paths[:maximum] + records = [(p, json.loads(p.read_bytes().decode("utf-8-sig"))) for p in paths] + cited = [(p, r, list(dict.fromkeys(q for u in r.get("source_urls", []) + if (q := qid_of(u))))) for p, r in records] + entities = cache.fetch(q for _, _, ids in cited for q in ids) + references: set[str] = set() + for _, record, ids in cited: + for qid in ids: + entity = entities[qid] + if not usable(entity, qid): + continue + for prop, field in MAPPINGS[category].items(): + if field not in DATES | {"homepage_url"} and absent(record.get(field)): + references.update(v["id"] for v in values(entity, prop) + if isinstance(v, dict) and isinstance(v.get("id"), str)) + labels = cache.fetch(sorted(references)) + counts: Counter[str] = Counter() + before_green = after_green = moved = changed = 0 + samples: list[str] = [] + for path, record, ids in cited: + updated = record.copy() + for qid in ids: + if usable(entities[qid], qid): + updated = fill(updated, category, entities[qid], labels) + rel = path.relative_to(root).as_posix() + before = score_record(Record(category, rel, record), date.today().year, {}).band == "green" + after = score_record(Record(category, rel, updated), date.today().year, {}).band == "green" + before_green += before + after_green += after + moved += after and not before + if updated == record: + continue + changed += 1 + counts.update(field for field in MAPPINGS[category].values() + if record.get(field) != updated.get(field)) + original = path.read_bytes() + rendered = serialize(updated, original) + if len(samples) < 5: + samples.append("".join(difflib.unified_diff( + original.decode("utf-8-sig").splitlines(keepends=True), + rendered.decode("utf-8-sig").splitlines(keepends=True), + fromfile=rel, tofile=rel))) + if apply: + path.write_bytes(rendered) + return {"category": category, "records": len(records), "changed_records": changed, + "fills": {f: counts[f] for f in MAPPINGS[category].values()}, + "green_before": before_green, "green_after": after_green, "moved_to_green": moved, + "applied": apply}, samples + + +def main() -> None: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--category", choices=list(MAPPINGS), required=True) + parser.add_argument("--apply", action="store_true") + parser.add_argument("--max", type=int, dest="maximum") + parser.add_argument("--report", type=Path, help="Write dry-run summary and five sample diffs") + args = parser.parse_args() + if args.maximum is not None and args.maximum < 0: + parser.error("--max must be nonnegative") + configure_stdout() + root = get_data_root() + with httpx.Client() as client: + cache = EntityCache(root / "_verify/state/wikidata_fill", client) + summary, samples = run(args.category, cache, root, apply=args.apply, maximum=args.maximum) + print(json.dumps(summary, indent=2)) + if args.report: + args.report.parent.mkdir(parents=True, exist_ok=True) + args.report.write_text( + f"# Wikidata fill: {args.category}\n\n" + "Dates require day precision and Gregorian calendar; month/year claims are skipped.\n\n" + + "```json\n" + json.dumps(summary, indent=2) + "\n```\n\n" + + "\n\n".join("```diff\n" + sample + "```" for sample in samples), encoding="utf-8") + + +if __name__ == "__main__": + main() diff --git a/tests/test_wikidata_fill.py b/tests/test_wikidata_fill.py new file mode 100644 index 0000000..b3017f7 --- /dev/null +++ b/tests/test_wikidata_fill.py @@ -0,0 +1,147 @@ +"""Regression tests for cited-entity fill and conservative serialization.""" +import json +from urllib.parse import parse_qs + +import httpx +import pytest + +from app.ingest.wikidata_fill import EntityCache, calendar_date, fill, run, serialize + + +def claim(value, rank="normal", snaktype="value"): + return {"rank": rank, "mainsnak": {"snaktype": snaktype, "datavalue": {"value": value}}} + + +def timestamp(precision=11, time="+2020-02-29T00:00:00Z"): + return {"precision": precision, "time": time, + "calendarmodel": "http://www.wikidata.org/entity/Q1985727"} + + +def test_mapping_rank_and_no_overwrite(): + labels = {"Q2": {"id": "Q2", "labels": {"en": {"value": "English label"}}}, + "Q3": {"id": "Q3", "labels": {"fr": {"value": "French only"}}}} + entity = {"id": "Q1", "claims": { + p: [claim({"id": "Q3"}), claim({"id": "Q2"}, "preferred"), + claim({"id": "Q3"}, "deprecated")] for p in + ("P178", "P123", "P306", "P275", "P136", "P277", "P407", "P127")}} + entity["claims"].update({"P577": [claim(timestamp())], "P571": [claim(timestamp())], + "P856": [claim("https://example.com")]}) + software = fill({"developers": ["Existing"], "publishers": None, + "licenses": [], "genres": "", "operating_systems": {}}, + "software", entity, labels) + assert software == {"developers": ["Existing"], "publishers": ["English label"], + "licenses": ["English label"], "genres": ["English label"], + "operating_systems": ["English label"], "release_date": "2020-02-29", + "programming_languages": ["English label"]} + website = fill({}, "website", entity, labels) + assert website == {"homepage_url": "https://example.com", "launch_date": "2020-02-29", + "languages": ["English label"], "owners": ["English label"]} + assert fill(website, "website", entity, {}) == website + + +@pytest.mark.parametrize("precision", [0, 8, 9, 10, 12]) +def test_precision_skips_inexpressible_dates(precision): + assert calendar_date(timestamp(precision)) is None + + +def test_invalid_dates_calendar_and_unknown_values(): + assert calendar_date(timestamp(time="+2021-02-29T00:00:00Z")) is None + assert calendar_date({**timestamp(), "calendarmodel": "Julian"}) is None + assert calendar_date("2020") is None + entity = {"claims": {"P577": [claim(timestamp(), "deprecated")], + "P178": [claim({}, snaktype="somevalue")], + "P856": [claim("javascript:alert(1)")]}} + assert fill({}, "software", entity, {}) == {} + assert fill({}, "website", entity, {}) == {} + + +def test_http_batches_cache_and_rate(tmp_path): + requests = [] + sleeps = [] + + def handler(request): + params = parse_qs(request.url.query.decode()) + ids = params["ids"][0].split("|") + requests.append(ids) + assert params["maxlag"] == ["5"] + assert params["languages"] == ["en"] + assert "TechEngine" in request.headers["User-Agent"] + assert "redirects" not in params + return httpx.Response(200, json={"entities": {qid: {"id": qid} for qid in ids}}) + + with httpx.Client(transport=httpx.MockTransport(handler)) as client: + cache = EntityCache(tmp_path, client, sleeps.append) + ids = [f"Q{i}" for i in range(1, 52)] + assert len(cache.fetch(ids + ids)) == 51 + assert [len(batch) for batch in requests] == [50, 1] + assert len(sleeps) == 1 and 0 <= sleeps[0] <= 1 + assert len(EntityCache(tmp_path, client).fetch(ids)) == 51 + assert len(requests) == 2 + + +def test_api_failure_is_not_cached(tmp_path): + with httpx.Client(transport=httpx.MockTransport( + lambda request: httpx.Response(200, json={"error": {"code": "maxlag"}}) + )) as client: + cache = EntityCache(tmp_path, client, lambda _: None) + with pytest.raises(ValueError, match="incomplete"): + cache.fetch(["Q1"]) + assert not list(tmp_path.glob("*.json")) + + +@pytest.mark.parametrize("entity", [{"id": "Q1", "redirect": "Q2"}, + {"id": "Q2"}, {"id": "Q1", "missing": ""}]) +def test_redirect_and_missing_are_skipped(tmp_path, entity): + root = tmp_path / "data" + path = root / "website" / "site.json" + path.parent.mkdir(parents=True) + record = {"slug": "site", "name": "Site", "verified": False, + "source_urls": ["https://www.wikidata.org/wiki/Q1"]} + path.write_text(json.dumps(record), encoding="utf-8") + entity["claims"] = {"P856": [claim("https://example.com")]} + with httpx.Client(transport=httpx.MockTransport( + lambda request: httpx.Response(200, json={"entities": {"Q1": entity}}) + )) as client: + summary, samples = run("website", EntityCache(tmp_path / "cache", client), root) + assert summary["changed_records"] == 0 + assert not samples + assert json.loads(path.read_text()) == record + + +def test_dry_run_and_apply_serialization(tmp_path): + path = tmp_path / "website" / "site.json" + path.parent.mkdir() + record = {"slug": "site", "name": "Site", "verified": False, + "source_urls": ["https://www.wikidata.org/wiki/Q1"]} + original = serialize(record, b"\r\n") + path.write_bytes(original) + entities = {"Q1": {"id": "Q1", "claims": {"P127": [claim({"id": "Q2"})]}}, + "Q2": {"id": "Q2", "labels": {"en": {"value": "Owner"}}}} + with httpx.Client(transport=httpx.MockTransport(lambda request: httpx.Response( + 200, json={"entities": {qid: entities[qid] for qid in + request.url.params["ids"].split("|")}} + ))) as client: + cache = EntityCache(tmp_path / "cache", client, lambda _: None) + summary, samples = run("website", cache, tmp_path) + assert summary["fills"]["owners"] == 1 and len(samples) == 1 + assert path.read_bytes() == original + run("website", cache, tmp_path, apply=True) + rendered = path.read_bytes() + assert rendered.endswith(b"\r\n") and b"\n" not in rendered.replace(b"\r\n", b"") + assert list(json.loads(rendered)) == list(record) + ["owners"] + assert b' "owners": [' in rendered + assert serialize(record, b'\xef\xbb\xbf\n').startswith(b'\xef\xbb\xbf') + +def test_compact_cache_keeps_only_fill_properties(tmp_path): + entity = {"id": "Q1", "claims": { + "P178": [{**claim({"id": "Q2"}), "references": [{"huge": "unneeded"}]}], + "P999999": [claim("unrelated")], + }, "labels": {"en": {"value": "Name"}}, "sitelinks": {"enwiki": {"title": "Name"}}} + with httpx.Client(transport=httpx.MockTransport(lambda request: httpx.Response( + 200, json={"entities": {"Q1": entity}} + ))) as client: + cached = EntityCache(tmp_path, client).fetch(["Q1"])["Q1"] + assert set(cached["claims"]) == {"P178"} + assert "references" not in cached["claims"]["P178"][0] + assert "sitelinks" not in cached + assert cached["labels"] == entity["labels"] From b39211852925ef37449e93b729d2feeae2ae46c2 Mon Sep 17 00:00:00 2001 From: Seungpyo1007 Date: Sun, 27 Sep 2026 17:08:26 +0900 Subject: [PATCH 2/4] fix(ingest): annotate compact entity cache payload Refs #99 Refs GetTechAPI/TechAPI#295 --- app/ingest/wikidata_fill.py | 6 ++++-- 1 file changed, 4 insertions(+), 2 deletions(-) diff --git a/app/ingest/wikidata_fill.py b/app/ingest/wikidata_fill.py index 2bb37c6..c21072a 100644 --- a/app/ingest/wikidata_fill.py +++ b/app/ingest/wikidata_fill.py @@ -32,8 +32,10 @@ def compact(entity: dict[str, Any]) -> dict[str, Any]: """Discard unrelated claims, qualifiers and references from the fill cache.""" - result = {key: entity[key] for key in ("id", "missing", "redirect", "lastrevid", "labels") - if key in entity} + result: dict[str, Any] = { + key: entity[key] for key in ("id", "missing", "redirect", "lastrevid", "labels") + if key in entity + } result["claims"] = { prop: [{"rank": s.get("rank", "normal"), "mainsnak": s.get("mainsnak", {})} for s in statements] From 706c2544808b6391d326a177d91d878954900012 Mon Sep 17 00:00:00 2001 From: Seungpyo1007 Date: Sun, 27 Sep 2026 17:24:11 +0900 Subject: [PATCH 3/4] fix(ingest): retry truncated Wikidata entity batches Refs #99 Refs GetTechAPI/TechAPI#295 --- app/ingest/wikidata_fill.py | 58 +++++++++++++++++++++++-------------- tests/test_wikidata_fill.py | 21 +++++++++++++- 2 files changed, 56 insertions(+), 23 deletions(-) diff --git a/app/ingest/wikidata_fill.py b/app/ingest/wikidata_fill.py index c21072a..291ae1f 100644 --- a/app/ingest/wikidata_fill.py +++ b/app/ingest/wikidata_fill.py @@ -79,28 +79,7 @@ def fetch(self, ids: Iterable[str]) -> dict[str, dict[str, Any]]: pending.append(qid) for start in range(0, len(pending), 50): batch = pending[start:start + 50] - for attempt in range(3): - if self.last_request is not None: - self.sleep(max(0.0, 1.0 - (time.monotonic() - self.last_request))) - self.last_request = time.monotonic() - try: - response = self.client.get("https://www.wikidata.org/w/api.php", params={ - "action": "wbgetentities", "ids": "|".join(batch), - "props": "info|claims|labels", "languages": "en", - "format": "json", "maxlag": "5", - }, headers={"User-Agent": USER_AGENT}, timeout=60) - response.raise_for_status() - payload = response.json() - entities = payload.get("entities") - if "error" in payload or not isinstance(entities, dict) or not all( - isinstance(entities.get(qid), dict) for qid in batch - ): - raise ValueError("Wikidata error or incomplete entity batch") - break - except (httpx.HTTPError, ValueError): - if attempt == 2: - raise - self.sleep(5.0 * (attempt + 1)) + entities = self._retrieve(batch) for qid in batch: entity = compact(entities[qid]) result[qid] = entity @@ -111,6 +90,41 @@ def fetch(self, ids: Iterable[str]) -> dict[str, dict[str, Any]]: print(f"Fetched {start + len(batch)}/{len(pending)} uncached entities", flush=True) return result + def _retrieve(self, batch: list[str]) -> dict[str, Any]: + for attempt in range(3): + if self.last_request is not None: + self.sleep(max(0.0, 1.0 - (time.monotonic() - self.last_request))) + self.last_request = time.monotonic() + try: + response = self.client.get("https://www.wikidata.org/w/api.php", params={ + "action": "wbgetentities", "ids": "|".join(batch), + "props": "info|claims|labels", "languages": "en", + "format": "json", "maxlag": "5", + }, headers={"User-Agent": USER_AGENT}, timeout=60) + response.raise_for_status() + payload = response.json() + entities = payload.get("entities") + if "error" in payload or not isinstance(entities, dict): + raise ValueError(f"Wikidata API error: {payload.get('error')}") + missing = [qid for qid in batch if not isinstance(entities.get(qid), dict)] + if missing and "truncated" not in json.dumps(payload.get("warnings", {})): + raise ValueError("Wikidata incomplete entity batch") + break + except (httpx.HTTPError, ValueError): + if attempt == 2: + raise + self.sleep(5.0 * (attempt + 1)) + if missing: + if len(batch) == 1: + raise ValueError(f"Wikidata entity exceeds response size limit: {batch[0]}") + # Only explicit size truncation permits smaller batches. Other + # incomplete responses remain uncached for retry. + size = max(1, len(batch) // 2) + for start in range(0, len(missing), size): + entities.update(self._retrieve(missing[start:start + size])) + complete: dict[str, Any] = entities + return complete + def values(entity: dict[str, Any], prop: str) -> list[Any]: statements = [s for s in entity.get("claims", {}).get(prop, []) diff --git a/tests/test_wikidata_fill.py b/tests/test_wikidata_fill.py index b3017f7..7d8dba0 100644 --- a/tests/test_wikidata_fill.py +++ b/tests/test_wikidata_fill.py @@ -84,7 +84,7 @@ def test_api_failure_is_not_cached(tmp_path): lambda request: httpx.Response(200, json={"error": {"code": "maxlag"}}) )) as client: cache = EntityCache(tmp_path, client, lambda _: None) - with pytest.raises(ValueError, match="incomplete"): + with pytest.raises(ValueError, match="API error"): cache.fetch(["Q1"]) assert not list(tmp_path.glob("*.json")) @@ -145,3 +145,22 @@ def test_compact_cache_keeps_only_fill_properties(tmp_path): assert "references" not in cached["claims"]["P178"][0] assert "sitelinks" not in cached assert cached["labels"] == entity["labels"] + + +def test_truncated_response_retries_omitted_entities_in_smaller_batches(tmp_path): + sizes = [] + + def handler(request): + ids = request.url.params["ids"].split("|") + sizes.append(len(ids)) + returned = ids[:1] if len(ids) == 50 else ids + payload = {"entities": {qid: {"id": qid} for qid in returned}} + if len(ids) == 50: + payload["warnings"] = {"result": {"*": "This result was truncated"}} + return httpx.Response(200, json=payload) + + with httpx.Client(transport=httpx.MockTransport(handler)) as client: + cache = EntityCache(tmp_path, client, lambda _: None) + assert len(cache.fetch(f"Q{i}" for i in range(1, 51))) == 50 + assert sizes == [50, 25, 24] + assert len(list(tmp_path.glob("*.json"))) == 50 From 18a009e517e68f6f95067d50478e518dc20244c7 Mon Sep 17 00:00:00 2001 From: Seungpyo1007 Date: Sun, 27 Sep 2026 17:51:15 +0900 Subject: [PATCH 4/4] fix(ingest): ignore pre-Web inception dates as website launch dates P571 is the owner's inception (e.g. a newspaper founded in 1785), so dates before 1991 are skipped for launch_date. Refs #99 --- app/ingest/wikidata_fill.py | 4 ++++ tests/test_wikidata_fill.py | 7 +++++++ 2 files changed, 11 insertions(+) diff --git a/app/ingest/wikidata_fill.py b/app/ingest/wikidata_fill.py index 291ae1f..db4a986 100644 --- a/app/ingest/wikidata_fill.py +++ b/app/ingest/wikidata_fill.py @@ -27,6 +27,7 @@ "P127": "owners"}, } DATES = {"release_date", "launch_date"} +WEB_EPOCH = "1991-01-01" PROPERTIES = {prop for mapping in MAPPINGS.values() for prop in mapping} @@ -159,6 +160,9 @@ def fill(record: dict[str, Any], category: str, entity: dict[str, Any], candidates = values(entity, prop) if field in DATES: dates = [parsed for v in candidates if (parsed := calendar_date(v))] + if field == "launch_date": + # P571 is the owner's inception; before the Web it can't be a site launch. + dates = [d for d in dates if d >= WEB_EPOCH] if dates: updated[field] = min(dates) elif field == "homepage_url": diff --git a/tests/test_wikidata_fill.py b/tests/test_wikidata_fill.py index 7d8dba0..f8b9dc4 100644 --- a/tests/test_wikidata_fill.py +++ b/tests/test_wikidata_fill.py @@ -164,3 +164,10 @@ def handler(request): assert len(cache.fetch(f"Q{i}" for i in range(1, 51))) == 50 assert sizes == [50, 25, 24] assert len(list(tmp_path.glob("*.json"))) == 50 + + +def test_pre_web_inception_is_not_a_website_launch(): + entity = {"id": "Q1", "claims": {"P571": [claim(timestamp(time="+1785-01-01T00:00:00Z"))]}} + assert "launch_date" not in fill({}, "website", entity, {}) + entity = {"id": "Q1", "claims": {"P577": [claim(timestamp(time="+1985-11-20T00:00:00Z"))]}} + assert fill({}, "software", entity, {})["release_date"] == "1985-11-20"