From df5649bfa82efb82792876133099403add439ef3 Mon Sep 17 00:00:00 2001 From: Seungpyo1007 Date: Sun, 27 Sep 2026 02:05:12 +0900 Subject: [PATCH] feat: cover all dataset categories and actual PR diffs Refs GetTechAPI/TechAPI#1 --- .../techapi-pr-validation-comment.yml | 56 ++++-- .github/workflows/techapi-verify-comment.yml | 30 +++- .github/workflows/verify-network.yml | 4 +- .gitignore | 3 + app/categories.py | 10 ++ app/dump.py | 17 +- app/dump_check.py | 84 +++++++++ app/validate.py | 31 +--- app/verify/cli.py | 43 ++--- app/verify/common.py | 13 +- app/verify/offline.py | 16 +- integrity_check.py | 13 +- tests/unit/test_bot_coverage.py | 160 ++++++++++++++++++ 13 files changed, 377 insertions(+), 103 deletions(-) create mode 100644 app/categories.py create mode 100644 app/dump_check.py create mode 100644 tests/unit/test_bot_coverage.py diff --git a/.github/workflows/techapi-pr-validation-comment.yml b/.github/workflows/techapi-pr-validation-comment.yml index d7cffed..f9f4a6e 100644 --- a/.github/workflows/techapi-pr-validation-comment.yml +++ b/.github/workflows/techapi-pr-validation-comment.yml @@ -21,6 +21,12 @@ on: type: string required: false default: "" + base_ref: + description: "TechAPI PR base branch (resolved from PR if omitted)" + required: false + base_sha: + description: "TechAPI PR base SHA (resolved from PR if omitted)" + required: false permissions: contents: read @@ -35,6 +41,8 @@ jobs: env: TECHAPI_COMMENT_TOKEN: ${{ secrets.TECHENGINEBOT_TOKEN || secrets.TECHAPI_TOKEN }} TECHAPI_PR_NUMBER: ${{ github.event.client_payload.pr_number || inputs.pr_number }} + TECHAPI_BASE_REF: ${{ github.event.client_payload.base_ref || inputs.base_ref }} + TECHAPI_BASE_SHA: ${{ github.event.client_payload.base_sha || inputs.base_sha }} TECHAPI_HEAD_SHA: ${{ github.event.client_payload.head_sha || inputs.head_sha }} TECHAPI_HEAD_REF: ${{ github.event.client_payload.head_ref || '' }} TECHAPI_PR_URL: ${{ github.event.client_payload.pr_url || inputs.pr_url }} @@ -45,6 +53,20 @@ jobs: - name: Checkout TechEngine uses: actions/checkout@v4 + - name: Resolve actual PR base + env: + GH_TOKEN: ${{ env.TECHAPI_COMMENT_TOKEN || github.token }} + shell: bash + run: | + set -euo pipefail + if [ -z "$TECHAPI_BASE_REF" ] || [ -z "$TECHAPI_BASE_SHA" ]; then + gh api "repos/GetTechAPI/TechAPI/pulls/$TECHAPI_PR_NUMBER" > pr-base.json + TECHAPI_BASE_REF=$(jq -r '.base.ref' pr-base.json) + TECHAPI_BASE_SHA=$(jq -r '.base.sha' pr-base.json) + fi + echo "TECHAPI_BASE_REF=$TECHAPI_BASE_REF" >> "$GITHUB_ENV" + echo "TECHAPI_BASE_SHA=$TECHAPI_BASE_SHA" >> "$GITHUB_ENV" + - name: Checkout TechAPI PR head uses: actions/checkout@v4 with: @@ -57,16 +79,18 @@ jobs: uses: actions/checkout@v4 with: repository: GetTechAPI/TechAPI - ref: develop + ref: ${{ env.TECHAPI_BASE_SHA }} path: TechAPI-base fetch-depth: 0 - name: Pin PR base to the merge base shell: bash run: | - git -C TechAPI fetch --no-tags origin develop - base_sha="$(git -C TechAPI merge-base origin/develop HEAD)" + git -C TechAPI fetch --no-tags origin "$TECHAPI_BASE_SHA" + base_sha="$(git -C TechAPI merge-base "$TECHAPI_BASE_SHA" HEAD)" + git -C TechAPI-base fetch --no-tags origin "$base_sha" git -C TechAPI-base checkout --detach "$base_sha" + echo "TECHAPI_DIFF_BASE=$base_sha" >> "$GITHUB_ENV" - uses: actions/setup-python@v6 with: @@ -165,15 +189,20 @@ jobs: PY ) + python -m app.dump_check --base "$TECHAPI_BASE_SHA" --base-ref "$TECHAPI_BASE_REF" > dump-check.log 2>&1 + dump_status=$? + cat dump-check.log >> validation.log + python -m app.dump_check --base "$TECHAPI_BASE_SHA" --base-ref "$TECHAPI_BASE_REF" --scope-only > scope.txt sed -i '/_status=/d' validation.log status="success" - if [ "${app_status:-1}" != "0" ] || [ "${integrity_status:-1}" != "0" ]; then + if [ "${app_status:-1}" != "0" ] || [ "${integrity_status:-1}" != "0" ] || [ "${dump_status:-1}" != "0" ]; then status="failure" fi echo "status=$status" >> "$GITHUB_OUTPUT" echo "app_status=${app_status:-1}" >> "$GITHUB_OUTPUT" + echo "dump_status=${dump_status:-1}" >> "$GITHUB_OUTPUT" echo "integrity_status=${integrity_status:-1}" >> "$GITHUB_OUTPUT" - name: Build TechAPI homepage @@ -210,16 +239,8 @@ jobs: HEAD = Path("TechAPI/data") BASE = Path("TechAPI-base/data") - CATEGORIES = ( - "brand", - "soc", - "smartphone", - "tablet", - "watch", - "pda", - "gpu", - "cpu", - ) + import os + from app.categories import CATEGORIES MAX_WARNINGS = 20 def load_json(path: Path) -> dict[str, Any]: @@ -250,7 +271,7 @@ jobs: "diff", "--name-status", "--no-renames", - "origin/develop...HEAD", + f"{os.environ['TECHAPI_BASE_SHA']}...HEAD", "--", "data", ], @@ -668,6 +689,8 @@ jobs: echo "" echo "## TechEngine change review: ${result}" echo + cat scope.txt + echo echo "- PR: #${TECHAPI_PR_NUMBER}" echo "- Ref: \`${TECHAPI_HEAD_REF:-detached}\`" echo "- Commit: \`${short_sha}\`" @@ -677,6 +700,7 @@ jobs: echo "| Check | Result |" echo "| --- | --- |" echo "| \`python -m app.validate\` | $([ "${{ steps.validate.outputs.app_status }}" = "0" ] && echo PASS || echo FAIL) |" + echo "| Changed dump JSON and manifest/index counts | $([ "${{ steps.validate.outputs.dump_status }}" = "0" ] && echo PASS || echo FAIL) |" echo "| New hard integrity anomalies vs PR base | $([ "${{ steps.validate.outputs.integrity_status }}" = "0" ] && echo PASS || echo FAIL) |" if [ "${site_changed}" = "true" ]; then echo "| \`cd TechAPI/site && npm ci && npm run build\` | $([ "${site_build_status}" = "0" ] && echo PASS || echo FAIL) |" @@ -693,6 +717,8 @@ jobs: echo "" echo "## TechEngine validation stats: ${result}" echo + cat scope.txt + echo echo "- PR: #${TECHAPI_PR_NUMBER}" echo "- Ref: \`${TECHAPI_HEAD_REF:-detached}\`" echo "- Commit: \`${short_sha}\`" diff --git a/.github/workflows/techapi-verify-comment.yml b/.github/workflows/techapi-verify-comment.yml index b5ef791..38d41d8 100644 --- a/.github/workflows/techapi-verify-comment.yml +++ b/.github/workflows/techapi-verify-comment.yml @@ -15,6 +15,12 @@ on: head_sha: description: "TechAPI commit SHA to verify" required: true + base_ref: + description: "TechAPI PR base branch (resolved from PR if omitted)" + required: false + base_sha: + description: "TechAPI PR base SHA (resolved from PR if omitted)" + required: false permissions: contents: read @@ -30,6 +36,8 @@ jobs: PYTHONIOENCODING: utf-8 TECHAPI_COMMENT_TOKEN: ${{ secrets.TECHENGINEBOT_TOKEN || secrets.TECHAPI_TOKEN }} TECHAPI_PR_NUMBER: ${{ github.event.client_payload.pr_number || inputs.pr_number }} + TECHAPI_BASE_REF: ${{ github.event.client_payload.base_ref || inputs.base_ref }} + TECHAPI_BASE_SHA: ${{ github.event.client_payload.base_sha || inputs.base_sha }} TECHAPI_HEAD_SHA: ${{ github.event.client_payload.head_sha || inputs.head_sha }} REQUESTED_BY: ${{ github.event.client_payload.requested_by || github.actor }} TECHAPI_COMMENT_ID: ${{ github.event.client_payload.comment_id }} @@ -59,6 +67,20 @@ jobs: - name: Checkout TechEngine uses: actions/checkout@v4 + - name: Resolve actual PR base + env: + GH_TOKEN: ${{ env.TECHAPI_COMMENT_TOKEN || github.token }} + shell: bash + run: | + set -euo pipefail + if [ -z "$TECHAPI_BASE_REF" ] || [ -z "$TECHAPI_BASE_SHA" ]; then + gh api "repos/GetTechAPI/TechAPI/pulls/$TECHAPI_PR_NUMBER" > pr-base.json + TECHAPI_BASE_REF=$(jq -r '.base.ref' pr-base.json) + TECHAPI_BASE_SHA=$(jq -r '.base.sha' pr-base.json) + fi + echo "TECHAPI_BASE_REF=$TECHAPI_BASE_REF" >> "$GITHUB_ENV" + echo "TECHAPI_BASE_SHA=$TECHAPI_BASE_SHA" >> "$GITHUB_ENV" + - name: Checkout TechAPI PR head uses: actions/checkout@v4 with: @@ -83,19 +105,21 @@ jobs: env: TECHAPI_DATA_DIR: ${{ github.workspace }}/TechAPI/data run: | - git -C TechAPI fetch origin main --depth=1 || true + git -C TechAPI fetch --no-tags origin "$TECHAPI_BASE_SHA" { echo 'report<> "$GITHUB_OUTPUT" diff --git a/.github/workflows/verify-network.yml b/.github/workflows/verify-network.yml index 92da15f..7389035 100644 --- a/.github/workflows/verify-network.yml +++ b/.github/workflows/verify-network.yml @@ -59,8 +59,8 @@ jobs: - name: Install TechEngine run: pip install -e . - - name: Tier 0 score (writes scores cache) - run: python -m app.verify score + - name: Tier 0 score (recomputed; no committed cache) + run: python -m app.verify score --no-cache - name: Tier 1 source-URL liveness run: python -m app.verify check-urls --max ${{ github.event.inputs.max_urls || '2000' }} diff --git a/.gitignore b/.gitignore index 3b680b6..2ab09b1 100644 --- a/.gitignore +++ b/.gitignore @@ -54,3 +54,6 @@ coverage.xml # Local Wikipedia SoC backfill cache and reports (never commit) .wikipedia-soc-backfill/ + +# Recomputable verifier cache +/data/_verify/state/ diff --git a/app/categories.py b/app/categories.py new file mode 100644 index 0000000..1b5101a --- /dev/null +++ b/app/categories.py @@ -0,0 +1,10 @@ +"""Canonical seed categories and their public collection names.""" + +CATEGORIES: tuple[str, ...] = ( + "brand", "soc", "smartphone", "tablet", "watch", "pda", "gpu", "cpu", + "laptop", "monitor", "software", "website", +) +COLLECTIONS = dict(zip(CATEGORIES, ( + "brands", "socs", "smartphones", "tablets", "watches", "pdas", "gpus", "cpus", + "laptops", "monitors", "software", "websites", +), strict=True)) diff --git a/app/dump.py b/app/dump.py index 9cc5bbb..7848ec1 100644 --- a/app/dump.py +++ b/app/dump.py @@ -17,23 +17,12 @@ from fastapi.testclient import TestClient +from app.categories import COLLECTIONS as CATEGORY_COLLECTIONS + OUTPUT_DIR = Path(__file__).resolve().parent.parent / "dump" # Collections that expose list + detail endpoints. -COLLECTIONS = [ - "brands", - "socs", - "smartphones", - "tablets", - "watches", - "pdas", - "gpus", - "cpus", - "laptops", - "monitors", - "software", - "websites", -] +COLLECTIONS = list(CATEGORY_COLLECTIONS.values()) # Collections with a /score sub-resource (§8) and a `scored` manifest count. SCORED = {"smartphones", "cpus", "gpus", "socs"} PAGE_LIMIT = 100 # API max page size (§7.3) diff --git a/app/dump_check.py b/app/dump_check.py new file mode 100644 index 0000000..dbd0fdc --- /dev/null +++ b/app/dump_check.py @@ -0,0 +1,84 @@ +"""Lightweight validation of a changed static dump against seed record counts.""" + +from __future__ import annotations + +import argparse +import json +import subprocess +from pathlib import Path +from typing import Any + +from app.categories import CATEGORIES, COLLECTIONS + + +def check_dump(repo: Path, base: str) -> list[str]: + changed = subprocess.run( + ["git", "diff", "--name-only", "--diff-filter=ACMR", f"{base}...HEAD", + "--", "site/public/v1/"], + cwd=repo, text=True, capture_output=True, check=True, + ).stdout.splitlines() + touched = subprocess.run( + ["git", "diff", "--name-only", f"{base}...HEAD", "--", "site/public/v1/"], + cwd=repo, text=True, capture_output=True, check=True, + ).stdout.splitlines() + if not touched: + return [] + errors: list[str] = [] + + def read(path: Path) -> Any: + try: + return json.loads(path.read_text(encoding="utf-8-sig")) + except (OSError, ValueError) as exc: + errors.append(f"{path.relative_to(repo).as_posix()}: {exc}") + return None + + for rel in changed: + if rel.endswith(".json"): + read(repo / rel) + root = repo / "site/public/v1" + manifest = read(root / "index.json") + collections = manifest.get("collections", {}) if isinstance(manifest, dict) else {} + if not isinstance(collections, dict): + errors.append("manifest collections must be an object") + collections = {} + for category in CATEGORIES: + resource = COLLECTIONS[category] + count = sum(1 for p in (repo / "data" / category).rglob("*.json") + if not p.name.startswith("_")) + entry = collections.get(resource) + index = read(root / resource / "index.json") + if not isinstance(entry, dict) or entry.get("count") != count: + errors.append(f"{resource}: manifest count must equal {count}") + if (not isinstance(index, dict) or index.get("count") != count + or not isinstance(index.get("results"), list) + or len(index["results"]) != count): + errors.append(f"{resource}: index count/results must equal {count}") + return errors + + +def scope(repo: Path, base_ref: str, base_sha: str) -> str: + count = sum(1 for category in CATEGORIES + for p in (repo / "data" / category).rglob("*.json") + if not p.name.startswith("_")) + return f"12/12 categories ? {count:,} records ? diff base {base_ref}@{base_sha[:7]}" + + +def main() -> int: + parser = argparse.ArgumentParser(description=__doc__) + parser.add_argument("--repo", type=Path, default=Path("TechAPI")) + parser.add_argument("--base", required=True) + parser.add_argument("--base-ref", default="unknown") + parser.add_argument("--scope-only", action="store_true") + args = parser.parse_args() + print(scope(args.repo, args.base_ref, args.base)) + if args.scope_only: + return 0 + errors = check_dump(args.repo, args.base) + for error in errors: + print(error) + print(f"Dump JSON/count check: {'FAIL' if errors else 'PASS (or no dump changes)'}") + return int(bool(errors)) + + +if __name__ == "__main__": + raise SystemExit(main()) diff --git a/app/validate.py b/app/validate.py index 0dff389..d6d2cf1 100644 --- a/app/validate.py +++ b/app/validate.py @@ -16,6 +16,7 @@ from pathlib import Path from typing import Any +from app.categories import CATEGORIES from app.data_root import get_data_root DATA_DIR = get_data_root() @@ -226,38 +227,16 @@ def _check_variant_path( def validate() -> list[str]: errors: list[str] = [] - brands = _load("brand") - socs = _load("soc") - phones = _load("smartphone") - tablets = _load("tablet") - watches = _load("watch") - pdas = _load("pda") - gpus = _load("gpu") - cpus = _load("cpu") - laptops = _load("laptop") - monitors = _load("monitor") - software = _load("software") - websites = _load("website") + loaded = {category: _load(category) for category in CATEGORIES} + (brands, socs, phones, tablets, watches, pdas, gpus, cpus, + laptops, monitors, software, websites) = (loaded[category] for category in CATEGORIES) brand_slugs = {rec["slug"] for _, rec in brands if "slug" in rec} soc_slugs = {rec["slug"] for _, rec in socs if "slug" in rec} cpu_slugs = {rec["slug"] for _, rec in cpus if "slug" in rec} gpu_slugs = {rec["slug"] for _, rec in gpus if "slug" in rec} - for category, records in ( - ("brand", brands), - ("soc", socs), - ("smartphone", phones), - ("tablet", tablets), - ("watch", watches), - ("pda", pdas), - ("gpu", gpus), - ("cpu", cpus), - ("laptop", laptops), - ("monitor", monitors), - ("software", software), - ("website", websites), - ): + for category, records in loaded.items(): _check_unique_slugs(category, records, errors) for fname, rec in brands: diff --git a/app/verify/cli.py b/app/verify/cli.py index 8c6a6a0..71f8857 100644 --- a/app/verify/cli.py +++ b/app/verify/cli.py @@ -42,25 +42,12 @@ def _now_iso() -> str: return datetime.now(UTC).strftime("%Y-%m-%dT%H:%M:%SZ") -def _changed_data_slugs() -> set[str]: - """Repo-relative data/ paths changed vs origin/main (for CI --changed). - - Direct two-tree diff (``origin/main HEAD``), NOT three-dot ``origin/main...HEAD``: - CI fetches main shallow (``--depth=1``), so there is no merge-base and the - three-dot form silently returns nothing. A direct tree diff only needs both - commit tips, which are always present. - - Runs git in the *data* repository (DATA_DIR's parent), so it works whether this - package lives in TechAPI (data alongside) or TechEngine (data in a separate - TechAPI checkout pointed at by TECHAPI_DATA_DIR). - """ - try: - out = subprocess.run( - ["git", "diff", "--name-only", "origin/main", "HEAD", "--", "data/"], - capture_output=True, text=True, check=True, cwd=DATA_DIR.parent, - ).stdout - except Exception: - out = "" +def _changed_data_slugs(base: str = "origin/main") -> set[str]: + """Changed seed paths against the PR merge base; git errors must stay visible.""" + out = subprocess.run( + ["git", "diff", "--name-only", f"{base}...HEAD", "--", "data/"], + capture_output=True, text=True, check=True, cwd=DATA_DIR.parent, + ).stdout # strip leading "data/" so it matches Record.path paths = set() for line in out.splitlines(): @@ -97,7 +84,7 @@ def cmd_score(args: argparse.Namespace) -> int: ts = _now_iso() categories = tuple(args.category) if args.category else CATEGORIES - changed = _changed_data_slugs() if args.changed else None + changed = _changed_data_slugs(args.base) if args.changed else None # The scores cache is a full-dataset snapshot; only rewrite it on a full run. full_scope = args.category is None and args.max is None and not args.changed @@ -210,7 +197,9 @@ def _print_markdown(hist: dict[str, Counter[str]], scored: int, hard_flags: Coun f"| {cat} | {bar} | {tot} | {c['green']} | {c['yellow']} | {c['red']} | {gpct:.1f}% |" ) gtot = sum(totals.values()) or 1 - print(f"**{scored} record(s) scored.**\n") + print(f"**{scored} record(s) assessed.**\n") + print("Laptop, monitor, software and website assess required fields and sources only; " + "domain consistency rules are unavailable and these categories cannot earn green.\n") # Overall distribution as a Mermaid pie (rendered by GitHub). Mermaid colors # slices pie1/pie2/pie3 in declaration order, so pin them to green/amber/red @@ -351,8 +340,8 @@ def cmd_status(args: argparse.Namespace) -> int: def cmd_report(args: argparse.Namespace) -> int: if not SCORES_PATH.exists(): - print("no scores cache — run `python -m app.verify score` first") - return 0 + return cmd_score(argparse.Namespace(category=None, max=None, unverified_only=False, + changed=False, no_cache=True, format="text")) hist: dict[str, Counter[str]] = defaultdict(Counter) hard_flags: Counter[str] = Counter() for entry in ledger.iter_entries(SCORES_PATH): @@ -584,13 +573,13 @@ def cmd_pr(args: argparse.Namespace) -> int: Tier 0 (offline score) + Tier 1 (source-URL liveness) + Tier 2 (external cross-reference) + Tier 3 (promotion decision, DRY-RUN — never writes). Network - tiers run only over the records changed vs origin/main, capped by --max. + tiers run only over the records changed vs the PR merge base, capped by --max. """ records = load_all() _, _, soc_release = foreign_key_sets(records) now_year = offline.now_year_today() - changed = _changed_data_slugs() + changed = _changed_data_slugs(args.base) changed_recs = [ rec for cat in CATEGORIES for rec in records[cat] if rec.slug and rec.path in changed @@ -725,10 +714,11 @@ def build_parser() -> argparse.ArgumentParser: sc.add_argument("--category", nargs="*", choices=CATEGORIES, help="limit to categories") sc.add_argument("--max", type=int, default=None, help="cap number scored") sc.add_argument("--unverified-only", action="store_true", help="skip verified:true records") - sc.add_argument("--changed", action="store_true", help="only records changed vs origin/main") + sc.add_argument("--changed", action="store_true", help="only records changed vs PR merge base") sc.add_argument("--no-cache", action="store_true", help="do not write the scores cache") sc.add_argument("--format", choices=["text", "md"], default="text", help="output format: text histogram (default) or markdown table") + sc.add_argument("--base", default="origin/main", help="PR base SHA or ref") sc.set_defaults(func=cmd_score) rp = sub.add_parser("report", help="summarize latest ledger state") @@ -764,6 +754,7 @@ def build_parser() -> argparse.ArgumentParser: pm.set_defaults(func=cmd_promote) pr = sub.add_parser("pr", help="all-tiers (0-3) markdown report for a PR's changed records") + pr.add_argument("--base", default="origin/main", help="PR base SHA or ref") pr.add_argument("--max", type=int, default=40, help="cap changed records for network tiers") pr.set_defaults(func=cmd_pr) diff --git a/app/verify/common.py b/app/verify/common.py index 8921245..cdef974 100644 --- a/app/verify/common.py +++ b/app/verify/common.py @@ -16,20 +16,9 @@ from pathlib import Path from typing import Any +from app.categories import CATEGORIES as CATEGORIES from app.validate import DATA_DIR, _load -# Categories the verifier knows about, in load order. Mirrors app.validate.validate. -CATEGORIES: tuple[str, ...] = ( - "brand", - "soc", - "smartphone", - "tablet", - "watch", - "pda", - "gpu", - "cpu", -) - VERIFY_DIR = DATA_DIR / "_verify" _RAW_CHIPSET_YEAR = re.compile(r"^[^,]+,\s*((?:19|20)\d{2})\s*,") LEDGER_PATH = VERIFY_DIR / "ledger.jsonl" # git-tracked: promotion decisions only diff --git a/app/verify/offline.py b/app/verify/offline.py index 4a9927d..e1a79b2 100644 --- a/app/verify/offline.py +++ b/app/verify/offline.py @@ -14,8 +14,11 @@ from __future__ import annotations from datetime import date +from pathlib import Path from typing import Any, NamedTuple +from app import validate + from . import hosts, signals from .common import Record @@ -67,7 +70,7 @@ def _get_path(data: dict[str, Any], path: str) -> Any: def _completeness(category: str, data: dict[str, Any]) -> float: fields = RICH_FIELDS.get(category, ()) if not fields: - return W_COMPLETENESS + return 0.0 present = sum(1 for f in fields if _get_path(data, f) not in (None, "", [], {})) return W_COMPLETENESS * present / len(fields) @@ -108,6 +111,17 @@ def score_record( completeness = _completeness(rec.category, data) sigs = signals.signals_for(rec.category, data, now_year, soc_release) consistency, flags, hard_failed = _consistency(sigs) + if rec.category not in RICH_FIELDS: + # Only assess defined structural fields; absent domain rules earn no credit. + required = getattr(validate, f"{rec.category.upper()}_REQUIRED", {"slug", "name"}) + completeness = W_COMPLETENESS * sum(k in data for k in required) / len(required) + consistency = 0.0 + flags.append("domain_rules_unavailable") + if (not required.issubset(data) or data.get("slug") != Path(rec.path).stem + or not validate.SLUG_RE.fullmatch(str(data.get("slug", ""))) + or (rec.verified and not urls)): + flags.append("!structural_integrity") + hard_failed = True host, best_tier = _host_score(urls) provenance = _provenance(data, best_tier) diff --git a/integrity_check.py b/integrity_check.py index 71b33f2..fbd8f08 100644 --- a/integrity_check.py +++ b/integrity_check.py @@ -1,4 +1,4 @@ -"""One-off data-integrity scan for TechAPI CPU+GPU (structural + benchmark anomaly). +"""One-off data-integrity scan for all TechAPI categories (structural + benchmark anomaly). Complements app/validate.py (schema) with: duplicate detection, slug/file match, verified-without-source, name/tier vs core-count consistency, single>multi sanity, @@ -20,6 +20,7 @@ """ from __future__ import annotations import os, json, math, re, statistics, sys +from app.categories import CATEGORIES # Em-dash etc. in section headers must not crash on legacy consoles (e.g. cp949). try: @@ -62,14 +63,18 @@ def mad_outliers(pairs, lo=0.34, hi=3.0): def section(t): print(f"\n### {t}") -cpus = load("cpu"); gpus = load("gpu") +records = {category: load(category) for category in CATEGORIES} +cpus = records["cpu"]; gpus = records["gpu"] +print(f"scope: {len(CATEGORIES)}/12 categories ? {sum(map(len, records.values()))} records") print(f"loaded CPU={len(cpus)} GPU={len(gpus)}") # --- 1. duplicates + slug/file + verified-no-source --- section("structural") -for comp, recs in (("cpu", cpus), ("gpu", gpus)): +for comp, recs in records.items(): slugs, names = {}, {} for p, fn, d in recs: + if d.get("verified") is True and not d.get("source_urls"): + hard(f" [{comp}] verified without sources: {fn}") slugs.setdefault(d.get("slug"), []).append(fn) names.setdefault(d.get("name"), []).append(fn) if d.get("slug") != fn: @@ -77,7 +82,7 @@ def section(t): print(f"\n### {t}") for s, fl in slugs.items(): if len(fl) > 1: hard(f" [{comp}] DUP slug {s}: {sorted(fl)}") for n, fl in names.items(): - if len(fl) > 1: hard(f" [{comp}] DUP name {n!r}: {sorted(fl)}") + if comp in ("cpu", "gpu") and len(fl) > 1: hard(f" [{comp}] DUP name {n!r}: {sorted(fl)}") # --- 2. AMD Ryzen line vs DESKTOP model tier-digit (2nd digit); APU/mobile excepted --- section("CPU name/tier consistency (desktop mainstream only)") diff --git a/tests/unit/test_bot_coverage.py b/tests/unit/test_bot_coverage.py new file mode 100644 index 0000000..0d126e2 --- /dev/null +++ b/tests/unit/test_bot_coverage.py @@ -0,0 +1,160 @@ +import json +import subprocess +from pathlib import Path +from types import SimpleNamespace + +import pytest + +from app import dump_check +from app.categories import CATEGORIES, COLLECTIONS +from app.verify import cli, offline +from app.verify.common import CATEGORIES as VERIFY_CATEGORIES +from app.verify.common import Record + + +def test_all_categories_share_registry(): + assert VERIFY_CATEGORIES is CATEGORIES + assert len(CATEGORIES) == len(set(CATEGORIES)) == 12 + assert set(CATEGORIES) == { + "smartphone", "tablet", "watch", "pda", "cpu", "gpu", "soc", "laptop", + "monitor", "software", "website", "brand", + } + + +@pytest.mark.parametrize("category", ["laptop", "monitor", "software", "website", "future"]) +def test_missing_domain_rules_never_earn_green(category): + score = offline.score_record(Record(category, "example.json", { + "slug": "example", "name": "Example", "source_urls": ["https://intel.com/example"], + }), 2026, {}) + assert score.band != "green" + assert score.subscores["consistency"] == 0 + assert "domain_rules_unavailable" in score.flags + assert offline._completeness(category, {}) == 0 + + +def test_changed_paths_use_supplied_base_and_propagate_errors(monkeypatch, tmp_path): + calls = [] + + def run(argv, **kwargs): + calls.append((argv, kwargs)) + return SimpleNamespace(stdout="data/laptop/example.json\ndata/_verify/status.json\n") + + monkeypatch.setattr(cli, "DATA_DIR", tmp_path / "data") + monkeypatch.setattr(cli.subprocess, "run", run) + assert "laptop/example.json" in cli._changed_data_slugs("develop-sha") + assert "develop-sha...HEAD" in calls[0][0] + assert calls[0][1]["cwd"] == tmp_path + + def fail(*args, **kwargs): + raise subprocess.CalledProcessError(128, "git") + + monkeypatch.setattr(cli.subprocess, "run", fail) + with pytest.raises(subprocess.CalledProcessError): + cli._changed_data_slugs("missing-base") + + +def write(path: Path, data): + path.parent.mkdir(parents=True, exist_ok=True) + path.write_text(json.dumps(data), encoding="utf-8") + + +def fixture_dump(tmp_path): + collections = {} + for category, resource in COLLECTIONS.items(): + write(tmp_path / "data" / category / "sample.json", {"slug": "sample"}) + write(tmp_path / "site/public/v1" / resource / "index.json", + {"count": 1, "results": [{"slug": "sample"}]}) + collections[resource] = {"count": 1} + write(tmp_path / "site/public/v1/index.json", {"collections": collections}) + + +def test_dump_checks_parse_manifest_and_indices(monkeypatch, tmp_path): + fixture_dump(tmp_path) + changed = "site/public/v1/laptops/sample/index.json" + write(tmp_path / changed, {"slug": "sample"}) + monkeypatch.setattr(dump_check.subprocess, "run", lambda *a, **kw: + SimpleNamespace(stdout=changed + "\n")) + assert dump_check.check_dump(tmp_path, "base") == [] + (tmp_path / changed).write_text("{broken", encoding="utf-8") + assert any(changed in e for e in dump_check.check_dump(tmp_path, "base")) + write(tmp_path / changed, {}) + write(tmp_path / "site/public/v1/laptops/index.json", {"count": 2, "results": []}) + assert any("laptops: index" in e for e in dump_check.check_dump(tmp_path, "base")) + write(tmp_path / "site/public/v1/index.json", {"collections": {}}) + assert sum("manifest count" in e for e in dump_check.check_dump(tmp_path, "base")) == 12 + + +def test_dump_deletion_still_checks_counts(monkeypatch, tmp_path): + fixture_dump(tmp_path) + replies = iter(["", "site/public/v1/laptops/index.json\n"]) + monkeypatch.setattr(dump_check.subprocess, "run", lambda *a, **kw: + SimpleNamespace(stdout=next(replies))) + (tmp_path / "site/public/v1/laptops/index.json").unlink() + assert any("laptops" in e for e in dump_check.check_dump(tmp_path, "base")) + + +def test_dump_no_changes_skips_missing_dump(monkeypatch, tmp_path): + monkeypatch.setattr(dump_check.subprocess, "run", lambda *a, **kw: + SimpleNamespace(stdout="")) + assert dump_check.check_dump(tmp_path, "base") == [] + + +def test_scope_includes_all_records_and_base(tmp_path): + fixture_dump(tmp_path) + assert dump_check.scope(tmp_path, "develop", "abcdef1234") == ( + "12/12 categories ? 12 records ? diff base develop@abcdef1" + ) + + +def test_pr_base_excludes_prior_develop_changes_and_release_includes_them(monkeypatch, tmp_path): + def git(*args): + return subprocess.run(["git", *args], cwd=tmp_path, check=True, + capture_output=True, text=True).stdout.strip() + + git("init", "-b", "main") + git("config", "user.name", "Test") + git("config", "user.email", "test@example.com") + write(tmp_path / "data/gpu/old.json", {"slug": "old"}) + git("add", ".") + git("commit", "-m", "initial") + main = git("rev-parse", "HEAD") + git("switch", "-c", "develop") + write(tmp_path / "data/laptop/prior.json", {"slug": "prior"}) + git("add", ".") + git("commit", "-m", "prior develop work") + develop = git("rev-parse", "HEAD") + git("switch", "-c", "feature") + write(tmp_path / "data/website/own.json", {"slug": "own"}) + git("add", ".") + git("commit", "-m", "own work") + monkeypatch.setattr(cli, "DATA_DIR", tmp_path / "data") + assert cli._changed_data_slugs(develop) == {"website/own.json"} + assert cli._changed_data_slugs(main) == {"laptop/prior.json", "website/own.json"} + + +def test_integrity_scans_non_chip_categories(tmp_path): + for category in ("laptop", "monitor", "software", "website"): + write(tmp_path / category / "a.json", + {"slug": "wrong", "name": "Example", "verified": True, "source_urls": []}) + write(tmp_path / category / "b.json", {"slug": "wrong", "name": "Example"}) + report = tmp_path / "hard.json" + result = subprocess.run( + ["python", "integrity_check.py", str(tmp_path), "--hard-report", str(report)], + capture_output=True, text=True, encoding="utf-8", check=True, + ) + assert "12/12 categories" in result.stdout + anomalies = json.loads(report.read_text(encoding="utf-8")) + for category in ("laptop", "monitor", "software", "website"): + assert any(f"[{category}] DUP slug" in item for item in anomalies) + assert any(f"[{category}] slug!=file" in item for item in anomalies) + assert any(f"[{category}] verified without sources" in item for item in anomalies) + assert not any("DUP name" in item for item in anomalies) + + +def test_report_recomputes_without_committed_cache(monkeypatch, tmp_path): + monkeypatch.setattr(cli, "SCORES_PATH", tmp_path / "missing.jsonl") + calls = [] + monkeypatch.setattr(cli, "cmd_score", lambda args: calls.append(args) or 0) + assert cli.cmd_report(SimpleNamespace()) == 0 + assert calls[0].no_cache is True + assert calls[0].changed is False