From e0f63ad03ac65386f198bbd5af19fafedf98f2ab Mon Sep 17 00:00:00 2001 From: Thomas Schmelzer Date: Sun, 30 Aug 2026 19:16:54 +0400 Subject: [PATCH] feat(board): add a Coverage column, read from CI artifacts Coverage comes from the `coverage-report` artifact CI already uploads, so the number is tied to a commit and a branch. Two sources were rejected first: * The .coverage files lying in the checkouts. 18 of 24 repos have one, aged 1 to 16 days. It records whenever someone last ran pytest in that directory - possibly a subset of tests, possibly on a feature branch. It would be a column that looks like a fact about the repo and is actually a fact about a shell history. * test LOC / code LOC. Cheap, complete, and not coverage. Fine as a ratio, dishonest under that heading. The branch filter is the load-bearing part. Artifacts come back newest-first across every ref, and in a repo that tags releases the newest is usually a tag build - rhiza's most recent coverage artifact is from v1.7.1, not main. Taking the latest would report a release build's coverage as the repo's, measured at a different commit. There is a test for exactly this. Cost, measured rather than estimated * Listing artifacts is one call per repo and always happens, so a report published between refreshes is picked up: +25 calls, ~150/hour. * The zip is downloaded only when the artifact id changes: 19 downloads on a cold pass, 0 on the next. That measurement also showed the documented API budget was wrong - the page claimed 119 calls per refresh and ~1430/hour for a fleet of 32. Counting requests through a full refresh gives 370 and ~2220/hour for a fleet of 25. Corrected, with a note saying so. Two metrics, not one. jq_ci_coverage_percent is the figure; jq_ci_coverage_lines is what CI actually measured, and the percentage is not interpretable without it. The denominator is NOT jq_local_code_lines: CI measures whatever it pointed --cov at, so rhiza reads 100% of 176 lines while LOC counts 1477. Both are right and they answer different questions. Absent, never zero, when a repo publishes no report - six of this fleet do not, and "nobody publishes coverage here" is not the finding "nothing is covered". The column reads `no report`. A malformed or oversized artifact is logged and skipped rather than failing a refresh that has already gathered everything else, with guards on both the download and the unpacked size. --- collector/jq_collector/__main__.py | 12 +- collector/jq_collector/github.py | 136 +++++++++++++++- collector/jq_collector/metrics.py | 21 +++ collector/jq_collector/state.py | 11 ++ collector/tests/conftest.py | 19 ++- collector/tests/test_coverage.py | 241 +++++++++++++++++++++++++++++ docs/configuration.md | 20 ++- docs/dashboard.md | 14 ++ grafana/dashboards/fleet.json | 46 +++++- 9 files changed, 497 insertions(+), 23 deletions(-) create mode 100644 collector/tests/test_coverage.py diff --git a/collector/jq_collector/__main__.py b/collector/jq_collector/__main__.py index da90546..bdcda6c 100644 --- a/collector/jq_collector/__main__.py +++ b/collector/jq_collector/__main__.py @@ -44,7 +44,17 @@ def _refresh_github(cfg: Config, store: Store) -> None: ref_cache = { name: (repo.head_sha, repo.rhiza_ref) for name, repo in snap.remote.items() if repo.head_sha } - remote, api, latest, excluded = gh.collect(cfg, ref_cache) + # (artifact id, percent) per repo: an unchanged artifact means the report + # behind it is byte-identical, so there is nothing to gain from pulling the + # zip down again. + coverage_cache = { + name: (repo.coverage_artifact, (repo.coverage, repo.coverage_lines)) + if repo.coverage is not None + else (repo.coverage_artifact, None) + for name, repo in snap.remote.items() + if repo.coverage_artifact + } + remote, api, latest, excluded = gh.collect(cfg, ref_cache, coverage_cache) try: store.update( remote=remote, diff --git a/collector/jq_collector/github.py b/collector/jq_collector/github.py index ce1fd34..532c3e1 100644 --- a/collector/jq_collector/github.py +++ b/collector/jq_collector/github.py @@ -1,10 +1,16 @@ """Read the state of the repos as GitHub sees them. -Per refresh this costs roughly ``5 * repos + workflows + open_pull_requests`` -REST calls (one branch, one protection, one alert listing, one workflow-run and -one pull listing each, plus a check-run listing per open PR). There are no -conditional requests, so the cost scales with the fleet and does not fall when -nothing has changed - see JQ_GITHUB_INTERVAL before growing either. +Per refresh this costs roughly ``6 * repos + workflows + open_pull_requests`` +REST calls (one branch, one protection, one alert listing, one workflow-run, +one artifact listing and one pull listing each, plus a check-run listing per +open PR). Measured on a fleet of 25: 370 calls in the steady state. There are +no conditional requests, so the cost scales with the fleet and does not fall +when nothing has changed - see JQ_GITHUB_INTERVAL before growing either. + +The coverage artifact is the one response that is not JSON. Listing artifacts +happens every refresh so a newly published report is picked up, but the zip is +downloaded only when the artifact id has changed - measured 19 downloads on a +cold pass and 0 on the next. ``jq_github_rate_limit_remaining`` is exported so the headroom is visible rather than assumed. """ @@ -13,8 +19,11 @@ import base64 import concurrent.futures +import io import logging +import zipfile from datetime import datetime +from xml.etree import ElementTree import httpx @@ -36,6 +45,17 @@ _ACCEPT = "application/vnd.github+json" _MAX_WORKERS = 8 +# The artifact CI uploads its coverage report as, and the file to read inside +# it. Repos that publish nothing by this name simply have no coverage on the +# board - which is the honest answer, and not the same as zero. +_COVERAGE_ARTIFACT = "coverage-report" + +# Guards on an archive we did not build. Coverage reports for a fleet this size +# are tens of kilobytes; anything near these is a bug or a bomb, and unpacking +# it would be the collector's problem rather than CI's. +_MAX_ARTIFACT_BYTES = 16 * 1024 * 1024 +_MAX_UNPACKED_BYTES = 64 * 1024 * 1024 + def _ts(value: str | None) -> float: if not value: @@ -97,6 +117,17 @@ def _json(self, path: str, **params: object) -> object | None: response.raise_for_status() return response.json() + def _bytes(self, path: str) -> bytes | None: + """GET returning raw bytes. 410 is added to the expected empties: that + is what an expired artifact returns, and artifacts expire on a schedule + nobody here controls.""" + response = self._get(path) + if response.status_code in (403, 404, 409, 410, 451): + log.info("%s -> %s", path, response.status_code) + return None + response.raise_for_status() + return response.content + def _paginate(self, path: str, **params: object) -> list[dict]: items: list[dict] = [] page = 1 @@ -315,6 +346,60 @@ def latest_runs(self, full_name: str, branch: str) -> list[dict]: for wid, run in newest.items() ] + def coverage_artifact(self, full_name: str, branch: str) -> int: + """Id of the newest ``coverage-report`` artifact built on ``branch``. + + Filtering on the branch is not optional. Artifacts are returned newest + first across *every* ref, and a tag build is usually the most recent + one - rhiza's newest coverage artifact is from ``v1.7.1``, not ``main``. + Taking the latest would quietly report a release build's coverage as + the repo's, which is a different number measured at a different commit. + + Zero when the repo publishes no such artifact, which most of a mixed + fleet does not. + """ + raw = self._json(f"/repos/{full_name}/actions/artifacts", per_page=100) + if not isinstance(raw, dict): + return 0 + best_id, best_at = 0, "" + for artifact in raw.get("artifacts") or []: + if artifact.get("name") != _COVERAGE_ARTIFACT or artifact.get("expired"): + continue + if ((artifact.get("workflow_run") or {}).get("head_branch")) != branch: + continue + created = artifact.get("created_at") or "" + if created >= best_at: + best_id, best_at = int(artifact.get("id") or 0), created + return best_id + + def coverage_percent(self, full_name: str, artifact_id: int) -> tuple[float, int] | None: + """``(line coverage as a percentage, lines measured)`` from that artifact. + + The line count is carried because the percentage alone is not + interpretable. CI measures whatever it pointed ``--cov`` at, which is + the package rather than everything tracked: rhiza reports 100% of 176 + lines, while the board's own LOC column counts 1477. Both are right and + they are answering different questions, so the denominator is exported + alongside rather than left to be guessed at. + + ``branch-rate`` is deliberately not read. Branch coverage is not + enabled in this fleet's CI, so it is a constant zero and putting it on + the board would invent a finding. + """ + blob = self._bytes(f"/repos/{full_name}/actions/artifacts/{artifact_id}/zip") + if blob is None: + return None + if len(blob) > _MAX_ARTIFACT_BYTES: + log.warning("%s: coverage artifact is %d bytes, skipping", full_name, len(blob)) + return None + try: + return _coverage(blob) + except (zipfile.BadZipFile, ElementTree.ParseError, ValueError) as exc: + # A malformed report is CI's problem, not a reason to fail a + # refresh that has already gathered everything else. + log.warning("%s: could not read coverage artifact: %s", full_name, exc) + return None + def open_pulls(self, full_name: str) -> tuple[int, list[PullRequest]]: """(total open PRs, detail for the first ``max_prs_per_repo``). @@ -396,6 +481,22 @@ def checks_state(self, full_name: str, sha: str) -> str: return "success" +def _coverage(blob: bytes) -> tuple[float, int] | None: + """``(percent, lines measured)`` out of the coverage.xml in a zipped artifact.""" + with zipfile.ZipFile(io.BytesIO(blob)) as bundle: + members = [m for m in bundle.infolist() if m.filename.endswith("coverage.xml")] + if not members: + return None + member = members[0] + if member.file_size > _MAX_UNPACKED_BYTES: + raise ValueError(f"coverage.xml unpacks to {member.file_size} bytes") + root = ElementTree.fromstring(bundle.read(member)) + rate = root.get("line-rate") + if rate is None: + return None + return round(float(rate) * 100, 1), int(root.get("lines-valid") or 0) + + def _inconclusive(run: dict) -> bool: return (run.get("conclusion") or "") in INCONCLUSIVE_CONCLUSIONS @@ -412,7 +513,9 @@ def _behind_count(tags: list[str], ref: str) -> int | None: def collect( - cfg: Config, ref_cache: dict[str, tuple[str, str]] + cfg: Config, + ref_cache: dict[str, tuple[str, str]], + coverage_cache: dict[str, tuple[int, tuple[float, int] | None]] | None = None, ) -> tuple[dict[str, RemoteRepo], GitHub, str, frozenset[str]]: """Build the remote half of the snapshot, keyed by ``owner/name``. @@ -425,7 +528,15 @@ def collect( previous refresh. The pointer can only have changed if the branch head moved, so an unchanged sha skips the fetch and the steady-state cost of correctness is zero extra calls. + + ``coverage_cache`` maps ``full_name -> (artifact_id, (percent, lines))``. Listing + the artifacts is one call per repo and always happens, so a report + published between refreshes is picked up; *downloading* one only happens + when the id has changed. That matters more than the call count - the + download is a zip, by far the largest response this collector handles, and + in the steady state it is never fetched at all. """ + coverage_cache = coverage_cache or {} api = GitHub(cfg) tags = api.release_tags(cfg.template_repo) latest = tags[0] if tags else "" @@ -461,6 +572,16 @@ def one(raw: dict) -> RemoteRepo: else: ref = api.template_ref(full_name) + artifact = api.coverage_artifact(full_name, branch) + cached_coverage = coverage_cache.get(full_name) + if artifact and cached_coverage is not None and cached_coverage[0] == artifact: + measured = cached_coverage[1] + elif artifact: + measured = api.coverage_percent(full_name, artifact) + else: + measured = None + coverage, coverage_lines = measured if measured else (None, 0) + workflows = tuple( WorkflowRun( name=r.get("_name") or r.get("name") or "unnamed", @@ -516,6 +637,9 @@ def one(raw: dict) -> RemoteRepo: ci_duration=representative.duration if representative else 0.0, ci_url=representative.url if representative else "", workflows=workflows, + coverage=coverage, + coverage_lines=coverage_lines, + coverage_artifact=artifact, open_issues=open_issues, open_pulls_total=pulls_total, pulls=tuple(pulls), diff --git a/collector/jq_collector/metrics.py b/collector/jq_collector/metrics.py index 49b780e..946e128 100644 --- a/collector/jq_collector/metrics.py +++ b/collector/jq_collector/metrics.py @@ -179,6 +179,18 @@ def render(snap: Snapshot): "Per workflow: when that run finished.", ["repo", "workflow"], ) + coverage = _gauge( + "jq_ci_coverage_percent", + "Line coverage from the newest default-branch coverage-report artifact. " + "Absent when the repo publishes none.", + ["repo"], + ) + coverage_lines = _gauge( + "jq_ci_coverage_lines", + "Lines CI measured for that coverage figure. The percentage is not " + "interpretable without it, and its denominator is not jq_local_code_lines.", + ["repo"], + ) wf_failing = _gauge( "jq_ci_workflows_failing", "How many of the repo's workflows are red on the default branch.", @@ -361,6 +373,13 @@ def render(snap: Snapshot): ci_dur.add_metric(ident, remote.ci_duration) wf_failing.add_metric(ident, bad) + # Absent, not zero, when there is no report. Zero would read as + # "nothing is covered", which is a finding; "nobody publishes a + # report here" is not one. + if remote.coverage is not None: + coverage.add_metric(ident, remote.coverage) + coverage_lines.add_metric(ident, remote.coverage_lines) + pr_count.add_metric(ident, remote.open_pulls_total) issue_count.add_metric(ident, remote.open_issues) # Red means red. A cancelled check is no verdict - the same rule the @@ -440,6 +459,8 @@ def render(snap: Snapshot): wf_ok, wf_at, wf_failing, + coverage, + coverage_lines, pr_count, issue_count, pr_failing, diff --git a/collector/jq_collector/state.py b/collector/jq_collector/state.py index 5727a4e..713e7cf 100644 --- a/collector/jq_collector/state.py +++ b/collector/jq_collector/state.py @@ -88,6 +88,17 @@ class RemoteRepo: ci_duration: float = 0.0 ci_url: str = "" + # Line coverage from the newest coverage-report artifact CI built on the + # default branch, and that artifact's id. None means the repo publishes no + # such artifact, or the newest one could not be read - not that it has no + # tests. The id is what lets a refresh skip re-downloading an unchanged + # report; see github.collect. + coverage: float | None = None + # Lines CI actually measured. The percentage is not interpretable without + # it - 100% of 176 lines and 100% of 3878 are different assurances. + coverage_lines: int = 0 + coverage_artifact: int = 0 + workflows: tuple[WorkflowRun, ...] = () open_issues: int = 0 # Open PRs as GitHub counts them, before max_prs_per_repo clipping. `pulls` diff --git a/collector/tests/conftest.py b/collector/tests/conftest.py index 135d247..ab693db 100644 --- a/collector/tests/conftest.py +++ b/collector/tests/conftest.py @@ -15,15 +15,28 @@ class FakeGitHub(GitHub): them, which keeps the fixtures readable. """ - def __init__(self, cfg: Config, responses: dict[str, object]) -> None: + def __init__( + self, + cfg: Config, + responses: dict[str, object], + blobs: dict[str, bytes] | None = None, + ) -> None: super().__init__(cfg) self.responses = responses + # Raw-bytes routes, for the endpoints that return an archive rather + # than JSON. Kept apart so a test that does not care never has to + # mention them. + self.blobs = blobs or {} self.calls: list[str] = [] def _json(self, path: str, **params: object) -> object | None: self.calls.append(path) return self.responses.get(path) + def _bytes(self, path: str) -> bytes | None: + self.calls.append(path) + return self.blobs.get(path) + @pytest.fixture def cfg() -> Config: @@ -32,8 +45,8 @@ def cfg() -> Config: @pytest.fixture def make_client(cfg): - def _make(responses: dict[str, object]) -> FakeGitHub: - return FakeGitHub(cfg, responses) + def _make(responses: dict[str, object], blobs: dict[str, bytes] | None = None) -> FakeGitHub: + return FakeGitHub(cfg, responses, blobs) return _make diff --git a/collector/tests/test_coverage.py b/collector/tests/test_coverage.py new file mode 100644 index 0000000..fd3af5a --- /dev/null +++ b/collector/tests/test_coverage.py @@ -0,0 +1,241 @@ +"""Reading test coverage out of what CI already publishes. + +The collector never runs anyone's tests. It reads the `coverage-report` +artifact CI uploads, which means the number is tied to a commit and a branch +rather than to whenever somebody last ran pytest in a checkout. + +The trap these pin is the branch. Artifacts come back newest-first across every +ref, and in a repo that tags releases the newest one is usually a tag build - +rhiza's most recent coverage artifact is from `v1.7.1`, not `main`. Taking the +latest would report a release build's coverage as the repo's. +""" + +from __future__ import annotations + +import io +import zipfile + +import pytest + +from jq_collector.github import _coverage + +ARTIFACTS = "/repos/o/r/actions/artifacts" + + +def artifact(aid: int, branch: str, created: str, name: str = "coverage-report", expired=False): + return { + "id": aid, + "name": name, + "expired": expired, + "created_at": created, + "workflow_run": {"head_branch": branch}, + } + + +def zipped(xml: str, filename: str = "coverage.xml") -> bytes: + buffer = io.BytesIO() + with zipfile.ZipFile(buffer, "w") as bundle: + bundle.writestr(filename, xml) + return buffer.getvalue() + + +REPORT = '' + + +def test_the_newest_default_branch_artifact_wins_not_the_newest_overall(make_client): + """A tag build is usually the most recent artifact. It is not the answer.""" + client = make_client( + { + ARTIFACTS: { + "artifacts": [ + artifact(3, "v1.7.1", "2026-08-27T04:09:22Z"), # newest, wrong branch + artifact(2, "main", "2026-08-27T04:07:41Z"), # what we want + artifact(1, "release-prep", "2026-08-27T04:03:09Z"), + ] + } + } + ) + assert client.coverage_artifact("o/r", "main") == 2 + + +def test_expired_and_unrelated_artifacts_are_ignored(make_client): + client = make_client( + { + ARTIFACTS: { + "artifacts": [ + artifact(9, "main", "2026-08-28T00:00:00Z", expired=True), + artifact(8, "main", "2026-08-27T00:00:00Z", name="book"), + artifact(7, "main", "2026-08-26T00:00:00Z"), + ] + } + } + ) + assert client.coverage_artifact("o/r", "main") == 7 + + +def test_a_repo_publishing_no_coverage_report_is_not_an_error(make_client): + """Most of a mixed fleet does not publish one; six of this one do not.""" + client = make_client({ARTIFACTS: {"artifacts": [artifact(1, "main", "x", name="book")]}}) + assert client.coverage_artifact("o/r", "main") == 0 + + +def test_the_percentage_and_its_denominator_are_both_read(make_client): + """100% of 176 lines and 100% of 3878 are different assurances.""" + client = make_client({}, {f"{ARTIFACTS}/5/zip": zipped(REPORT)}) + assert client.coverage_percent("o/r", 5) == (87.3, 472) + + +def test_a_nested_coverage_xml_is_found(make_client): + """CI uploads it as _tests/coverage.xml, so it is not at the archive root.""" + client = make_client({}, {f"{ARTIFACTS}/5/zip": zipped(REPORT, "_tests/coverage.xml")}) + assert client.coverage_percent("o/r", 5) == (87.3, 472) + + +def test_an_expired_artifact_download_is_survivable(make_client): + """410 Gone is normal - artifacts expire on a schedule nobody here sets.""" + client = make_client({}, {}) + assert client.coverage_percent("o/r", 5) is None + + +@pytest.mark.parametrize( + "blob", + [ + b"not a zip at all", + zipped(""), # no line-rate + zipped("not xml", "coverage.xml"), # unparseable + zipped(REPORT, "something-else.txt"), # no coverage.xml inside + ], + ids=["not-a-zip", "no-line-rate", "bad-xml", "no-coverage-xml"], +) +def test_a_malformed_report_is_ci_s_problem_not_a_failed_refresh(make_client, blob): + """Everything else in the refresh has already been gathered by this point.""" + client = make_client({}, {f"{ARTIFACTS}/5/zip": blob}) + assert client.coverage_percent("o/r", 5) is None + + +def test_an_absurdly_large_report_is_refused(): + """A zip bomb would be the collector's problem, not CI's.""" + from jq_collector import github + + blob = zipped("" + " " * 1000) + original = github._MAX_UNPACKED_BYTES + github._MAX_UNPACKED_BYTES = 10 + try: + with pytest.raises(ValueError, match="unpacks to"): + _coverage(blob) + finally: + github._MAX_UNPACKED_BYTES = original + + +# -- the cache, which is what keeps this affordable -------------------------- + + +class StubAPI: + """Just enough of GitHub for collect(), counting the expensive call. + + Listing artifacts is one cheap call per repo and always happens, so a + report published between refreshes is picked up. Downloading one is a zip - + by far the largest response this collector handles - and must not happen + again while the artifact id is unchanged. + """ + + rate_remaining = rate_limit = rate_reset = 0.0 + + def __init__(self, artifact_id: int = 42) -> None: + self.artifact_id = artifact_id + self.downloads: list[str] = [] + + def list_repos(self): + return [ + { + "full_name": "o/r", + "name": "r", + "default_branch": "main", + "owner": {"login": "o"}, + "visibility": "public", + "open_issues_count": 0, + } + ] + + def release_tags(self, _full_name): + return [] + + def branch_sha(self, *_): + return "sha" + + def template_ref(self, *_): + return "" + + def branch_protection(self, *_): + return None, False + + def open_alerts(self, *_): + return None + + def latest_runs(self, *_): + return [] + + def open_pulls(self, *_): + return 0, [] + + def recent_merges(self, *_): + return [] + + def coverage_artifact(self, *_): + return self.artifact_id + + def coverage_percent(self, full_name, _artifact_id): + self.downloads.append(full_name) + return (87.3, 472) + + def close(self): + pass + + +@pytest.fixture +def stubbed(monkeypatch): + from jq_collector import github + + stub = StubAPI() + monkeypatch.setattr(github, "GitHub", lambda _cfg: stub) + return github, stub + + +def test_a_cold_refresh_downloads_the_report(stubbed, cfg): + github, stub = stubbed + remote, _api, _latest, _excluded = github.collect(cfg, {}, {}) + + assert stub.downloads == ["o/r"] + assert remote["o/r"].coverage == 87.3 + assert remote["o/r"].coverage_lines == 472 + assert remote["o/r"].coverage_artifact == 42 + + +def test_an_unchanged_artifact_is_not_downloaded_again(stubbed, cfg): + github, stub = stubbed + cache = {"o/r": (42, (87.3, 472))} + + remote, *_ = github.collect(cfg, {}, cache) + + assert stub.downloads == [], "the zip was pulled again for an unchanged artifact" + assert remote["o/r"].coverage == 87.3 + assert remote["o/r"].coverage_lines == 472 + + +def test_a_new_artifact_is_downloaded(stubbed, cfg): + github, stub = stubbed + remote, *_ = github.collect(cfg, {}, {"o/r": (41, (10.0, 100))}) + + assert stub.downloads == ["o/r"] + assert remote["o/r"].coverage == 87.3 + + +def test_a_repo_that_stops_publishing_loses_its_coverage(stubbed, cfg): + """Absent, not the last value it happened to have.""" + github, stub = stubbed + stub.artifact_id = 0 + + remote, *_ = github.collect(cfg, {}, {"o/r": (42, (87.3, 472))}) + + assert remote["o/r"].coverage is None + assert stub.downloads == [] diff --git a/docs/configuration.md b/docs/configuration.md index 7526b54..fd18bb5 100644 --- a/docs/configuration.md +++ b/docs/configuration.md @@ -98,12 +98,20 @@ checkouts, so only the GitHub panels have anything to say. ## API budget -A refresh costs roughly `3 × repos + open PRs` REST calls. Measured on this -fleet of 32: **119 calls** for a steady-state refresh, or ~1430/hour at the -default cadence, against an authenticated budget of 5000. A cold start adds one -pointer read per repo; after that the sha-cache skips them — measured 32 pointer -reads on the first refresh and **1** on the second, that one being the single -repo whose default branch had moved. `jq_github_rate_limit_remaining` is on the *Collector health* row so the +A refresh costs roughly `6 × repos + workflows + open PRs` REST calls. Measured +on this fleet of 25: **370 calls** for a steady-state refresh, or ~2220/hour at +the default 600s cadence, against an authenticated budget of 5000. + +> An earlier version of this page said 119 calls and ~1430/hour. That was +> understated; the numbers above were measured by counting requests through a +> full refresh rather than derived from the formula. + +Two caches keep it there. A cold start adds one pointer read per repo; after +that the sha-cache skips them — measured 32 pointer reads on the first refresh +and **1** on the second, that one being the single repo whose default branch +had moved. Coverage adds one artifact listing per repo (25 calls, ~150/hour), +but the report itself — a zip, the largest response here — is downloaded only +when the artifact id changes: **19 downloads cold, 0 on the next refresh.** `jq_github_rate_limit_remaining` is on the *Collector health* row so the headroom is visible rather than assumed. Lower `JQ_GITHUB_INTERVAL` only if that number stays comfortable. diff --git a/docs/dashboard.md b/docs/dashboard.md index 68df28c..c04ceed 100644 --- a/docs/dashboard.md +++ b/docs/dashboard.md @@ -52,6 +52,20 @@ Working copy, so uncommitted work counts. A file is test code if it sits under `tests/`, `test/` or `testing/`, or if it is named `test_*` / `*_test.*` — both conventions are in this fleet. +**Coverage is different in kind from its neighbours.** Every other column is +read off your working copy and refreshes within a minute. Coverage comes from +the newest `coverage-report` artifact CI built **on the default branch**, so it +lags a push by a CI run and says nothing about uncommitted work. Repos that +publish no such artifact read `no report`, which is not `0%` — six of this +fleet are in that position. + +Its denominator is whatever CI pointed `--cov` at, and **that is not the LOC +column**: rhiza reads 100% of 176 measured lines while LOC counts 1477. Both +are right; they answer different questions. `jq_ci_coverage_lines` carries the +denominator so the percentage can be read honestly. Note also that the branch +filter is load-bearing — artifacts come back newest-first across every ref, and +in a repo that tags releases the newest is usually a tag build. + **The two commit counts are taken on the default branch**, not on whatever branch the clone is parked on: a repo's cadence is what landed on main, not what you happen to have checked out. *Unreleased* counts commits since the newest tag diff --git a/grafana/dashboards/fleet.json b/grafana/dashboards/fleet.json index 3978f86..4d4f9e2 100644 --- a/grafana/dashboards/fleet.json +++ b/grafana/dashboards/fleet.json @@ -2090,7 +2090,7 @@ { "type": "table", "title": "Size and cadence", - "description": "One row per checkout. LOC and Tests are lines of tracked source in the working copy - uncommitted work included, config and prose excluded - split on whether the file lives in a test tree or is named like a test. The two commit counts are taken on the default branch, not on whatever branch the clone is parked on. 'Unreleased' counts commits since the newest tag *in this clone*, so a release published since the last fetch is not reflected yet - read it next to the fetch age under Local working copies. All five are re-measured only when the clone moves, so a flat line here means a quiet repo, not a stuck collector.", + "description": "One row per checkout. LOC and Tests are lines of tracked source in the working copy - uncommitted work included, config and prose excluded - split on whether the file lives in a test tree or is named like a test. Coverage is different in kind: it comes from the newest coverage-report artifact CI built on the default branch, so it lags a push by a CI run and says nothing about uncommitted work. Its denominator is whatever CI pointed --cov at, which is NOT the LOC column - rhiza reads 100% of 176 measured lines while LOC counts 1477. jq_ci_coverage_lines carries that denominator. 'no report' means the repo publishes no such artifact, which is not the same as 0%. The two commit counts are taken on the default branch, not on whatever branch the clone is parked on. 'Unreleased' counts commits since the newest tag *in this clone*, so a release published since the last fetch is not reflected yet. LOC, Tests and the commit counts are re-measured only when the clone moves, so a flat line there means a quiet repo, not a stuck collector.", "gridPos": { "h": 13, "w": 24, @@ -2137,6 +2137,12 @@ "expr": "jq_local_last_release_info{repo=~\"$repo\"}", "instant": true, "format": "table" + }, + { + "refId": "G", + "expr": "jq_ci_coverage_percent{repo=~\"$repo\"}", + "instant": true, + "format": "table" } ], "transformations": [ @@ -2151,7 +2157,7 @@ "id": "filterFieldsByName", "options": { "include": { - "pattern": "^(repo|ref|Value #A|Value #B|Value #C|Value #D|Value #E)$" + "pattern": "^(repo|ref|Value #A|Value #B|Value #C|Value #D|Value #E|Value #G)$" } } }, @@ -2163,10 +2169,11 @@ "repo": 0, "Value #A": 1, "Value #B": 2, - "Value #C": 3, - "Value #D": 4, - "Value #E": 5, - "ref": 6 + "Value #G": 3, + "Value #C": 4, + "Value #D": 5, + "Value #E": 6, + "ref": 7 }, "renameByName": { "repo": "Repo", @@ -2175,7 +2182,8 @@ "Value #C": "Last commit", "Value #D": "Commits (30d)", "Value #E": "Unreleased", - "ref": "Release" + "ref": "Release", + "Value #G": "Coverage" } } }, @@ -2272,6 +2280,30 @@ "value": "\u2013" } ] + }, + { + "matcher": { + "id": "byName", + "options": "Coverage" + }, + "properties": [ + { + "id": "unit", + "value": "percent" + }, + { + "id": "decimals", + "value": 1 + }, + { + "id": "noValue", + "value": "no report" + }, + { + "id": "custom.width", + "value": 110 + } + ] } ] },