Skip to content
Merged
Show file tree
Hide file tree
Changes from all commits
Commits
File filter

Filter by extension

Filter by extension

Conversations
Failed to load comments.
Loading
Jump to
Jump to file
Failed to load files.
Loading
Diff view
Diff view
12 changes: 11 additions & 1 deletion collector/jq_collector/__main__.py
Original file line number Diff line number Diff line change
Expand Up @@ -44,7 +44,17 @@ def _refresh_github(cfg: Config, store: Store) -> None:
ref_cache = {
name: (repo.head_sha, repo.rhiza_ref) for name, repo in snap.remote.items() if repo.head_sha
}
remote, api, latest, excluded = gh.collect(cfg, ref_cache)
# (artifact id, percent) per repo: an unchanged artifact means the report
# behind it is byte-identical, so there is nothing to gain from pulling the
# zip down again.
coverage_cache = {
name: (repo.coverage_artifact, (repo.coverage, repo.coverage_lines))
if repo.coverage is not None
else (repo.coverage_artifact, None)
for name, repo in snap.remote.items()
if repo.coverage_artifact
}
remote, api, latest, excluded = gh.collect(cfg, ref_cache, coverage_cache)
try:
store.update(
remote=remote,
Expand Down
136 changes: 130 additions & 6 deletions collector/jq_collector/github.py
Original file line number Diff line number Diff line change
@@ -1,10 +1,16 @@
"""Read the state of the repos as GitHub sees them.

Per refresh this costs roughly ``5 * repos + workflows + open_pull_requests``
REST calls (one branch, one protection, one alert listing, one workflow-run and
one pull listing each, plus a check-run listing per open PR). There are no
conditional requests, so the cost scales with the fleet and does not fall when
nothing has changed - see JQ_GITHUB_INTERVAL before growing either.
Per refresh this costs roughly ``6 * repos + workflows + open_pull_requests``
REST calls (one branch, one protection, one alert listing, one workflow-run,
one artifact listing and one pull listing each, plus a check-run listing per
open PR). Measured on a fleet of 25: 370 calls in the steady state. There are
no conditional requests, so the cost scales with the fleet and does not fall
when nothing has changed - see JQ_GITHUB_INTERVAL before growing either.

The coverage artifact is the one response that is not JSON. Listing artifacts
happens every refresh so a newly published report is picked up, but the zip is
downloaded only when the artifact id has changed - measured 19 downloads on a
cold pass and 0 on the next.
``jq_github_rate_limit_remaining`` is exported so the headroom is visible
rather than assumed.
"""
Expand All @@ -13,8 +19,11 @@

import base64
import concurrent.futures
import io
import logging
import zipfile
from datetime import datetime
from xml.etree import ElementTree

import httpx

Expand All @@ -36,6 +45,17 @@
_ACCEPT = "application/vnd.github+json"
_MAX_WORKERS = 8

# The artifact CI uploads its coverage report as, and the file to read inside
# it. Repos that publish nothing by this name simply have no coverage on the
# board - which is the honest answer, and not the same as zero.
_COVERAGE_ARTIFACT = "coverage-report"

# Guards on an archive we did not build. Coverage reports for a fleet this size
# are tens of kilobytes; anything near these is a bug or a bomb, and unpacking
# it would be the collector's problem rather than CI's.
_MAX_ARTIFACT_BYTES = 16 * 1024 * 1024
_MAX_UNPACKED_BYTES = 64 * 1024 * 1024


def _ts(value: str | None) -> float:
if not value:
Expand Down Expand Up @@ -97,6 +117,17 @@ def _json(self, path: str, **params: object) -> object | None:
response.raise_for_status()
return response.json()

def _bytes(self, path: str) -> bytes | None:
"""GET returning raw bytes. 410 is added to the expected empties: that
is what an expired artifact returns, and artifacts expire on a schedule
nobody here controls."""
response = self._get(path)
if response.status_code in (403, 404, 409, 410, 451):
log.info("%s -> %s", path, response.status_code)
return None
response.raise_for_status()
return response.content

def _paginate(self, path: str, **params: object) -> list[dict]:
items: list[dict] = []
page = 1
Expand Down Expand Up @@ -315,6 +346,60 @@ def latest_runs(self, full_name: str, branch: str) -> list[dict]:
for wid, run in newest.items()
]

def coverage_artifact(self, full_name: str, branch: str) -> int:
"""Id of the newest ``coverage-report`` artifact built on ``branch``.

Filtering on the branch is not optional. Artifacts are returned newest
first across *every* ref, and a tag build is usually the most recent
one - rhiza's newest coverage artifact is from ``v1.7.1``, not ``main``.
Taking the latest would quietly report a release build's coverage as
the repo's, which is a different number measured at a different commit.

Zero when the repo publishes no such artifact, which most of a mixed
fleet does not.
"""
raw = self._json(f"/repos/{full_name}/actions/artifacts", per_page=100)
if not isinstance(raw, dict):
return 0
best_id, best_at = 0, ""
for artifact in raw.get("artifacts") or []:
if artifact.get("name") != _COVERAGE_ARTIFACT or artifact.get("expired"):
continue
if ((artifact.get("workflow_run") or {}).get("head_branch")) != branch:
continue
created = artifact.get("created_at") or ""
if created >= best_at:
best_id, best_at = int(artifact.get("id") or 0), created
return best_id

def coverage_percent(self, full_name: str, artifact_id: int) -> tuple[float, int] | None:
"""``(line coverage as a percentage, lines measured)`` from that artifact.

The line count is carried because the percentage alone is not
interpretable. CI measures whatever it pointed ``--cov`` at, which is
the package rather than everything tracked: rhiza reports 100% of 176
lines, while the board's own LOC column counts 1477. Both are right and
they are answering different questions, so the denominator is exported
alongside rather than left to be guessed at.

``branch-rate`` is deliberately not read. Branch coverage is not
enabled in this fleet's CI, so it is a constant zero and putting it on
the board would invent a finding.
"""
blob = self._bytes(f"/repos/{full_name}/actions/artifacts/{artifact_id}/zip")
if blob is None:
return None
if len(blob) > _MAX_ARTIFACT_BYTES:
log.warning("%s: coverage artifact is %d bytes, skipping", full_name, len(blob))
return None
try:
return _coverage(blob)
except (zipfile.BadZipFile, ElementTree.ParseError, ValueError) as exc:
# A malformed report is CI's problem, not a reason to fail a
# refresh that has already gathered everything else.
log.warning("%s: could not read coverage artifact: %s", full_name, exc)
return None

def open_pulls(self, full_name: str) -> tuple[int, list[PullRequest]]:
"""(total open PRs, detail for the first ``max_prs_per_repo``).

Expand Down Expand Up @@ -396,6 +481,22 @@ def checks_state(self, full_name: str, sha: str) -> str:
return "success"


def _coverage(blob: bytes) -> tuple[float, int] | None:
"""``(percent, lines measured)`` out of the coverage.xml in a zipped artifact."""
with zipfile.ZipFile(io.BytesIO(blob)) as bundle:
members = [m for m in bundle.infolist() if m.filename.endswith("coverage.xml")]
if not members:
return None
member = members[0]
if member.file_size > _MAX_UNPACKED_BYTES:
raise ValueError(f"coverage.xml unpacks to {member.file_size} bytes")
root = ElementTree.fromstring(bundle.read(member))
rate = root.get("line-rate")
if rate is None:
return None
return round(float(rate) * 100, 1), int(root.get("lines-valid") or 0)


def _inconclusive(run: dict) -> bool:
return (run.get("conclusion") or "") in INCONCLUSIVE_CONCLUSIONS

Expand All @@ -412,7 +513,9 @@ def _behind_count(tags: list[str], ref: str) -> int | None:


def collect(
cfg: Config, ref_cache: dict[str, tuple[str, str]]
cfg: Config,
ref_cache: dict[str, tuple[str, str]],
coverage_cache: dict[str, tuple[int, tuple[float, int] | None]] | None = None,
) -> tuple[dict[str, RemoteRepo], GitHub, str, frozenset[str]]:
"""Build the remote half of the snapshot, keyed by ``owner/name``.

Expand All @@ -425,7 +528,15 @@ def collect(
previous refresh. The pointer can only have changed if the branch head
moved, so an unchanged sha skips the fetch and the steady-state cost of
correctness is zero extra calls.

``coverage_cache`` maps ``full_name -> (artifact_id, (percent, lines))``. Listing
the artifacts is one call per repo and always happens, so a report
published between refreshes is picked up; *downloading* one only happens
when the id has changed. That matters more than the call count - the
download is a zip, by far the largest response this collector handles, and
in the steady state it is never fetched at all.
"""
coverage_cache = coverage_cache or {}
api = GitHub(cfg)
tags = api.release_tags(cfg.template_repo)
latest = tags[0] if tags else ""
Expand Down Expand Up @@ -461,6 +572,16 @@ def one(raw: dict) -> RemoteRepo:
else:
ref = api.template_ref(full_name)

artifact = api.coverage_artifact(full_name, branch)
cached_coverage = coverage_cache.get(full_name)
if artifact and cached_coverage is not None and cached_coverage[0] == artifact:
measured = cached_coverage[1]
elif artifact:
measured = api.coverage_percent(full_name, artifact)
else:
measured = None
coverage, coverage_lines = measured if measured else (None, 0)

workflows = tuple(
WorkflowRun(
name=r.get("_name") or r.get("name") or "unnamed",
Expand Down Expand Up @@ -516,6 +637,9 @@ def one(raw: dict) -> RemoteRepo:
ci_duration=representative.duration if representative else 0.0,
ci_url=representative.url if representative else "",
workflows=workflows,
coverage=coverage,
coverage_lines=coverage_lines,
coverage_artifact=artifact,
open_issues=open_issues,
open_pulls_total=pulls_total,
pulls=tuple(pulls),
Expand Down
21 changes: 21 additions & 0 deletions collector/jq_collector/metrics.py
Original file line number Diff line number Diff line change
Expand Up @@ -179,6 +179,18 @@ def render(snap: Snapshot):
"Per workflow: when that run finished.",
["repo", "workflow"],
)
coverage = _gauge(
"jq_ci_coverage_percent",
"Line coverage from the newest default-branch coverage-report artifact. "
"Absent when the repo publishes none.",
["repo"],
)
coverage_lines = _gauge(
"jq_ci_coverage_lines",
"Lines CI measured for that coverage figure. The percentage is not "
"interpretable without it, and its denominator is not jq_local_code_lines.",
["repo"],
)
wf_failing = _gauge(
"jq_ci_workflows_failing",
"How many of the repo's workflows are red on the default branch.",
Expand Down Expand Up @@ -361,6 +373,13 @@ def render(snap: Snapshot):
ci_dur.add_metric(ident, remote.ci_duration)
wf_failing.add_metric(ident, bad)

# Absent, not zero, when there is no report. Zero would read as
# "nothing is covered", which is a finding; "nobody publishes a
# report here" is not one.
if remote.coverage is not None:
coverage.add_metric(ident, remote.coverage)
coverage_lines.add_metric(ident, remote.coverage_lines)

pr_count.add_metric(ident, remote.open_pulls_total)
issue_count.add_metric(ident, remote.open_issues)
# Red means red. A cancelled check is no verdict - the same rule the
Expand Down Expand Up @@ -440,6 +459,8 @@ def render(snap: Snapshot):
wf_ok,
wf_at,
wf_failing,
coverage,
coverage_lines,
pr_count,
issue_count,
pr_failing,
Expand Down
11 changes: 11 additions & 0 deletions collector/jq_collector/state.py
Original file line number Diff line number Diff line change
Expand Up @@ -88,6 +88,17 @@ class RemoteRepo:
ci_duration: float = 0.0
ci_url: str = ""

# Line coverage from the newest coverage-report artifact CI built on the
# default branch, and that artifact's id. None means the repo publishes no
# such artifact, or the newest one could not be read - not that it has no
# tests. The id is what lets a refresh skip re-downloading an unchanged
# report; see github.collect.
coverage: float | None = None
# Lines CI actually measured. The percentage is not interpretable without
# it - 100% of 176 lines and 100% of 3878 are different assurances.
coverage_lines: int = 0
coverage_artifact: int = 0

workflows: tuple[WorkflowRun, ...] = ()
open_issues: int = 0
# Open PRs as GitHub counts them, before max_prs_per_repo clipping. `pulls`
Expand Down
19 changes: 16 additions & 3 deletions collector/tests/conftest.py
Original file line number Diff line number Diff line change
Expand Up @@ -15,15 +15,28 @@ class FakeGitHub(GitHub):
them, which keeps the fixtures readable.
"""

def __init__(self, cfg: Config, responses: dict[str, object]) -> None:
def __init__(
self,
cfg: Config,
responses: dict[str, object],
blobs: dict[str, bytes] | None = None,
) -> None:
super().__init__(cfg)
self.responses = responses
# Raw-bytes routes, for the endpoints that return an archive rather
# than JSON. Kept apart so a test that does not care never has to
# mention them.
self.blobs = blobs or {}
self.calls: list[str] = []

def _json(self, path: str, **params: object) -> object | None:
self.calls.append(path)
return self.responses.get(path)

def _bytes(self, path: str) -> bytes | None:
self.calls.append(path)
return self.blobs.get(path)


@pytest.fixture
def cfg() -> Config:
Expand All @@ -32,8 +45,8 @@ def cfg() -> Config:

@pytest.fixture
def make_client(cfg):
def _make(responses: dict[str, object]) -> FakeGitHub:
return FakeGitHub(cfg, responses)
def _make(responses: dict[str, object], blobs: dict[str, bytes] | None = None) -> FakeGitHub:
return FakeGitHub(cfg, responses, blobs)

return _make

Expand Down
Loading
Loading