-
Notifications
You must be signed in to change notification settings - Fork 0
Expand file tree
/
Copy pathrun_tests.py
More file actions
396 lines (337 loc) · 17.4 KB
/
Copy pathrun_tests.py
File metadata and controls
396 lines (337 loc) · 17.4 KB
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
150
151
152
153
154
155
156
157
158
159
160
161
162
163
164
165
166
167
168
169
170
171
172
173
174
175
176
177
178
179
180
181
182
183
184
185
186
187
188
189
190
191
192
193
194
195
196
197
198
199
200
201
202
203
204
205
206
207
208
209
210
211
212
213
214
215
216
217
218
219
220
221
222
223
224
225
226
227
228
229
230
231
232
233
234
235
236
237
238
239
240
241
242
243
244
245
246
247
248
249
250
251
252
253
254
255
256
257
258
259
260
261
262
263
264
265
266
267
268
269
270
271
272
273
274
275
276
277
278
279
280
281
282
283
284
285
286
287
288
289
290
291
292
293
294
295
296
297
298
299
300
301
302
303
304
305
306
307
308
309
310
311
312
313
314
315
316
317
318
319
320
321
322
323
324
325
326
327
328
329
330
331
332
333
334
335
336
337
338
339
340
341
342
343
344
345
346
347
348
349
350
351
352
353
354
355
356
357
358
359
360
361
362
363
364
365
366
367
368
369
370
371
372
373
374
375
376
377
378
379
380
381
382
383
384
385
386
387
388
389
390
391
392
393
394
395
396
#!/usr/bin/env python3
"""TOS centralized test harness — parallel.
Runs EVERY test in one pass and reports one total:
• TOS-Dev unit tests usr/lib/tests/test_*.lua
• Optional Utilities tests ../TOS-Extras (modules, build, pane-ui)
• Build tooling tests build/test_*.py (pytest — see below)
Almost everything here is Lua, spawned as `lua <file>`, because the tests
load and run the real TOS modules. build/sync-emulator.py is plain Python
with no TOS module to load, so its test is plain Python too — run via
`python -m pytest` instead. This is also why it can't just live where a
bare `pytest` from the monorepo root would find it on its own: pytest's
default norecursedirs skips any directory literally named "build", and
sync-emulator.py's test sits in build/ beside the script it tests. Lua
tests have the identical problem for a different reason (pytest cannot
run Lua at all), which is the whole reason this file exists — so a
Python test that needs its own explicit invocation is not a new class
of gap, it is the same one, and this is where it gets closed.
The tests themselves stay in Lua and always will: they load and execute
the real TOS modules (kernel/pkg.lua, panels/ui.lua, calc/sheet.lua, …),
which is the whole point of them. Python is here only as the RUNNER,
because the work is ~110 independent `lua` process spawns and the shell
version ran them one at a time.
Why not just parallelise run_tests.sh? Because correct parallelism needs
three things bash makes awkward: deterministic output ordering regardless
of completion order, per-test timeouts that actually kill the child, and
an isolation rule for the one test that touches shared disk state. Those
are the interesting parts of this file.
Usage:
python run_tests.py # all tests, parallel
python run_tests.py -j 4 # cap workers
python run_tests.py --serial # one at a time (bisecting a flake)
python run_tests.py -k calc -k mesh # only matching names
python run_tests.py -v # print each test's output too
Dev-only: excluded from TOS-Release by build-release.{sh,cmd} and guarded
by usr/lib/tests/test_release_excludes.lua — keep it that way.
"""
from __future__ import annotations
import argparse
import os
import re
import subprocess
import sys
import time
from concurrent.futures import ThreadPoolExecutor
from dataclasses import dataclass
from pathlib import Path
DEV_DIR = Path(__file__).resolve().parent
# TOS-Extras sits in one of two places and both are correct. In the local
# monorepo it is a SIBLING of TOS-Dev; on the published dev branch it is
# INSIDE the repo root, because a contributor gets one clone and a sibling
# would be outside it. Prefer the sibling so the monorepo keeps working,
# fall back to the nested copy so a contributor's add-on tests actually run
# -- without this they get the add-on source but none of its 21 tests.
_SIBLING_EXTRAS = DEV_DIR.parent / "TOS-Extras"
_NESTED_EXTRAS = DEV_DIR / "TOS-Extras"
EXTRAS_DIR = _SIBLING_EXTRAS if _SIBLING_EXTRAS.is_dir() else _NESTED_EXTRAS
DEFAULT_TIMEOUT = int(os.environ.get("TEST_TIMEOUT", "120"))
# A pass needs BOTH the marker AND a clean exit (inherited from the shell
# harness, and it was there for a reason): marker-only matching let a test
# print its success line and then crash in teardown while still counting
# as a pass.
PASS_MARKERS = ("All tests passed.", "*** ALL TESTS PASSED ***")
FAIL_MARKER = "*** TESTS FAILED ***"
# Likewise a skip requires a clean exit, so a real failure whose output
# happens to mention "run inside TOS" isn't misfiled as skipped.
SKIP_PATTERN = re.compile(r"not available; run inside TOS|run inside TOS", re.I)
# ── Tests that need the add-on source tree ───────────────────────────
# TOS-Extras is a SIBLING of this directory and is not published to any
# branch yet, so a contributor cloning `dev` alone has ~14 tests fail for
# a tree they were never given. Those are not failures of their checkout;
# reporting them as such teaches people that red is normal, which is the
# fastest way to lose a test suite.
#
# Rather than maintain a hand-written list of which tests need Extras --
# which drifts the moment someone adds one -- infer it from the test's
# own SOURCE: a test that names TOS-Extras is one that drives it. That
# signal is exact (all 14 such tests reference the path they search) and
# self-maintaining, where matching on failure OUTPUT is not: most of them
# fail with a package-specific message like "could not load
# tape-authenticator" that never mentions the tree at all.
#
# Gated on the tree actually being absent, so on a full checkout this
# never fires and a genuine failure is still reported as a failure.
# Absent means absent from BOTH places -- see EXTRAS_DIR above. Getting
# this wrong in the nested layout would skip 14 tests the contributor can
# actually run, which is the failure this whole mechanism exists to avoid.
EXTRAS_ABSENT = not EXTRAS_DIR.is_dir()
def needs_extras(test: Path) -> bool:
try:
return "TOS-Extras" in test.read_text(encoding="utf-8", errors="replace")
except OSError:
return False
# ── Tests that need the vendored mod API table ───────────────────────
# Same situation, one tree over. test_component_api.lua checks every
# component call against Reference/oc-component-api/oc_api.lua, which is
# vendored in the monorepo as a SIBLING of TOS-Dev and is not on the
# published dev branch -- so on a fresh clone that test was the one red
# line in an otherwise green suite. The test's own rule stands: where
# Reference/ exists, a missing table is still a loud failure (deleted or
# moved). Only a Reference/ tree absent from both places is a skip.
_SIBLING_REFERENCE = DEV_DIR.parent / "Reference"
_NESTED_REFERENCE = DEV_DIR / "Reference"
REFERENCE_ABSENT = not (_SIBLING_REFERENCE.is_dir() or _NESTED_REFERENCE.is_dir())
def needs_reference(test: Path) -> bool:
try:
return "Reference/oc-component-api" in test.read_text(encoding="utf-8", errors="replace")
except OSError:
return False
# ── Isolation ────────────────────────────────────────────────────────
# Almost every test is a pure function of its source tree: it spawns its
# own `lua`, reads files, and writes nothing. `test_build_disk.lua` is the
# exception — it RUNS THE ASSEMBLER into dist/.test-build and
# dist/.test-install, then asserts on what landed there. Two copies racing
# in the same scratch dirs would be a spectacular source of phantom
# failures, and worse, of phantom PASSES.
#
# It's also the slowest single test (it shells out to the builder several
# times), so rather than serialise everything, mark the exclusive tests
# and run them FIRST, alone, before the parallel pool starts. One-off cost,
# zero risk, and the pool still gets the long tail.
#
# Three other tests were briefly listed here, on the theory that their
# intermittent failures were pool contention. That was wrong, and the
# record is worth keeping straight: they enumerate the tree with io.popen,
# and were calling POSIX `ls` and `find`. Native Lua routes io.popen
# through cmd.exe whatever shell started the suite, so those resolved only
# when Git's bin happened to be on PATH -- deterministic failure from
# cmd.exe, and intermittent failure under 16 concurrent PATH lookups from
# Git Bash. Same cause, two faces. They now call `dir`/`cd`, which are cmd
# builtins and need no PATH lookup at all, and they have been stable in
# the pool since. Serialising them fixed nothing; it only hid it.
EXCLUSIVE = {"test_build_disk.lua"}
@dataclass
class Result:
path: str # display path, relative + forward slashes
status: str # "pass" | "fail" | "skip" | "timeout"
seconds: float
output: str
returncode: int
@property
def ok(self) -> bool:
return self.status in ("pass", "skip")
def discover() -> list[tuple[Path, Path]]:
"""(test file, working directory) pairs, in a stable order.
The cwd matters: each suite's tests resolve sibling modules with
relative paths, so Extras tests must run from the Extras root exactly
as the shell harness did (`cd "$EXTRAS"`).
"""
found: list[tuple[Path, Path]] = []
dev_tests = sorted((DEV_DIR / "usr" / "lib" / "tests").glob("test_*.lua"))
found += [(t, DEV_DIR) for t in dev_tests]
# Python-based tests for the build tooling (sync-emulator.py has no TOS
# module to load, so its test is plain Python — see the module
# docstring for why pytest can't just find this on its own).
py_tests = sorted((DEV_DIR / "build").glob("test_*.py"))
found += [(t, DEV_DIR) for t in py_tests]
if EXTRAS_DIR.is_dir():
extras: list[Path] = []
extras += sorted(EXTRAS_DIR.glob("modules/*/test_*.lua"))
extras += sorted(EXTRAS_DIR.glob("build/test_*.lua"))
extras += sorted(EXTRAS_DIR.glob("pane-ui/test_*.lua"))
# Add-ons that don't live under modules/ still ship tests. These two
# were written and then silently never run for want of a glob —
# rbmk/test_rbmk.lua alone is 72 assertions the suite was reporting
# nothing about. Keep this list in step with run_tests.sh.
extras += sorted(EXTRAS_DIR.glob("rbmk/test_*.lua"))
extras += sorted(EXTRAS_DIR.glob("cluster/*/test_*.lua"))
found += [(t, EXTRAS_DIR) for t in extras]
return found
def display_path(test: Path, cwd: Path) -> str:
# Where the file really is, relative to TOS-Dev. The prefix used to be
# hard-coded "../TOS-Extras/", which is right for the monorepo's SIBLING
# layout and wrong for the published dev branch, where TOS-Extras is
# nested inside the repo: every add-on test was shown at a path outside
# the checkout. relpath gives "../TOS-Extras/..." and "TOS-Extras/..."
# respectively, from the same line. (cwd is unused now, kept so callers
# and -k filtering are unchanged.)
return Path(os.path.relpath(test, DEV_DIR)).as_posix()
def classify(out: str, rc: int) -> str:
has_pass = any(m in out for m in PASS_MARKERS)
has_fail = FAIL_MARKER in out
has_skip = bool(SKIP_PATTERN.search(out))
if rc == 0 and has_pass:
return "pass"
if rc == 0 and has_skip:
return "skip"
# Passed then died in teardown — a real failure the marker would hide.
if has_pass and rc != 0:
return "fail"
if has_fail or rc != 0:
return "fail"
# Clean exit, no recognisable marker: surface it rather than assume.
return "fail"
def classify_py(rc: int) -> str:
# pytest's own exit code is the whole story: 0 is every test passed,
# 5 is "no tests were collected" (a renamed test function silently
# vanishing counts as a failure here, not a quiet no-op), anything
# else is a real failure or error. There is no skip marker to hunt
# for the way the Lua harness's SKIP_PATTERN does — a Python test
# that wants to skip uses pytest's own skip mechanism, which already
# exits 0.
return "pass" if rc == 0 else "fail"
def run_one(test: Path, cwd: Path, timeout: int) -> Result:
started = time.monotonic()
is_py = test.suffix == ".py"
cmd = [sys.executable, "-m", "pytest", "-q", str(test)] if is_py else ["lua", str(test)]
try:
proc = subprocess.run(
cmd,
cwd=str(cwd),
capture_output=True,
text=True,
errors="replace",
timeout=timeout,
)
out = (proc.stdout or "") + (proc.stderr or "")
rc = proc.returncode
status = classify_py(rc) if is_py else classify(out, rc)
# A failure that is really "you were not given this tree" -- see
# the EXTRAS_ABSENT note above. Only ever downgrades a fail, and
# only when the sibling tree genuinely is not there.
if status == "fail" and EXTRAS_ABSENT and needs_extras(test):
status = "skip"
if status == "fail" and REFERENCE_ABSENT and needs_reference(test):
status = "skip"
except subprocess.TimeoutExpired as e:
out = (e.stdout or "") if isinstance(e.stdout, str) else ""
out += f"\n*** TIMED OUT after {timeout}s ***"
rc, status = 124, "timeout"
except FileNotFoundError:
print(f"error: `{cmd[0]}` not found in PATH", file=sys.stderr)
raise SystemExit(2)
return Result(display_path(test, cwd), status, time.monotonic() - started, out, rc)
SYMBOL = {"pass": "ok ", "skip": "skip", "fail": "FAIL", "timeout": "TIME"}
def report(r: Result, verbose: bool) -> None:
print(f" {SYMBOL[r.status]} {r.path} ({r.seconds:.1f}s)")
if r.status in ("fail", "timeout"):
# Show the failing assertions, then a tail for context — the same
# information the shell harness surfaced, minus the noise.
lines = r.output.splitlines()
interesting = [l for l in lines if "FAIL:" in l or "Results:" in l]
for line in interesting[:20]:
print(f" {line.strip()}")
if not interesting:
for line in lines[-8:]:
print(f" {line.rstrip()}")
print(f" (exit {r.returncode})")
elif verbose:
for line in r.output.splitlines():
print(f" {line.rstrip()}")
def main() -> int:
ap = argparse.ArgumentParser(description="Run the TOS test suite in parallel.")
ap.add_argument("-j", "--jobs", type=int, default=0,
help="worker count (default: CPU count, capped at 16)")
ap.add_argument("--serial", action="store_true",
help="run one at a time (use when bisecting a flake)")
ap.add_argument("-k", "--filter", action="append", default=[],
help="only tests whose path contains this (repeatable)")
ap.add_argument("-v", "--verbose", action="store_true",
help="print every test's output, not just failures")
ap.add_argument("--timeout", type=int, default=DEFAULT_TIMEOUT,
help=f"per-test timeout in seconds (default {DEFAULT_TIMEOUT})")
args = ap.parse_args()
tests = discover()
if args.filter:
tests = [(t, c) for (t, c) in tests
if any(f.lower() in display_path(t, c).lower() for f in args.filter)]
if not tests:
print("No tests found.")
return 1
workers = 1 if args.serial else (args.jobs or min(16, os.cpu_count() or 4))
exclusive = [(t, c) for (t, c) in tests if t.name in EXCLUSIVE]
shared = [(t, c) for (t, c) in tests if t.name not in EXCLUSIVE]
print(f"Running {len(tests)} test file(s) "
f"with {workers} worker(s), {args.timeout}s timeout")
if exclusive and not args.serial:
names = ", ".join(t.name for t, _ in exclusive)
print(f" ({names} run first, alone — they write shared scratch dirs)")
print()
started = time.monotonic()
results: list[Result] = []
# Exclusive first, alone.
for test, cwd in exclusive:
r = run_one(test, cwd, args.timeout)
results.append(r)
report(r, args.verbose)
# Then the pool. Results are collected and printed in DISCOVERY order,
# not completion order — a parallel run whose output reshuffles every
# time is unreadable, and undiffable against a previous run.
if shared:
with ThreadPoolExecutor(max_workers=workers) as pool:
futures = [pool.submit(run_one, t, c, args.timeout) for t, c in shared]
for f in futures:
r = f.result()
results.append(r)
report(r, args.verbose)
elapsed = time.monotonic() - started
passed = sum(1 for r in results if r.status == "pass")
skipped = sum(1 for r in results if r.status == "skip")
failed = [r for r in results if not r.ok]
print()
print("-" * 43)
if EXTRAS_ABSENT:
label = "SKIP(needs TOS-Extras)"
elif REFERENCE_ABSENT and skipped:
label = "SKIP(needs Reference)"
else:
label = "SKIP(needs-TOS)"
print(f"PASS={passed} FAIL={len(failed)} {label}={skipped}"
f" in {elapsed:.1f}s")
if failed:
print("Failed/unclear:")
for r in failed:
print(f" {r.path} ({r.status})")
# Say WHY, once, rather than leaving a contributor to guess what a
# skip count means on a fresh clone.
if EXTRAS_ABSENT and skipped:
print()
print(f"note: the sibling TOS-Extras/ tree is not present, so {skipped} test(s)")
print(" that drive add-on packages were skipped rather than failed.")
print(" That tree is not published yet; see CONTRIBUTING.md. Everything")
print(" testable without it ran.")
elif REFERENCE_ABSENT and skipped:
print()
print("note: Reference/oc-component-api/ is not present (it is vendored in the")
print(f" maintainer's monorepo, not on the dev branch), so {skipped} test(s)")
print(" that check component calls against it were skipped, not failed.")
# Slowest few — worth knowing which tests dominate the wall clock.
if args.verbose or not failed:
slowest = sorted(results, key=lambda r: r.seconds, reverse=True)[:3]
if slowest and slowest[0].seconds >= 0.5:
times = ", ".join(f"{r.path.split('/')[-1]} {r.seconds:.1f}s"
for r in slowest)
print(f"Slowest: {times}")
return 1 if failed else 0
if __name__ == "__main__":
sys.exit(main())