mirror of
https://github.com/rcourtman/Pulse.git
synced 2026-10-03 12:47:49 +00:00
The weights from #2324 came from one local run and left the api shards at 4.5, 8.8, 17.1 and 12.5 minutes on CI. Most of the 17.1 was not test time. That shard ran `-run . -skip <1231 names>`, and the test binary caches only its last compiled pattern, so alternating the two recompiled the 62 KB skip regex for every test and subtest (about 50 ms each under -race locally, 30s against 1.3s for 300 trivial tests). Two runs put that shard at 887s and 875s against roughly 450s of tests. The skip path now passes -skip alone. Every api shard now runs go test -json through .github/scripts/record-internal-api-test-seconds.py. It prints what plain go test would (package lines and the output of failing or unfinished tests), exits non-zero on any failure behind pipefail, and writes each top-level test's seconds to an internal-api-test-seconds-N artifact that is uploaded even when the shard fails. .github/scripts/refresh-internal-api-test-seconds.py rebuilds the weights file from those artifacts (gh run download, median across runs, DEFAULT_WEIGHT set to the mean of the unlisted tests, which the selector now reads from the file). Until those artifacts exist, the weights are a local -race -json run with each region of go test's order scaled to what CI reported for it. Equal-count quarters averaged 107, 437, 100 and 787s over four runs, and the weighted cut gave 131s for the first 1187 tests, 350s for the next 23 and 608s for the last 21 over two. The last quarter and the last 21 tests disagree on how the tail splits, so the tail takes the midpoint of an exact fit and a pooled factor. With these weights four shards still predict a slowest shard near 10 minutes, and a 20% slower runner on the tail would take it past the 11 minutes of Frontend and rest-1. Five shards predict about 131, 325, 280-340, 330-355 and 330-365s of tests, or 4.5 to 8.5 minutes a job with about 2.3 minutes of setup, so API_SHARD_COUNT goes to 5. Every test still runs exactly once in contiguous go test order, and all five slices passed locally from their new starting points. The required Backend tests (api) verdict keeps its name.
104 lines
3.5 KiB
Python
Executable file
104 lines
3.5 KiB
Python
Executable file
#!/usr/bin/env python3
|
|
"""Print `go test -json` like plain `go test` and record per-test seconds.
|
|
|
|
Usage: go test -json ... | record-internal-api-test-seconds.py <seconds-file>
|
|
|
|
`go test -json` reports every test the way `-v` does, which for internal/api
|
|
means thousands of lines per shard. This filter keeps the log as readable as
|
|
plain `go test`: package lines (build errors, `ok`, `FAIL`, panics outside a
|
|
test) print straight away, and a test's own output is held until the test
|
|
ends, printed if it failed and dropped if it passed or was skipped. Output of
|
|
a test that never finished (a panic or timeout killed the binary) is printed
|
|
at the end, so a crash is never hidden.
|
|
|
|
Every finished top-level test is written to <seconds-file> as
|
|
`<TestName> <elapsed seconds>`, after a `# package-seconds <elapsed>` comment
|
|
holding the whole test binary's run time, the same shape as
|
|
.github/scripts/internal-api-test-seconds.txt, so
|
|
refresh-internal-api-test-seconds.py can rebuild the shard weights from CI.
|
|
|
|
The exit status is non-zero when any test or package failed, but the caller
|
|
must still run with `set -o pipefail` so a `go test` failure that produced no
|
|
JSON (for example a vet or build error) fails the step.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import json
|
|
import sys
|
|
from collections import defaultdict
|
|
|
|
|
|
def main() -> int:
|
|
if len(sys.argv) != 2:
|
|
print(f"usage: {sys.argv[0]} <seconds-file>", file=sys.stderr)
|
|
return 2
|
|
|
|
held: dict[str, list[str]] = defaultdict(list)
|
|
running: set[str] = set()
|
|
seconds: dict[str, float] = {}
|
|
package_seconds = None
|
|
failed = False
|
|
out = sys.stdout
|
|
|
|
for raw in sys.stdin:
|
|
try:
|
|
event = json.loads(raw)
|
|
except ValueError:
|
|
event = None
|
|
if not isinstance(event, dict):
|
|
out.write(raw)
|
|
out.flush()
|
|
continue
|
|
|
|
action = event.get("Action")
|
|
test = event.get("Test") or ""
|
|
top = test.split("/", 1)[0]
|
|
text = event.get("Output") or ""
|
|
|
|
if not top:
|
|
# Package level: build output, `ok`/`FAIL` summary, panics and
|
|
# timeouts reported outside a test.
|
|
if text:
|
|
out.write(text)
|
|
out.flush()
|
|
if action == "fail":
|
|
failed = True
|
|
if action in ("pass", "fail") and isinstance(event.get("Elapsed"), (int, float)):
|
|
package_seconds = float(event["Elapsed"])
|
|
continue
|
|
|
|
if action == "run" and test == top:
|
|
running.add(top)
|
|
if text:
|
|
held[top].append(text)
|
|
if test != top or action not in ("pass", "fail", "skip"):
|
|
continue
|
|
|
|
running.discard(top)
|
|
lines = held.pop(top, [])
|
|
if action == "fail":
|
|
failed = True
|
|
out.write("".join(lines))
|
|
out.flush()
|
|
elapsed = event.get("Elapsed")
|
|
if isinstance(elapsed, (int, float)):
|
|
seconds[top] = float(elapsed)
|
|
|
|
for top in sorted(running):
|
|
failed = True
|
|
out.write(f"--- {top} did not finish; its output follows\n")
|
|
out.write("".join(held.get(top, [])))
|
|
out.flush()
|
|
|
|
with open(sys.argv[1], "w", encoding="utf-8") as handle:
|
|
if package_seconds is not None:
|
|
handle.write(f"# package-seconds {package_seconds:.2f}\n")
|
|
for name, value in seconds.items():
|
|
handle.write(f"{name} {value:.2f}\n")
|
|
|
|
return 1 if failed else 0
|
|
|
|
|
|
if __name__ == "__main__":
|
|
raise SystemExit(main())
|