mirror of
https://github.com/rcourtman/Pulse.git
synced 2026-10-04 13:52:24 +00:00
test(slo): only enforce latency budgets on hosted runners
The shared release preflight worker exports GITHUB_ACTIONS=true so isolated single-repository checkouts take their private-sibling skip. That flag also made the load/SLO latency guards hard-fail on a contended shared host, where the tests' own documented local-contention skip is the intended behaviour. Distinguish a real GitHub Actions run by GITHUB_RUN_ID, which only a hosted runner sets, so the worker keeps the cross-repo skip while latency overruns skip instead of failing. Hosted CI enforcement is unchanged. Change-source: pulse-maintainer
This commit is contained in:
parent
50f744fe6b
commit
59d3da9768
5 changed files with 80 additions and 13 deletions
|
|
@ -442,8 +442,43 @@ func buildLargeDeploymentState(t *testing.T, numNodes int) *models.State {
|
|||
return state
|
||||
}
|
||||
|
||||
// latencyBudgetEnforced reports whether a load/SLO overrun should fail the run.
|
||||
// Only a controlled GitHub-hosted runner qualifies. The shared release
|
||||
// preflight worker exports GITHUB_ACTIONS=true for its isolated
|
||||
// single-repository checkout (scripts/release-preflight-worker.sh), but it is a
|
||||
// contended shared host where an overrun cannot be attributed to a regression,
|
||||
// so the documented local-contention skip applies there. GITHUB_RUN_ID is set
|
||||
// only by a real GitHub Actions run, never by the worker.
|
||||
func latencyBudgetEnforced() bool {
|
||||
return os.Getenv("GITHUB_ACTIONS") == "true" && os.Getenv("GITHUB_RUN_ID") != ""
|
||||
}
|
||||
|
||||
func TestLatencyBudgetEnforcedOnlyOnHostedRunner(t *testing.T) {
|
||||
t.Run("shared preflight worker skips", func(t *testing.T) {
|
||||
t.Setenv("GITHUB_ACTIONS", "true")
|
||||
t.Setenv("GITHUB_RUN_ID", "")
|
||||
if latencyBudgetEnforced() {
|
||||
t.Fatal("a shared preflight worker must use the local-contention skip")
|
||||
}
|
||||
})
|
||||
t.Run("hosted runner enforces", func(t *testing.T) {
|
||||
t.Setenv("GITHUB_ACTIONS", "true")
|
||||
t.Setenv("GITHUB_RUN_ID", "123456")
|
||||
if !latencyBudgetEnforced() {
|
||||
t.Fatal("a GitHub-hosted run must enforce the latency budget")
|
||||
}
|
||||
})
|
||||
t.Run("local run skips", func(t *testing.T) {
|
||||
t.Setenv("GITHUB_ACTIONS", "")
|
||||
t.Setenv("GITHUB_RUN_ID", "")
|
||||
if latencyBudgetEnforced() {
|
||||
t.Fatal("a local run must use the local-contention skip")
|
||||
}
|
||||
})
|
||||
}
|
||||
|
||||
func effectiveLoadMinCount(localMinCount, githubActionsMinCount int64) int64 {
|
||||
if githubActionsMinCount > 0 && os.Getenv("GITHUB_ACTIONS") == "true" {
|
||||
if githubActionsMinCount > 0 && latencyBudgetEnforced() {
|
||||
return githubActionsMinCount
|
||||
}
|
||||
return localMinCount
|
||||
|
|
@ -459,7 +494,7 @@ func effectiveLoadMinCount(localMinCount, githubActionsMinCount int64) int64 {
|
|||
// the call sites.
|
||||
func failOrSkipLoadOverrun(t *testing.T, format string, args ...interface{}) {
|
||||
t.Helper()
|
||||
if os.Getenv("GITHUB_ACTIONS") == "true" {
|
||||
if latencyBudgetEnforced() {
|
||||
t.Errorf(format, args...)
|
||||
return
|
||||
}
|
||||
|
|
@ -467,7 +502,7 @@ func failOrSkipLoadOverrun(t *testing.T, format string, args ...interface{}) {
|
|||
}
|
||||
|
||||
func effectiveLoadP95Budget(endpoint string, localTarget time.Duration) time.Duration {
|
||||
if os.Getenv("GITHUB_ACTIONS") != "true" {
|
||||
if !latencyBudgetEnforced() {
|
||||
return localTarget
|
||||
}
|
||||
switch endpoint {
|
||||
|
|
|
|||
|
|
@ -92,14 +92,14 @@ func TestLoad_500Node_ConcurrentResources(t *testing.T) {
|
|||
t.Logf("p50=%v p95=%v p99=%v", p50, p95, p99)
|
||||
|
||||
target := 3 * time.Second
|
||||
if os.Getenv("GITHUB_ACTIONS") == "true" {
|
||||
if latencyBudgetEnforced() {
|
||||
target = 4 * time.Second
|
||||
}
|
||||
if p95 > target {
|
||||
resourceLoadOverrun(t, "p95 latency %v exceeds %v budget for 500-node concurrent resources load", p95, target)
|
||||
}
|
||||
minimum := int64(100)
|
||||
if os.Getenv("GITHUB_ACTIONS") == "true" {
|
||||
if latencyBudgetEnforced() {
|
||||
minimum = 40
|
||||
}
|
||||
if totalCount < minimum {
|
||||
|
|
@ -151,9 +151,20 @@ func buildResourceLoadState(t *testing.T, numNodes int) *models.State {
|
|||
return state
|
||||
}
|
||||
|
||||
// latencyBudgetEnforced reports whether a load overrun should fail the run.
|
||||
// Only a controlled GitHub-hosted runner qualifies. The shared release
|
||||
// preflight worker exports GITHUB_ACTIONS=true for its isolated
|
||||
// single-repository checkout (scripts/release-preflight-worker.sh), but it is a
|
||||
// contended shared host where an overrun cannot be attributed to a regression,
|
||||
// so the documented local-contention skip applies there. GITHUB_RUN_ID is set
|
||||
// only by a real GitHub Actions run, never by the worker.
|
||||
func latencyBudgetEnforced() bool {
|
||||
return os.Getenv("GITHUB_ACTIONS") == "true" && os.Getenv("GITHUB_RUN_ID") != ""
|
||||
}
|
||||
|
||||
func resourceLoadOverrun(t *testing.T, format string, args ...interface{}) {
|
||||
t.Helper()
|
||||
if os.Getenv("GITHUB_ACTIONS") == "true" {
|
||||
if latencyBudgetEnforced() {
|
||||
t.Errorf(format, args...)
|
||||
return
|
||||
}
|
||||
|
|
|
|||
|
|
@ -5,7 +5,6 @@ import (
|
|||
"fmt"
|
||||
"net/http"
|
||||
"net/http/httptest"
|
||||
"os"
|
||||
"path/filepath"
|
||||
"reflect"
|
||||
"sort"
|
||||
|
|
@ -853,7 +852,7 @@ const (
|
|||
)
|
||||
|
||||
func effectiveAPISLOTarget(localTarget, githubActionsTarget time.Duration) time.Duration {
|
||||
if githubActionsTarget > 0 && os.Getenv("GITHUB_ACTIONS") == "true" {
|
||||
if githubActionsTarget > 0 && latencyBudgetEnforced() {
|
||||
return githubActionsTarget
|
||||
}
|
||||
return localTarget
|
||||
|
|
@ -936,7 +935,7 @@ func assertLatencySLO(t *testing.T, label string, latencies []time.Duration, tar
|
|||
if p95 <= target {
|
||||
return
|
||||
}
|
||||
if os.Getenv("GITHUB_ACTIONS") == "true" {
|
||||
if latencyBudgetEnforced() {
|
||||
t.Errorf("SLO VIOLATION: p95=%v exceeds target %v", p95, target)
|
||||
return
|
||||
}
|
||||
|
|
|
|||
|
|
@ -217,15 +217,26 @@ func assertLatencySLO(t *testing.T, label string, latencies []time.Duration, tar
|
|||
if p95 <= target {
|
||||
return
|
||||
}
|
||||
if os.Getenv("GITHUB_ACTIONS") == "true" {
|
||||
if latencyBudgetEnforced() {
|
||||
t.Errorf("SLO VIOLATION: p95=%v exceeds target %v", p95, target)
|
||||
return
|
||||
}
|
||||
t.Skipf("p95=%v exceeds target %v (median=%v): host CPU contention from parallel builds inflates wall-clock latency, so this overrun cannot be attributed to a regression; re-run on a quiet machine for a strict check (CI enforces the budget unconditionally)", p95, target, p50)
|
||||
}
|
||||
|
||||
// latencyBudgetEnforced reports whether an SLO overrun should fail the run.
|
||||
// Only a controlled GitHub-hosted runner qualifies. The shared release
|
||||
// preflight worker exports GITHUB_ACTIONS=true for its isolated
|
||||
// single-repository checkout (scripts/release-preflight-worker.sh), but it is a
|
||||
// contended shared host where an overrun cannot be attributed to a regression,
|
||||
// so the documented local-contention skip applies there. GITHUB_RUN_ID is set
|
||||
// only by a real GitHub Actions run, never by the worker.
|
||||
func latencyBudgetEnforced() bool {
|
||||
return os.Getenv("GITHUB_ACTIONS") == "true" && os.Getenv("GITHUB_RUN_ID") != ""
|
||||
}
|
||||
|
||||
func effectiveMonitoringSLOTarget(localTarget, githubActionsTarget time.Duration) time.Duration {
|
||||
if githubActionsTarget > 0 && os.Getenv("GITHUB_ACTIONS") == "true" {
|
||||
if githubActionsTarget > 0 && latencyBudgetEnforced() {
|
||||
return githubActionsTarget
|
||||
}
|
||||
return localTarget
|
||||
|
|
|
|||
|
|
@ -134,8 +134,19 @@ const (
|
|||
|
||||
const sloIterations = 200
|
||||
|
||||
// latencyBudgetEnforced reports whether an SLO overrun should fail the run.
|
||||
// Only a controlled GitHub-hosted runner qualifies. The shared release
|
||||
// preflight worker exports GITHUB_ACTIONS=true for its isolated
|
||||
// single-repository checkout (scripts/release-preflight-worker.sh), but it is a
|
||||
// contended shared host where an overrun cannot be attributed to a regression,
|
||||
// so the documented local-contention skip applies there. GITHUB_RUN_ID is set
|
||||
// only by a real GitHub Actions run, never by the worker.
|
||||
func latencyBudgetEnforced() bool {
|
||||
return os.Getenv("GITHUB_ACTIONS") == "true" && os.Getenv("GITHUB_RUN_ID") != ""
|
||||
}
|
||||
|
||||
func effectiveSLOTarget(localTarget time.Duration, githubActionsTarget time.Duration) time.Duration {
|
||||
if os.Getenv("GITHUB_ACTIONS") == "true" {
|
||||
if latencyBudgetEnforced() {
|
||||
return githubActionsTarget
|
||||
}
|
||||
return localTarget
|
||||
|
|
@ -222,7 +233,7 @@ func assertLatencySLO(t *testing.T, label string, latencies []time.Duration, tar
|
|||
if p95 <= target {
|
||||
return
|
||||
}
|
||||
if os.Getenv("GITHUB_ACTIONS") == "true" {
|
||||
if latencyBudgetEnforced() {
|
||||
t.Errorf("SLO VIOLATION: p95=%v exceeds target %v", p95, target)
|
||||
return
|
||||
}
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue