fix(ci): keep package builds out of the macOS app test budget (#153911)

* ci: run Swift package tests alongside the macOS app

* ci: account for both hosted Swift test phases

* test(ci): verify cache ownership across Swift phases
This commit is contained in:
Peter Steinberger 2026-09-20 11:55:08 -07:00 • committed by GitHub
parent f429ed76ef
commit 2bf6257853
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
7 changed files with 74 additions and 26 deletions

View file

@ -232,12 +232,15 @@ These are intentionally guarded by `test/scripts/ci-workflow-guards.test.ts`:
backend, after runtime preparation completes. Native allocation can be smaller
than the runner label. Native proof must cover available CPUs/RAM, fixture
memory and cleanup. This adds no runner registrations.
- macOS Swift regular PR/main and PR `release_gate` CI retains the complete
shared/app test workload plus lint/schema guards in one `tests` phase.
- macOS Swift regular PR/main and PR `release_gate` CI runs complete app tests
plus lint/schema guards in `tests`, alongside independent OpenClawKit trait,
OpenClawKit test, and Swabble test graphs in `packages`.
Ordinary full-scope manual validation adds independent release compilation,
moves the guards to `release`, and retains health renders in `tests`.
Both phases use GitHub-hosted `macos-26`, `max-parallel: 2`, and the existing
30-minute budget. Build caches stay phase-owned; the sole eligible shared
All phases use GitHub-hosted `macos-26`, `max-parallel: 2`, and the existing
30-minute budget. This adds one hosted job and no Blacksmith registrations;
measure complete hosted timing including duplicated setup. Packages do not
restore or save app build products. Build caches stay phase-owned; the sole eligible shared
SwiftPM cache writer is regular `tests` or full-validation `release`.
- Android regular CI uses four test/lint rows, including benchmark compilation
in the Kotlin-lint row when benchmark/build/dependency inputs change or the
@ -251,8 +254,8 @@ These are intentionally guarded by `test/scripts/ci-workflow-guards.test.ts`:
npm qualification still defers native jobs. All iOS build phases and screenshot
shards use `macos-26` from the first attempt.
The conservative full-tier non-Node inventory, including Control UI performance, is
86 rows, or 87 for historical UI targets. Excluding those four hosted rows
plus both macOS Swift phases and the always-hosted aggregate gate leaves at
87 rows, or 88 for historical UI targets. Excluding those four hosted rows
plus all three macOS Swift phases and the always-hosted aggregate gate leaves at
most 80 potentially eligible jobs. The enforced Node caps therefore give
150 registrations per main run and 210 per PR:
`4 × 150 + 21 × 210 = 5,010` in the retained peak arrival envelope.

View file

@ -2023,7 +2023,7 @@ jobs:
"check-docs": count(manifest.run_check_docs),
"skills-python": count(manifest.run_skills_python_job),
"macos-node": count(manifest.run_macos_node, manifest.macos_node_matrix.include.length),
"macos-swift": count(manifest.run_macos_swift),
"macos-swift": count(manifest.run_macos_swift, 2),
"ios-build": count(manifest.run_ios_build),
"docker-seed-e2e": count(manifest.run_docker_seed_e2e && eventName === "pull_request" &&
process.env.OPENCLAW_CI_HEAD_REPOSITORY !== process.env.OPENCLAW_CI_REPOSITORY),
@ -4891,13 +4891,13 @@ jobs:
name: ${{ format('macos-swift ({0})', matrix.phase) }}
needs: [preflight]
if: needs.preflight.outputs.run_macos_swift == 'true'
# Debug/coverage tests remain required; full validation adds an independent
# optimized build whose products are not inputs to the test graph.
# Independent package builds must not consume the app test job's budget.
# Full validation also compiles the optimized app, within the same two-job cap.
strategy:
fail-fast: false
max-parallel: 2
matrix:
phase: ${{ fromJSON(github.event_name == 'workflow_dispatch' && !inputs.release_gate && needs.preflight.outputs.release_scope == 'full' && '["release","tests"]' || '["tests"]') }}
phase: ${{ fromJSON(github.event_name == 'workflow_dispatch' && !inputs.release_gate && needs.preflight.outputs.release_scope == 'full' && '["release","tests","packages"]' || '["tests","packages"]') }}
# Unassigned Blacksmith Mac jobs can hold both non-canceling main slots.
# Use the existing hosted image and budget from the first attempt.
runs-on: macos-26
@ -5018,6 +5018,7 @@ jobs:
- name: Detect Swift toolchain cache key
id: swift-toolchain
if: matrix.phase != 'packages'
run: |
set -euo pipefail
xcode_version="$(xcodebuild -version | tr '\n' ' ' | sed 's/ */ /g; s/ $//')"
@ -5037,7 +5038,7 @@ jobs:
- name: Restore Swift build directory cache
id: swift-build-cache
if: needs.preflight.outputs.cache_mode != 'off'
if: matrix.phase != 'packages' && needs.preflight.outputs.cache_mode != 'off'
uses: actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9 # v6.1.0
with:
path: apps/macos/.build
@ -5049,6 +5050,7 @@ jobs:
- name: Validate Swift build cache
id: validate-swift-build-cache
if: matrix.phase != 'packages'
run: |
set -euo pipefail
cache_valid=true
@ -5115,7 +5117,7 @@ jobs:
swift build --package-path apps/macos --product OpenClaw --configuration release
- name: OpenClawKit Talk-trait opt-out (no ElevenLabsKit when default traits disabled)
if: matrix.phase == 'tests'
if: matrix.phase == 'packages'
run: |
set -euo pipefail
# Guard: chat-only consumers build OpenClawKit with the Talk trait
@ -5132,7 +5134,7 @@ jobs:
# openclawkit-tests-contract-v1: the target owns an independently runnable package suite.
- name: OpenClawKit tests
if: matrix.phase == 'tests' && needs.preflight.outputs.run_openclawkit_tests == 'true'
if: matrix.phase == 'packages' && needs.preflight.outputs.run_openclawkit_tests == 'true'
run: |
set -euo pipefail
openclawkit_scratch="$(mktemp -d "$RUNNER_TEMP/openclawkit.XXXXXX")"
@ -5149,7 +5151,7 @@ jobs:
swift test "${openclawkit_test_args[@]}"
- name: Swabble tests
if: matrix.phase == 'tests' && env.HISTORICAL_TARGET != 'true'
if: matrix.phase == 'packages' && env.HISTORICAL_TARGET != 'true'
run: |
set -euo pipefail
swabble_scratch="$(mktemp -d "$RUNNER_TEMP/swabble.XXXXXX")"
@ -5239,7 +5241,7 @@ jobs:
# total five minutes, leaving five for runner startup, cancellation and cleanup.
- name: Check Swift cache save budget
id: swift-cache-budget
if: needs.preflight.outputs.cache_write_allowed == 'true'
if: matrix.phase != 'packages' && needs.preflight.outputs.cache_write_allowed == 'true'
continue-on-error: true
timeout-minutes: 1
env:

View file

@ -21,6 +21,8 @@ Core lint includes `src/**/*.test-support.cjs` in type-aware checks through the
Android native resource preparation uses the Mermaid renderer's filtered dependency install, including optional build tooling. Pnpm retains root dependencies but omits unrelated plugin packages; Gradle still builds the assets and runs the selected native tests and lint. Historical targets keep their compatibility path.
macOS Swift CI runs the app and independent package suites in separate [native phases](/ci/pipeline#macos-swift-phases), retaining every test and the existing concurrency and timeout limits.
Short hybrid jobs use a [40-row base threshold and 45-row hosted admission limit](/ci/capacity#bounded-hybrid-hosted-offload), with unchanged coverage and Blacksmith fallback when optional work does not fit.
Real-Gateway browser checks use [job budgets matched to their selected runner](/ci/runners#blacksmith-runner-capacity).

View file

@ -108,7 +108,7 @@ Failed-job-only hybrid retries retain their original matrix and its 360-second a
The final Node matrix admits longer estimated jobs first across compact and plugin descriptors. Plugin estimates reuse the extension batch cost owner, including existing process boundaries; runtime preparation is charged separately from the same prerequisite table used by compact jobs. Equal estimates and historical descriptors without estimates keep their original order. The 96-job concurrency ceiling bounds active jobs, while the manifest caps bound total admissions. In run `33449014227`, all 96 slots were occupied when the late QA job started; that dependency delay was matrix admission, not evidence of runner-registration throttling.
Expanded serial large/small jobs admit 210 predicted seconds; eligible hybrid parallel bins admit 360. All profiles retain the shared 90-row compact cap. The 150-second file-split and default exclusive-group budgets stay unchanged; complete non-build CLI bins alone may use the 250-second ordinary hybrid admission budget. The PR-only performance lifecycle file retains its 136-second fallback from native spans of 127.288/135.808 seconds in runs 33532741896/33545657559; canonical pushes omit that tooling family. Trusted contributor forks can use the GitHub profile on Blacksmith, so every profile participates in the same registration bound. The widest current workflow profiles retain up to 86 other potential rows (14 nonmatrix and 72 matrix), or 87 for historical targets without the UI named-project contract. The conservative cap-based envelope already includes twelve Control UI shards plus the browser-extension row on every profile. Excluding the four unconditionally hosted iOS rows, two hosted macOS Swift phases, and the hosted aggregate gate gives the conservative ceiling of 80 potentially eligible rows. This includes the new Control UI performance job; keep the ceiling rather than spending savings from consolidated checks. With the final Node caps, the bounds are 150 registrations per main run and 210 per PR. Two active main slots, both pending successors and the observed peak of 21 non-skipped PR arrivals give `4 × 150 + 21 × 210 = 5,010` registrations in five minutes. This leaves 990 within the 6,000 reference operating target for release work, adjacent repositories and carryover; it does not prove those arrivals fit. The earlier 19-arrival estimate is obsolete. Using the prior 4,826-registration reference, the bounded 2026-09-02 cohort audit counted 321 unassigned Blacksmith jobs and reserved nine auxiliary rows, giving `4,826 + 321 + 9 = 5,156` planned registrations and an 844-row allowance below that reference. Its 40 exact attempts covered 4,830 jobs; queued observations spanned 21:50:48–21:57:11 UTC and were not simultaneous. Already-assigned jobs, old approval-waiting runs, unobserved retries and unlisted organization work remain outside that cohort, so this is a conditional planning bound rather than a live organization balance. Evaluate a single PR trial using its actual emitted rows separately from the rollout model. Budget all six npm qualification jobs and the relevant full-release children; a shared-token quota response or unused bucket does not establish organization-wide usage or physical runner capacity.
Expanded serial large/small jobs admit 210 predicted seconds; eligible hybrid parallel bins admit 360. All profiles retain the shared 90-row compact cap. The 150-second file-split and default exclusive-group budgets stay unchanged; complete non-build CLI bins alone may use the 250-second ordinary hybrid admission budget. The PR-only performance lifecycle file retains its 136-second fallback from native spans of 127.288/135.808 seconds in runs 33532741896/33545657559; canonical pushes omit that tooling family. Trusted contributor forks can use the GitHub profile on Blacksmith, so every profile participates in the same registration bound. The widest current workflow profiles retain up to 87 other potential rows (14 nonmatrix and 73 matrix), or 88 for historical targets without the UI named-project contract. The conservative cap-based envelope already includes twelve Control UI shards plus the browser-extension row on every profile. Excluding the four unconditionally hosted iOS rows, three hosted macOS Swift phases, and the hosted aggregate gate gives the conservative ceiling of 80 potentially eligible rows. This includes the new Control UI performance job; keep the ceiling rather than spending savings from consolidated checks. With the final Node caps, the bounds are 150 registrations per main run and 210 per PR. Two active main slots, both pending successors and the observed peak of 21 non-skipped PR arrivals give `4 × 150 + 21 × 210 = 5,010` registrations in five minutes. This leaves 990 within the 6,000 reference operating target for release work, adjacent repositories and carryover; it does not prove those arrivals fit. The earlier 19-arrival estimate is obsolete. Using the prior 4,826-registration reference, the bounded 2026-09-02 cohort audit counted 321 unassigned Blacksmith jobs and reserved nine auxiliary rows, giving `4,826 + 321 + 9 = 5,156` planned registrations and an 844-row allowance below that reference. Its 40 exact attempts covered 4,830 jobs; queued observations spanned 21:50:48–21:57:11 UTC and were not simultaneous. Already-assigned jobs, old approval-waiting runs, unobserved retries and unlisted organization work remain outside that cohort, so this is a conditional planning bound rather than a live organization balance. Evaluate a single PR trial using its actual emitted rows separately from the rollout model. Budget all six npm qualification jobs and the relevant full-release children; a shared-token quota response or unused bucket does not establish organization-wide usage or physical runner capacity.
`checks-ui-e2e` emits thirteen rows for every newly planned target with the named-project contract: twelve combined weighted Control UI shards and one browser-extension row. This width applies across backend profiles, attempts, frozen targets, and missing attempt metadata. Control UI shards use the 16-class; the browser row uses the 8-class unless admitted to hosted Ubuntu by the bounded hybrid plan. Historical targets without the contract retain four total rows on the Blacksmith planner profile or fourteen on GitHub and hybrid profiles. The 2026-09-02 inventory at `49fb9c5` contains 359 files: 329 parallel bundle consumers, three parallel self-owned files, seven serial bundle consumers, and 20 serial private source/custom-build files. Ordinary CI excludes seven real-Gateway files, leaving 352. Four native projects represent resource ownership without adding jobs or execution phases: `ui-e2e-bundled` and `ui-e2e-standalone` share group 0 with at most two workers total, then `ui-e2e-serial` and `ui-e2e-serial-standalone` share group 1 with one worker. Local throttling and explicit worker limits still apply. The shared weighted sequencer charges each file by its measured duration divided by that project's effective worker count and assigns every discovered specification once across the selected Control UI rows. The root config keeps the complete inventory visible for discovery. Serial scheduling still protects private source servers that share a Vite optimizer cache, real Gateways, and the runtime-budget measurement; test cases, deadlines, and isolation are unchanged.

View file

@ -74,6 +74,26 @@ the job's uploaded artifacts.
| `openclaw-performance` | Separate workflow: daily/on-demand Kova runtime performance reports with mock-provider, deep-profile, and GPT 5.6 live lanes | Scheduled and manual dispatch |
| `docs-external-links` | Separate workflow: Docs External Link Audit checks external documentation links with lychee and uploads a report; it reports findings without failing, so it never blocks a pull request | Scheduled and manual dispatch |
### macOS Swift phases
`macos-swift (tests)` builds and runs the app's complete default- and named-profile
test partitions with coverage. `macos-swift (packages)` independently runs the
OpenClawKit Talk-trait opt-out build, OpenClawKit tests, and Swabble tests. These
separate package graphs previously ran before the app build in one job; a hosted
baseline spent 7m57s on them in a 21m48s job. Separating them gives app compilation
and tests their own 30-minute budget without removing coverage or increasing
test-process parallelism.
Both phases use `macos-26`, with at most two concurrent jobs. Full manual
validation adds the existing `release` phase under the same cap. This adds one
hosted Mac job and its checkout/setup cost per selected run, with no additional
Blacksmith registrations. Compare complete hosted timings, including queue and
setup time, before treating the removed serial work as an observed speedup.
Only the app phases restore the app build cache. SwiftPM dependency caches remain
restore-only in `packages`; the existing primary phase owns shared cache writes.
The aggregate gate requires every selected phase to succeed.
Ordinary Markdown and MDX pages under `docs/`, plus root `README.md`, retain
their separate `check-docs` coverage beside precise pull-request Node tests.
Page deletions and renames preserve this targeting. Explicit Node owners for

View file

@ -60,6 +60,7 @@ function runClockStep(owner: Step, now: number, started = "") {
}
type Context = {
cacheMode?: string;
allowed?: string;
authorized?: boolean;
phase?: string;
@ -78,7 +79,10 @@ function selected(owner: Step, context: Context = {}) {
const result = runInNewContext((owner.if ?? "true").replace(/\.([a-zA-Z_][\w-]*)/g, '["$1"]'), {
needs: {
preflight: {
outputs: { cache_write_allowed: context.authorized === false ? "false" : "true" },
outputs: {
cache_mode: context.cacheMode ?? "restore",
cache_write_allowed: context.authorized === false ? "false" : "true",
},
},
},
matrix: { phase: context.phase ?? "tests" },
@ -168,6 +172,10 @@ describe("macOS optional Swift cache lifetime", () => {
);
it("retains phase ownership and historical metadata compatibility", () => {
expect(selected(budget, { phase: "packages" })).toBe(false);
for (const owner of [packageSave, metadata, buildSave]) {
expect(selected(owner, { phase: "packages", allowed: "" }), owner.name).toBe(false);
}
expect(selected(packageSave, { primary: "release" })).toBe(false);
expect(selected(packageSave, { primary: "release", phase: "release" })).toBe(true);
expect(selected(buildSave, { primary: "release" })).toBe(true);
@ -193,7 +201,12 @@ describe("macOS optional Swift cache lifetime", () => {
["Restore Swift build directory cache", buildSave],
] as const) {
const restore = step(restoreName);
expect(restore.if).toBe("needs.preflight.outputs.cache_mode != 'off'");
for (const phase of ["tests", "release", "packages"]) {
expect(selected(restore, { phase }), `${restoreName}: ${phase}`).toBe(
phase !== "packages" || save === packageSave,
);
expect(selected(restore, { phase, cacheMode: "off" })).toBe(false);
}
expect(restore.uses).toBe("actions/cache/restore@55cc8345863c7cc4c66a329aec7e433d2d1c52a9");
expect(save.uses).toBe("actions/cache/save@55cc8345863c7cc4c66a329aec7e433d2d1c52a9");
expect(save.with?.path).toBe(restore.with?.path);

View file

@ -3176,7 +3176,7 @@ NODE
});
it.each([
["macos-swift", false, "workflow_dispatch", false, ["release", "tests"]],
["macos-swift", false, "workflow_dispatch", false, ["release", "tests", "packages"]],
["ios-build", false, "workflow_dispatch", false, ["release", "tests"]],
["ios-build", true, "workflow_dispatch", false, ["tests"]],
["ios-build", false, "pull_request", false, ["smoke"]],
@ -3220,11 +3220,11 @@ NODE
"Swift lint",
"Swift build (release)",
],
tests: [
tests: ["Swift test"],
packages: [
"OpenClawKit Talk-trait opt-out (no ElevenLabsKit when default traits disabled)",
"OpenClawKit tests",
"Swabble tests",
"Swift test",
],
}
: {
@ -3330,7 +3330,8 @@ NODE
const phases: string[] = Array.isArray(job.strategy.matrix.phase)
? job.strategy.matrix.phase
: evaluateWorkflowExpression(job.strategy.matrix.phase, context);
expect(phases).toEqual(full ? ["release", "tests"] : ["tests"]);
expect(phases).toEqual(full ? ["release", "tests", "packages"] : ["tests", "packages"]);
expect(job.strategy["max-parallel"]).toBe(2);
const env = Object.fromEntries(
Object.entries(job.env).map(([key, value]) => [
key,
@ -3373,12 +3374,19 @@ NODE
for (const name of [
"OpenClawKit Talk-trait opt-out (no ElevenLabsKit when default traits disabled)",
"OpenClawKit tests",
"Swift test",
]) {
expect(selectedPhases(name), name).toEqual(["tests"]);
expect(selectedPhases(name), name).toEqual(["packages"]);
}
expect(selectedPhases("Swabble tests")).toEqual(historical ? [] : ["tests"]);
expect(selectedPhases("Swift test")).toEqual(["tests"]);
expect(selectedPhases("Swabble tests")).toEqual(historical ? [] : ["packages"]);
expect(selectedPhases("Swift build (release)")).toEqual(full ? ["release"] : []);
for (const name of [
"Detect Swift toolchain cache key",
"Restore Swift build directory cache",
"Validate Swift build cache",
]) {
expect(selectedPhases(name), name).toEqual(full ? ["release", "tests"] : ["tests"]);
}
expect(selectedPhases("Render isolated macOS health fixtures")).toEqual(
full ? ["tests"] : [],
);
@ -14567,7 +14575,7 @@ printf '%s\n' "\${CURL_SUCCESS_IP:-203.0.113.7}"
expect(swiftLint.run).toContain("swiftlint lint --config config/swiftlint.yml");
expect(swiftLint.run).toContain('elif [[ "$HISTORICAL_TARGET" == "true" ]]');
expect(openClawKitTests.if).toBe(
"matrix.phase == 'tests' && needs.preflight.outputs.run_openclawkit_tests == 'true'",
"matrix.phase == 'packages' && needs.preflight.outputs.run_openclawkit_tests == 'true'",
);
const checkShard = workflow.jobs["check-shard"].steps.find(