diff --git a/.github/scripts/run-autofix-review-verification.sh b/.github/scripts/run-autofix-review-verification.sh index 8f3c3c5bef..4ca5790d58 100755 --- a/.github/scripts/run-autofix-review-verification.sh +++ b/.github/scripts/run-autofix-review-verification.sh @@ -119,7 +119,7 @@ reject_fix() { if [[ "${preexisting}" == 'true' ]]; then # NOT retryable: the repair agent is only allowed to amend this round's # fix, and a failure that exists without the fix is outside that boundary - # by definition — the 18-minute repair budget cannot reach it. The remedy + # by definition — the 45-minute repair budget cannot reach it. The remedy # is a base update (merge main into the branch), not a repair. echo "preexisting=true" >> "${GITHUB_OUTPUT}" elif [[ "${retryable}" == 'true' ]]; then @@ -1047,7 +1047,7 @@ fi # a DEFECT-CLAIM round only when resolved-comments.txt marks a finding # resolved-in-code whose thread is Critical-tagged or belongs to a # CHANGES_REQUESTED review (matched in rc.json/rv.json). Those rounds get a -# non-retryable rejection on all-green — the 18-minute repair pass cannot +# non-retryable rejection on all-green — the 45-minute repair pass cannot # make a nonexistent defect reproduce; the next full round re-reads the # feedback with the evidence in LAST_REJECTION and can decline or escalate # instead. Every OTHER src+test round (a refactor pinning existing diff --git a/.github/workflows/qwen-autofix.yml b/.github/workflows/qwen-autofix.yml index 784400e61b..9073e3697f 100644 --- a/.github/workflows/qwen-autofix.yml +++ b/.github/workflows/qwen-autofix.yml @@ -2057,7 +2057,7 @@ jobs: # Keep the predicate as narrow as the job's own `if:` — concurrency is # evaluated BEFORE it, so without the do_review conjunct a dispatch with # `phase: issue` + `pr_number: N` (route emits pr_number unconditionally) - # would park this skipped job in that PR's shared slot behind a 300-minute + # would park this skipped job in that PR's shared slot behind a 330-minute # address round, stalling the issue phase that `needs` it. concurrency: group: "qwen-pr-head-write-${{ needs.route.outputs.do_review == 'true' && needs.route.outputs.pr_number || github.run_id }}" @@ -2362,11 +2362,11 @@ jobs: # Pending-check staleness bound (invariant across candidate PRs, computed # once): ignore a check stuck far past any legitimate runtime. The bound # must sit ABOVE real check durations here — review-pr can take ~50m and - # a review-address JOB runs up to its 300-minute cap — so an active run + # a review-address JOB runs up to its 330-minute cap — so an active run # keeps blocking and is never aged out mid-flight (which would enqueue - # the PR against a live check and double-process the feedback). 330 holds + # the PR against a live check and double-process the feedback). 360 holds # a 30-minute margin over that cap. - PENDING_STALE_MIN=330 + PENDING_STALE_MIN=360 PENDING_CUTOFF="$(date -u -d "${PENDING_STALE_MIN} minutes ago" +%Y-%m-%dT%H:%M:%SZ)" # Repetition-guard cutoff for the stale-base update marker (invariant @@ -3445,7 +3445,7 @@ jobs: # write+ (internal) authors at scan AND address time. That is an # Full rationale → qwen-autofix.md#af-038 runs-on: '${{ (github.repository == ''QwenLM/qwen-code'' && vars.MAINTAINER_ECS_RUNNER_DISABLED != ''true'' && (github.event_name != ''pull_request'' && github.event_name != ''pull_request_review'' || github.event.pull_request.head.repo.full_name == github.repository || contains(fromJSON(''["OWNER","MEMBER","COLLABORATOR"]''), github.event.pull_request.author_association))) && fromJSON(''["self-hosted", "linux", "x64", "ecs-qwen"]'') || fromJSON(''["ubuntu-latest"]'') }}' - timeout-minutes: 300 + timeout-minutes: 330 permissions: contents: 'read' strategy: @@ -3702,7 +3702,7 @@ jobs: ;; esac # The label routing pins ecs-qwen, but a mis-labelled registration - # must not silently claim a PAT-bearing 300-minute job — assert the + # must not silently claim a PAT-bearing 330-minute job — assert the # pool by name on the self-hosted branch too. if [[ "${RUNNER_ENVIRONMENT}" == 'self-hosted' ]]; then case "${RUNNER_NAME}" in @@ -4946,7 +4946,10 @@ jobs: # the CLI without any sandbox (#9527 review). if: |- ${{ always() && steps.verify.outputs.retryable == 'true' && steps.sandbox_image.outcome == 'success' }} - timeout-minutes: 20 + # 55m: the 45-minute budget below plus the same 10-minute margin the + # primary attempt keeps under its own backstop, so the internal kill + # path still writes `agent-timeout` before the step cap fires. + timeout-minutes: 55 env: QWEN_SANDBOX_IMAGE: '${{ steps.sandbox_image.outputs.image }}' DOCKER_HOST: '' @@ -4958,7 +4961,15 @@ jobs: OPENAI_MODEL: '${{ vars.QWEN_AUTOFIX_MODEL || vars.QWEN_PR_REVIEW_MODEL }}' NO_PROXY: '127.0.0.1,localhost,::1' QWEN_HOME: '${{ runner.temp }}/qwen-autofix-review-home' - QWEN_TIMEOUT_MS: '1080000' + # 45m. The repair attempt starts from a deterministic rejection — + # an opaque check failure, not the structured review feedback the + # primary attempt is handed — so it must re-derive which change + # caused it before it can amend anything. At the previous 18m it + # ran out mid-diagnosis on every observed rejection while the + # primary attempt (120m) had already succeeded, and the round's + # work was discarded. Still a fraction of the primary budget: a + # repair that cannot land in 45m is a handoff, not a longer retry. + QWEN_TIMEOUT_MS: '2700000' CONFLICT: '${{ steps.prepare.outputs.conflict }}' BASE: 'main' SETTINGS_JSON: |- diff --git a/scripts/tests/qwen-autofix-workflow.test.js b/scripts/tests/qwen-autofix-workflow.test.js index 64bc9cf90d..15ceb42e05 100644 --- a/scripts/tests/qwen-autofix-workflow.test.js +++ b/scripts/tests/qwen-autofix-workflow.test.js @@ -597,9 +597,9 @@ describe('qwen-autofix workflow', () => { expect(reviewScanJob).toContain('echo "targets=[]" >> "${GITHUB_OUTPUT}"'); expect(reviewScanJob).toContain('active checks in flight; skipping until'); // Staleness bound must sit above legitimate check runtimes (a review-address - // job runs up to its 300-minute cap) so an active run is never aged out + // job runs up to its 330-minute cap) so an active run is never aged out // mid-flight. - expect(reviewScanJob).toContain('PENDING_STALE_MIN=330'); + expect(reviewScanJob).toContain('PENDING_STALE_MIN=360'); // The staleness filter itself, including the comparison operator: a check only // blocks if its start is newer than the cutoff. Asserting `> $cut` too means a // flipped comparison (which would age out live checks → double-processing) is @@ -4178,7 +4178,7 @@ describe('qwen-autofix workflow', () => { // No rollup entries → dispatchable. expect(runMarkerCheck([])).toBe('pass'); - // A stranded marker must NOT keep blocking through the 330-minute + // A stranded marker must NOT keep blocking through the 360-minute // HAS_PENDING_CHECKS gate after its TTL expired: replay the gate's jq // over fixture rollups. const pendingGate = reviewScanJob.match( @@ -4228,7 +4228,7 @@ describe('qwen-autofix workflow', () => { checkRun('build', 'IN_PROGRESS', '2026-08-17T07:50:00Z'), ]), ).toBe('true'); - // ...a check stuck past the 330-minute horizon is aged out... + // ...a check stuck past the 360-minute horizon is aged out... expect( runPendingGate([ checkRun('build', 'IN_PROGRESS', '2026-08-17T01:00:00Z'), @@ -14672,9 +14672,9 @@ exit 1 expect(repairDeterministicRejectionStep).toContain( "steps.verify.outputs.retryable == 'true'", ); - expect(repairDeterministicRejectionStep).toContain('timeout-minutes: 20'); + expect(repairDeterministicRejectionStep).toContain('timeout-minutes: 55'); expect(repairDeterministicRejectionStep).toContain( - "QWEN_TIMEOUT_MS: '1080000'", + "QWEN_TIMEOUT_MS: '2700000'", ); const settingsJson = (step) => step.match(/SETTINGS_JSON: \|-\n([\s\S]*?)\n {8}run: \|-/)?.[1] ?? '';