diff --git a/pytest.ini b/pytest.ini index 2ede9fb3e..9814b3365 100644 --- a/pytest.ini +++ b/pytest.ini @@ -10,3 +10,4 @@ addopts = --capture=no norecursedirs = tests/manual tests/evals scripts/cdp-download-poc tests/sdk eval infra/gcp-agent prompt_evaluation internal-tools markers = workday_offline: opt-in offline Workday deterministic regression suite (needs a Playwright chromium; run with -m workday_offline) + boundary: reads skyvern-frontend/ or docs/ from the repo, so it must run on FE/docs-only PRs that skip the Python suite diff --git a/skyvern/forge/taskv3/loop.py b/skyvern/forge/taskv3/loop.py index 70156ee71..55f8cccf3 100644 --- a/skyvern/forge/taskv3/loop.py +++ b/skyvern/forge/taskv3/loop.py @@ -203,7 +203,20 @@ PERCEPTION_STALL_REASON_PREFIX = "perception_stall:" # returning different content, or a download landing; a first-time probe has no baseline and is # evidence of nothing, so varied-selector probing cannot launder repetition into "progress". ACTION_LOOP_NUDGE_AFTER = 3 -ACTION_LOOP_TERMINATE_AFTER = 6 +# 8, not 6: RAW repeats of one action key peak at 6 across 50 completed prod runs while stuck runs +# reach 36, so 6 sat on the completed population's edge. DO NOT LOWER THIS. The effective +# post-clearing counter — what this constant is actually compared against — was then measured by +# replaying the clearing rule over per-call telemetry, and NO SEPARATION WAS OBSERVED: two completed +# runs peaked at 2 and 4, four stuck runs at 1, 3, 3 and below (upper bounds, computed identically +# on both sides), and the highest observed value fell in a COMPLETED run. n=6 is far too small to +# establish inversion as a property of either population — that spread is also consistent with +# noise — but it is no evidence FOR a lower threshold either, and lowering a safety threshold +# requires positive evidence. 8 is above everything observed, which is why the change is safe; it +# is also why nothing in the sample fires. REVISIT if the effective distribution ever becomes +# measurable at population scale; do not lower it before then. Repeat count may not be a stuck-ness +# signal at all: SKY-15602 tracks the real problem, which is that the loop has no definition of +# goal progress, only of page change. +ACTION_LOOP_TERMINATE_AFTER = 8 # Facetable sibling of PERCEPTION_STALL_REASON_PREFIX; same dashboard contract. ACTION_LOOP_REASON_PREFIX = "action_loop:" @@ -1706,14 +1719,22 @@ async def run_agent_tool_loop( # (e.g. "next") on every page relies on THIS clear to survive — page_transitioned alone # deliberately does not clear the action-loop guard (see below), so only a progressed # snapshot does. - _clear_action_state() + ring = ledger.content_only.get(action_key) if content_only_digest is not None else None + # INVARIANT: this test and the budget-evidence test below must both refuse a return to + # the ring. They answer one question — did the run reach ground it has not already + # covered? — and relaxing either alone silently reopens SKY-14998: a page that CYCLES + # moves on every probe, so `progressed` holds every round, and an unconditional clear + # let the action DRIVING the oscillation reset its own counter forever (one production + # key ran 11 times against a threshold of 6 while the nudge fired once). + returned_to_known_ground = ring is not None and content_only_digest in ring + if not returned_to_known_ground: + _clear_action_state() # Two landed digests that differ are positive evidence, exactly like a fingerprint # mismatch — and the only movement evidence there is when page_fingerprint is absent. canonical.progress(_ProgressEvidence.PERCEPTION_DIGEST) # Evidence requires NEW content: a URL-only flip (history.pushState) still clears the # repeat guards above but earns no budget, and neither does a return to a content state # in the probe's recent ring (a panel toggling open and shut). - ring = ledger.content_only.get(action_key) if content_only_digest is not None else None if ring and content_only_digest not in ring: _note_page_change_evidence() if content_only_digest is not None: diff --git a/skyvern/forge/taskv3/tools.py b/skyvern/forge/taskv3/tools.py index a8b992e88..b78fdb53f 100644 --- a/skyvern/forge/taskv3/tools.py +++ b/skyvern/forge/taskv3/tools.py @@ -760,14 +760,29 @@ def _lone_duplicate_candidate(rows: list[dict[str, Any]]) -> int | None: present_vals = {str(o.get("val")) for o in rows if o.get("val") is not None} if len(present_vals) >= 2: return None - # Sets must be NESTED, not equal: an a11y copy often mirrors only a subset of the primary row's - # attributes ({"55", "New York"} beside {"55"} is one candidate). Two sets neither containing the - # other carry a genuine disagreement and refuse. - val_sets = [frozenset(str(v) for v in o.get("vals") or []) for o in rows] - non_empty_sets = [s for s in val_sets if s] - for i, a in enumerate(non_empty_sets): - for b in non_empty_sets[i + 1 :]: - if not (a <= b or b <= a): + # Two-rule agreement over the surface+attribute-keyed entries. (1) A key present on BOTH rows + # must carry one value — crossed values across surfaces refuse, while a value carried on one + # row's surface only says nothing against the other row. (2) The depth-blind floor: one row's + # flattened attr=value pairs must nest inside the other's. Rows declaring identity on disjoint + # attributes cannot be confirmed the same, and an agreeing generic attribute (a shared title) + # beside disjoint identifiers is ordinary markup, not identity evidence — the pairs don't nest, + # so it refuses. A row declaring NOTHING still constrains nothing (its empty set nests, so an + # attribute-less a11y copy collapses). + val_maps: list[dict[str, str]] = [] + for o in rows: + entries: dict[str, str] = {} + for raw in o.get("vals") or []: + key, _, val_part = str(raw).partition("=") + entries[key] = val_part + val_maps.append(entries) + for i, first in enumerate(val_maps): + for second in val_maps[i + 1 :]: + for shared in first.keys() & second.keys(): + if first[shared] != second[shared]: + return None + flat_first = {f"{k.partition(':')[2]}={v}" for k, v in first.items()} + flat_second = {f"{k.partition(':')[2]}={v}" for k, v in second.items()} + if not (flat_first <= flat_second or flat_second <= flat_first): return None n = rows[0].get("n") return n if isinstance(n, int) else None @@ -873,21 +888,43 @@ def _ambiguous_rows_error( ) -def _row_value_suffix(o: dict[str, Any]) -> str: +def _row_value_suffix(o: dict[str, Any], rows: list[dict[str, Any]]) -> str: """Every present distinguishing surface for one row — value, accessible label, then any other declared value (data-code, data-key, ...) — each truncated to 60 chars like the row text already is, so a page-controlled attribute can never blow a refusal message up wholesale. Surfaces are additive (not first-match), since the veto may have come from a surface other than the first one present. + `rows` is the same-text row set the refusal lists: a surface that agrees across every row + distinguishes nothing, so the row-label clause prints only when the names disagree, and the + capped vals render entries that differ across rows first. """ parts = "" if o.get("val") is not None: parts += f" (value {str(o.get('val'))[:60]!r})" if o.get("label"): parts += f" (label {str(o.get('label'))[:60]!r})" + # The ancestor's name is its own surface: when it differs from the preferred display label (a + # shared leaf label masking distinct row names) AND disagrees across the rows, the disagreeing + # name is the one that converts — gated like the veto itself, on cross-row disagreement. + ancestor_name = (o.get("labels") or [None, None])[1] + ancestor_names = {str((p.get("labels") or [None, None])[1]) for p in rows if (p.get("labels") or [None, None])[1]} + if ancestor_name and str(ancestor_name) != str(o.get("label") or "") and len(ancestor_names) >= 2: + parts += f" (row label {str(ancestor_name)[:60]!r})" # `vals` entries are attribute-keyed for comparison; render only the value part, and subtract # what the value surface already showed so a row never prints one value twice. shown_already = {str(o.get("val")).strip()} if o.get("val") is not None else set() - vals = [bare[:60] for v in (o.get("vals") or []) if (bare := str(v).split("=", 1)[-1]) not in shown_already] + raw_vals = [str(v) for v in (o.get("vals") or [])] + # The caller is being asked to pick between these rows: entries every row carries identically + # cannot be what tells them apart, so the ones that differ print first (the cap must not hide + # the distinguishing value behind agreeing generic ones). + common_to_all = set(raw_vals) + for p in rows: + if p is not o: + common_to_all &= {str(v) for v in (p.get("vals") or [])} + vals = [ + bare[:60] + for v in sorted(raw_vals, key=lambda v: v in common_to_all) + if (bare := v.split("=", 1)[-1]) not in shown_already + ] if vals: shown_vals = vals[:3] more = "; ..." if len(vals) > 3 else "" @@ -911,7 +948,7 @@ def _identical_text_rows_error( # and the stale tags may re-land on different rows at the next scan. return f'[data-tv3-sugg="{o.get("n")}"] ' if tags_live else "" - listing = "; ".join(f"{_sel(o)}{str(o.get('text') or '')[:60]!r}{_row_value_suffix(o)}" for o in shown) + listing = "; ".join(f"{_sel(o)}{str(o.get('text') or '')[:60]!r}{_row_value_suffix(o, rows)}" for o in shown) more = len(rows) - len(shown) next_step = ( 'click the intended row directly by its [data-tv3-sugg="N"] selector; the typed query was left ' @@ -4325,13 +4362,17 @@ _MENU_OPTION_TEXTS_JS = ( // The DOM `.value` IDL is the spec submission value ALREADY -- an explicit `value=""` and an // absent attribute (which falls back to the option's own text) are genuinely distinct submission // values even though both can display identical text, so this always reads it, never null. + // NOT trimmed: the DOM preserves whitespace in option submission values, so "x" and "x " are + // genuinely distinct — the trim rule covers attribute-authoring drift, not submission values. val = String(el.value); } else { // Same presence rule as the OPTION branch: an attribute that EXISTS is a present value even // when empty -- `data-value=""` next to `data-value="x"` is a real disagreement, not absence. const dv = el.getAttribute('data-value'); const va = el.getAttribute('value'); - val = dv !== null ? dv : va; + // data-value is authoring metadata (whitespace drift between copies is presentational); a + // `value` attribute is a submission value and stays byte-exact like the OPTION branch above. + val = dv !== null ? String(dv).trim() : va !== null ? String(va) : null; } // The collapse's OWN value read: the verifier's declaredValues drops all-digit and >40-char // values on purpose (they must not CONFIRM a commit), but for telling two same-text rows apart a @@ -4344,6 +4385,11 @@ _MENU_OPTION_TEXTS_JS = ( const vetoVals = []; try { for (const node of new Set([el, opt || el])) { + // Keyed by SURFACE and attribute ("a:" the option row, "l:" a leaf inside it): crossed + // values across surfaces are disagreements a flat set cannot see. The key must come from + // DOM structure, not tagging depth — a self-tagged row and its twin tagged at a leaf must + // read the row's attributes under the same key, or equal renders refuse each other. + const where = node === (opt || el) ? 'a:' : 'l:'; for (const a of node.attributes) { if (!VETO_ATTR.test(a.name)) continue; // Trimmed for comparison: whitespace drift between a portal copy and an inline copy is @@ -4354,9 +4400,7 @@ _MENU_OPTION_TEXTS_JS = ( // veto blind to distinct long values, while the fingerprint still tells apart any pair // differing in prefix or length. Only identical-prefix-same-length pairs read as equal. if (v.length > 512) v = v.slice(0, 512) + '#len' + v.length; - // Keyed by attribute: crossed values across surfaces (data-code="A" data-key="B" beside - // data-code="B" data-key="A") are a disagreement an unkeyed set cannot see. - vetoVals.push(a.name + '=' + v); + vetoVals.push(where + a.name + '=' + v); } } } catch (e) { /* attributes unreadable: no veto values */ } @@ -4367,7 +4411,8 @@ _MENU_OPTION_TEXTS_JS = ( setsize: Number.isFinite(setsize) && setsize > 0 ? setsize : 0, pos: pos, val: val, - vals: vetoVals.slice(0, 12), + // 9 allowlisted attributes x 2 nodes = 18 possible entries; 24 can never truncate. + vals: vetoVals.slice(0, 24), // Names are FULL veto inputs (truncation happens only where a message renders them): a slice // here would read two labels diverging past the cut as equal and collapse distinct rows. label: (accessibleName(el) || accessibleName(opt)) || null, @@ -5180,6 +5225,9 @@ async () => { // diverge whenever a retained control is dropped later for having no selector that names it. let hiddenKept = 0; let hiddenListed = 0; + // Candidates the visibility gates below drop. Those drops are silent, so a page whose whole app + // shell is behind a boot gate renders exactly like an empty one; this is what tells the two apart. + let hiddenDropped = 0; let phantomDropped = 0; let truncated = 0; let truncatedInComponents = 0; @@ -5293,12 +5341,12 @@ async () => { // like v1 (whose center_x check is only reached for a non-zero rect), so the zero-size // skinned-proxy carve-out below still runs for an off-screen-positioned skinned control. const centerX = (gr.left + gr.width) / 2 + window.scrollX; - if (ownGated && gr.width !== 0 && gr.height !== 0 && centerX < 0 && !_hScrolledAncestor(gateEl)) { continue; } + if (ownGated && gr.width !== 0 && gr.height !== 0 && centerX < 0 && !_hScrolledAncestor(gateEl)) { hiddenDropped++; continue; } // v1's isElementStyleVisibilityVisible (domUtils.js) drops a control whose own computed // visibility is not 'visible'. Scoped to non-zero-rect elements so the zero-size skinned-proxy // carve-out below still runs; visibility is read per-element, so a visibility:visible child of a // hidden ancestor is kept. A native checkbox/radio judges the parent here instead of itself. - if (ownGated && gr.width !== 0 && gr.height !== 0 && window.getComputedStyle(gateEl).visibility !== 'visible') { continue; } + if (ownGated && gr.width !== 0 && gr.height !== 0 && window.getComputedStyle(gateEl).visibility !== 'visible') { hiddenDropped++; continue; } let hidden = false; if (r.width === 0 || r.height === 0) { // Design systems skin a native SELECT/checkbox/radio/file input at zero size behind a styled @@ -5319,6 +5367,7 @@ async () => { // only when it actually renders visible content, as v1's recursion does -- an empty, // all-hidden, or all-off-canvas host is a phantom. } else { + hiddenDropped++; continue; } } @@ -5834,7 +5883,7 @@ async () => { } } catch (e) { iframeInfo.total = 0; iframeInfo.inComponents = 0; iframeInfo.entries.length = 0; iframeInfo.unread = 0; iframeInfo.failed = true; } - return JSON.stringify({ url: location.href, title: document.title, text: texts, textFull: texts.map((t) => { const f = fullText.get(t); return f && f !== t ? f : null; }), textTruncated: textFull, textDropped: textDropped, iframes: iframeInfo, dropped: dropped, truncated: truncated, truncatedInComponents: truncatedInComponents, unnamedAnonymous: unnamedAnonymous, unnamedBudget: unnamedBudget, unnamedDuplicated: unnamedDuplicated, unnamedUnverifiable: unnamedUnverifiable, unnamedUnsafe: unnamedUnsafe, unreadableRoot: sawUnreadableRoot, undiscoveredRoots: undiscoveredRoots, rootCount: allRoots.length - 1, hiddenListed: hiddenListed, phantomDropped: phantomDropped, markersMinted: markersWritten, markersReused: markersReused, pageMutated: mutated, elements: out }); + return JSON.stringify({ url: location.href, title: document.title, text: texts, textFull: texts.map((t) => { const f = fullText.get(t); return f && f !== t ? f : null; }), textTruncated: textFull, textDropped: textDropped, iframes: iframeInfo, dropped: dropped, truncated: truncated, truncatedInComponents: truncatedInComponents, unnamedAnonymous: unnamedAnonymous, unnamedBudget: unnamedBudget, unnamedDuplicated: unnamedDuplicated, unnamedUnverifiable: unnamedUnverifiable, unnamedUnsafe: unnamedUnsafe, unreadableRoot: sawUnreadableRoot, undiscoveredRoots: undiscoveredRoots, rootCount: allRoots.length - 1, hiddenListed: hiddenListed, hiddenDropped: hiddenDropped, phantomDropped: phantomDropped, markersMinted: markersWritten, markersReused: markersReused, pageMutated: mutated, elements: out }); } """ ) @@ -6577,6 +6626,20 @@ def build_browser_tools( lines.append( f"note: {hidden_kept} native control(s) hidden behind styled proxies are listed with [hidden-native]" ) + hidden_dropped = data.get("hiddenDropped") or 0 + # Scoped to total blindness: present-but-hidden chrome (closed menus, inactive tabs) is on + # nearly every page, so an unconditional note would cost the prefix on every call to say + # nothing. With no element listed the count is the whole signal -- it separates an app shell + # still behind its boot gate from a page that genuinely has no controls, which is the one + # distinction the model cannot otherwise make and will poll wait->observe for turns to guess. + # Deliberately time-neutral: a boot gate resolves on its own and a stuck one never does, and + # this cannot tell which, so the wording must not imply that waiting is what fixes it. + if not elements and hidden_dropped: + lines.append( + f"note: the page has {hidden_dropped} control(s) that are present but not visible " + "(CSS-hidden or positioned off-canvas): the DOM is populated but none of it is " + "currently actionable" + ) phantom_dropped = data.get("phantomDropped") or 0 if phantom_dropped: lines.append( @@ -6753,6 +6816,7 @@ def build_browser_tools( summary = { "text_dropped": text_dropped, "hidden_listed": hidden_kept, + "hidden_dropped": hidden_dropped, "phantom_dropped": phantom_dropped, "iframes_in_component_roots": iframe_info.get("inComponents") or 0, "undiscovered_roots": data.get("undiscoveredRoots") or 0, @@ -8753,7 +8817,8 @@ def build_browser_tools( def _listed_row(o: dict[str, Any], canon_text: str) -> str: entry = f'[data-tv3-menu="{o.get("n")}"] {str(o.get("text") or "")[:60]!r}' if text_counts[canon_text] >= 2: - entry += _row_value_suffix(o) + same_text = [p for p, pt in zip(shown, shown_canon_texts) if pt == canon_text] + entry += _row_value_suffix(o, same_text) return entry listing = "; ".join(_listed_row(o, t) for o, t in zip(shown, shown_canon_texts)) diff --git a/tests/unit/test_api_docs_taxonomy.py b/tests/unit/test_api_docs_taxonomy.py index d834ca44a..ddab6a23c 100644 --- a/tests/unit/test_api_docs_taxonomy.py +++ b/tests/unit/test_api_docs_taxonomy.py @@ -15,6 +15,10 @@ sync workflow, and `scripts/sync_openapi_docs.py --check`. import json from pathlib import Path +import pytest + +pytestmark = pytest.mark.boundary + COMMITTED_SPEC = Path(__file__).resolve().parents[2] / "docs" / "api-reference" / "openapi.json" # Capitalized resource tags — kept in lockstep with .agents/skills/api-docs-audit/SKILL.md. diff --git a/tests/unit/test_docs_webhook_verification_snippets.py b/tests/unit/test_docs_webhook_verification_snippets.py index afd6fd31b..dda995831 100644 --- a/tests/unit/test_docs_webhook_verification_snippets.py +++ b/tests/unit/test_docs_webhook_verification_snippets.py @@ -18,6 +18,8 @@ import pytest from skyvern.forge.sdk.core.security import generate_skyvern_webhook_signature +pytestmark = pytest.mark.boundary + REPO_ROOT = Path(__file__).resolve().parents[2] SNIPPET = REPO_ROOT / "docs/snippets/webhook-signature-verification.mdx" diff --git a/tests/unit/test_taskv3_loop.py b/tests/unit/test_taskv3_loop.py index c6048c8e9..6d56bb963 100644 --- a/tests/unit/test_taskv3_loop.py +++ b/tests/unit/test_taskv3_loop.py @@ -28,7 +28,9 @@ from skyvern.forge.taskv3 import loop as loop_module from skyvern.forge.taskv3.loop import ( ACTION_BUDGET_EXTENDED_EVENT, ACTION_BUDGET_EXTENSION_REFUSED_EVENT, + ACTION_LOOP_NUDGE_AFTER, ACTION_LOOP_REASON_PREFIX, + ACTION_LOOP_TERMINATE_AFTER, CANONICAL_SURVIVAL_EVENT, FAILURE_EVIDENCE_MIN_TOOL_CALLS, FAILURE_EVIDENCE_MIN_TURNS, @@ -3543,6 +3545,44 @@ async def test_action_loop_catches_varied_probe_evasion() -> None: assert caller.calls <= 15 +def test_action_loop_terminate_threshold_is_pinned_to_its_measured_value() -> None: + # The other action-loop tests read this constant so they track policy instead of drifting, which + # leaves nothing asserting the VALUE — someone could set it to 50 and the suite would stay green. + # Pinning it makes any change deliberate and visible in a diff, and sends the reader to the + # do-not-lower note at the constant: the effective post-clearing counter was measured, showed no + # separation between completed and stuck runs, and its highest observed value fell in a completed + # run — so there is no positive evidence supporting a lower threshold. + assert ACTION_LOOP_TERMINATE_AFTER == 8 + assert ACTION_LOOP_NUDGE_AFTER < ACTION_LOOP_TERMINATE_AFTER + + +@pytest.mark.asyncio +async def test_action_loop_survives_a_page_that_oscillates_between_two_known_states() -> None: + # The production shape the guard was structurally blind to (SKY-14998, tsk in wr_568475173904014164): + # one action key ran 11 times against a terminate threshold of 6, and the repeat nudge fired + # exactly ONCE at repeat_count=3 before the run died on the token cap. A page that CYCLES rather + # than freezes moves on every probe, so `snap.progressed` held every round and wiped the whole + # repeat ledger — the action driving the oscillation reset its own counter forever. Returning to + # a state this probe has already seen is not progress and must not clear the guard. + from skyvern.forge.taskv3.loop import ACTION_LOOP_REASON_PREFIX + + panel = ["url=x text: 'filters panel open'", "url=x text: 'filters panel shut'"] + contents = [panel[i % 2] for i in range(16)] + script: list[list[tuple[str, dict[str, Any]]]] = [] + for _ in range(16): + script.append([("click", {"selector": "#apply"})]) + script.append([("observe", {})]) + clicks: list[tuple[str, dict[str, Any]]] = [] + tools = [_billable_tool("click", clicks), _perception_tool("observe", contents), make_finish_tool()] + outcome, _ = await _run(script, tools, max_turns=200, max_tool_calls=500) + + assert outcome.status == "terminated" + assert outcome.reason.startswith(ACTION_LOOP_REASON_PREFIX) + assert "#apply" in outcome.reason + # Bounded well below the 16 the script offers, and below the 11 the production run reached. + assert len(clicks) <= 10, len(clicks) + + @pytest.mark.asyncio async def test_pagination_with_changing_page_content_never_trips_action_loop() -> None: # Healthy pagination clicks the same Next selector many times, but each page's observe differs — @@ -3637,7 +3677,8 @@ async def test_action_loop_warn_and_terminate_emit_facetable_logs() -> None: warned = [entry for entry in logs if entry["event"] == "taskv3 loop action repeat nudged"] terminated = [entry for entry in logs if entry["event"] == "taskv3 loop action repeated"] assert len(warned) == 1 and warned[0]["tool"] == "click" and warned[0]["repeat_count"] == 3 - assert len(terminated) == 1 and terminated[0]["tool"] == "click" and terminated[0]["repeat_count"] == 6 + assert len(terminated) == 1 and terminated[0]["tool"] == "click" + assert terminated[0]["repeat_count"] == ACTION_LOOP_TERMINATE_AFTER @pytest.mark.asyncio @@ -4071,7 +4112,9 @@ async def test_warn_always_precedes_terminate_even_after_single_batch_burst() -> from skyvern.forge.taskv3.loop import ACTION_LOOP_REASON_PREFIX script = [ - [("click", {"selector": "#submit"})] * 5, + # The burst must cross the terminate threshold inside ONE turn, or the property under test + # (no verdict before a delivered warning) is never exercised. + [("click", {"selector": "#submit"})] * ACTION_LOOP_TERMINATE_AFTER, [("click", {"selector": "#submit"})], [("click", {"selector": "#submit"})], ] @@ -4080,7 +4123,8 @@ async def test_warn_always_precedes_terminate_even_after_single_batch_burst() -> outcome, _ = await _run(script, tools, max_turns=200, max_tool_calls=500) assert outcome.status == "terminated" assert outcome.reason.startswith(ACTION_LOOP_REASON_PREFIX) - assert len(clicks) == 7 # 5 burst + 1 post-warn-queue + 1 post-warn-delivery + # burst + 1 post-warn-queue + 1 post-warn-delivery + assert len(clicks) == ACTION_LOOP_TERMINATE_AFTER + 2 warns = [m for m in outcome.messages if m.get("role") == "user" and "#submit" in str(m.get("content"))] assert len(warns) == 1 assert outcome.messages.index(warns[0]) < len(outcome.messages) - 1 @@ -4095,11 +4139,11 @@ async def test_action_loop_counts_errored_attempts() -> None: clicks: list[tuple[str, dict[str, Any]]] = [] click = _recording_tool("click", clicks, raises=True) click.billable = True - script = [[("click", {"selector": "#dead"})] for _ in range(10)] + script = [[("click", {"selector": "#dead"})] for _ in range(ACTION_LOOP_TERMINATE_AFTER + 4)] outcome, _ = await _run(script, [click, make_finish_tool()], max_turns=200, max_tool_calls=500) assert outcome.status == "terminated" assert outcome.reason.startswith(ACTION_LOOP_REASON_PREFIX) - assert len(clicks) == 6 + assert len(clicks) == ACTION_LOOP_TERMINATE_AFTER @pytest.mark.asyncio diff --git a/tests/unit/test_taskv3_tools.py b/tests/unit/test_taskv3_tools.py index 3b43d6f06..a38e229f9 100644 --- a/tests/unit/test_taskv3_tools.py +++ b/tests/unit/test_taskv3_tools.py @@ -1679,6 +1679,7 @@ async def test_observe_result_carries_count_only_summary_for_the_call_record() - assert set(summary) == { "text_dropped", "hidden_listed", + "hidden_dropped", "phantom_dropped", "iframes_in_component_roots", "undiscovered_roots", @@ -16226,12 +16227,15 @@ def _button_listbox_html( anchor_id: str = "cc", labels: list[str | None] | None = None, labelledby: list[str | None] | None = None, - wrap_in_span: bool = False, + wrap_in_span: bool | list[bool] = False, span_label: str | None = None, + span_attrs: list[dict[str, str]] | None = None, extra_attrs: list[dict[str, str]] | None = None, ) -> str: labels = labels or [None] * len(rows) extra_attrs = extra_attrs or [{} for _ in rows] + span_attrs = span_attrs or [{} for _ in rows] + wraps = wrap_in_span if isinstance(wrap_in_span, list) else [wrap_in_span] * len(rows) # `labelledby[i]` is the TEXT of a hidden div this row's aria-labelledby points at (not an id) -- # the id is minted from the row's own position so each row gets a distinct label target. labelledby = labelledby or [None] * len(rows) @@ -16248,7 +16252,9 @@ def _button_listbox_html( for k, v in extra_attrs[i].items(): attrs += f" {k}={v!r}" span_attr = f" aria-label={span_label!r}" if span_label else "" - inner = f'{text}' if wrap_in_span else text + for k, v in span_attrs[i].items(): + span_attr += f" {k}={v!r}" + inner = f'{text}' if wraps[i] else text return f'
Welcome back.
" + '' + '' + '