Task V3 combobox collapse: node-keyed veto maps and refusal refinements (SKY-15514) (#8441)

This commit is contained in:
pedrohsdb 2026-09-03 18:02:13 -07:00 • committed by GitHub
parent 7e6c631a49
commit 1076d3fed0
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
7 changed files with 510 additions and 30 deletions

View file

@ -10,3 +10,4 @@ addopts = --capture=no
norecursedirs = tests/manual tests/evals scripts/cdp-download-poc tests/sdk eval infra/gcp-agent prompt_evaluation internal-tools
markers =
workday_offline: opt-in offline Workday deterministic regression suite (needs a Playwright chromium; run with -m workday_offline)
boundary: reads skyvern-frontend/ or docs/ from the repo, so it must run on FE/docs-only PRs that skip the Python suite

View file

@ -203,7 +203,20 @@ PERCEPTION_STALL_REASON_PREFIX = "perception_stall:"
# returning different content, or a download landing; a first-time probe has no baseline and is
# evidence of nothing, so varied-selector probing cannot launder repetition into "progress".
ACTION_LOOP_NUDGE_AFTER = 3
ACTION_LOOP_TERMINATE_AFTER = 6
# 8, not 6: RAW repeats of one action key peak at 6 across 50 completed prod runs while stuck runs
# reach 36, so 6 sat on the completed population's edge. DO NOT LOWER THIS. The effective
# post-clearing counter — what this constant is actually compared against — was then measured by
# replaying the clearing rule over per-call telemetry, and NO SEPARATION WAS OBSERVED: two completed
# runs peaked at 2 and 4, four stuck runs at 1, 3, 3 and below (upper bounds, computed identically
# on both sides), and the highest observed value fell in a COMPLETED run. n=6 is far too small to
# establish inversion as a property of either population — that spread is also consistent with
# noise — but it is no evidence FOR a lower threshold either, and lowering a safety threshold
# requires positive evidence. 8 is above everything observed, which is why the change is safe; it
# is also why nothing in the sample fires. REVISIT if the effective distribution ever becomes
# measurable at population scale; do not lower it before then. Repeat count may not be a stuck-ness
# signal at all: SKY-15602 tracks the real problem, which is that the loop has no definition of
# goal progress, only of page change.
ACTION_LOOP_TERMINATE_AFTER = 8
# Facetable sibling of PERCEPTION_STALL_REASON_PREFIX; same dashboard contract.
ACTION_LOOP_REASON_PREFIX = "action_loop:"
@ -1706,14 +1719,22 @@ async def run_agent_tool_loop(
# (e.g. "next") on every page relies on THIS clear to survive — page_transitioned alone
# deliberately does not clear the action-loop guard (see below), so only a progressed
# snapshot does.
_clear_action_state()
ring = ledger.content_only.get(action_key) if content_only_digest is not None else None
# INVARIANT: this test and the budget-evidence test below must both refuse a return to
# the ring. They answer one question — did the run reach ground it has not already
# covered? — and relaxing either alone silently reopens SKY-14998: a page that CYCLES
# moves on every probe, so `progressed` holds every round, and an unconditional clear
# let the action DRIVING the oscillation reset its own counter forever (one production
# key ran 11 times against a threshold of 6 while the nudge fired once).
returned_to_known_ground = ring is not None and content_only_digest in ring
if not returned_to_known_ground:
_clear_action_state()
# Two landed digests that differ are positive evidence, exactly like a fingerprint
# mismatch — and the only movement evidence there is when page_fingerprint is absent.
canonical.progress(_ProgressEvidence.PERCEPTION_DIGEST)
# Evidence requires NEW content: a URL-only flip (history.pushState) still clears the
# repeat guards above but earns no budget, and neither does a return to a content state
# in the probe's recent ring (a panel toggling open and shut).
ring = ledger.content_only.get(action_key) if content_only_digest is not None else None
if ring and content_only_digest not in ring:
_note_page_change_evidence()
if content_only_digest is not None:

View file

@ -760,14 +760,29 @@ def _lone_duplicate_candidate(rows: list[dict[str, Any]]) -> int | None:
present_vals = {str(o.get("val")) for o in rows if o.get("val") is not None}
if len(present_vals) >= 2:
return None
# Sets must be NESTED, not equal: an a11y copy often mirrors only a subset of the primary row's
# attributes ({"55", "New York"} beside {"55"} is one candidate). Two sets neither containing the
# other carry a genuine disagreement and refuse.
val_sets = [frozenset(str(v) for v in o.get("vals") or []) for o in rows]
non_empty_sets = [s for s in val_sets if s]
for i, a in enumerate(non_empty_sets):
for b in non_empty_sets[i + 1 :]:
if not (a <= b or b <= a):
# Two-rule agreement over the surface+attribute-keyed entries. (1) A key present on BOTH rows
# must carry one value — crossed values across surfaces refuse, while a value carried on one
# row's surface only says nothing against the other row. (2) The depth-blind floor: one row's
# flattened attr=value pairs must nest inside the other's. Rows declaring identity on disjoint
# attributes cannot be confirmed the same, and an agreeing generic attribute (a shared title)
# beside disjoint identifiers is ordinary markup, not identity evidence — the pairs don't nest,
# so it refuses. A row declaring NOTHING still constrains nothing (its empty set nests, so an
# attribute-less a11y copy collapses).
val_maps: list[dict[str, str]] = []
for o in rows:
entries: dict[str, str] = {}
for raw in o.get("vals") or []:
key, _, val_part = str(raw).partition("=")
entries[key] = val_part
val_maps.append(entries)
for i, first in enumerate(val_maps):
for second in val_maps[i + 1 :]:
for shared in first.keys() & second.keys():
if first[shared] != second[shared]:
return None
flat_first = {f"{k.partition(':')[2]}={v}" for k, v in first.items()}
flat_second = {f"{k.partition(':')[2]}={v}" for k, v in second.items()}
if not (flat_first <= flat_second or flat_second <= flat_first):
return None
n = rows[0].get("n")
return n if isinstance(n, int) else None
@ -873,21 +888,43 @@ def _ambiguous_rows_error(
)
def _row_value_suffix(o: dict[str, Any]) -> str:
def _row_value_suffix(o: dict[str, Any], rows: list[dict[str, Any]]) -> str:
"""Every present distinguishing surface for one row — value, accessible label, then any other
declared value (data-code, data-key, ...) — each truncated to 60 chars like the row text already is,
so a page-controlled attribute can never blow a refusal message up wholesale. Surfaces are additive
(not first-match), since the veto may have come from a surface other than the first one present.
`rows` is the same-text row set the refusal lists: a surface that agrees across every row
distinguishes nothing, so the row-label clause prints only when the names disagree, and the
capped vals render entries that differ across rows first.
"""
parts = ""
if o.get("val") is not None:
parts += f" (value {str(o.get('val'))[:60]!r})"
if o.get("label"):
parts += f" (label {str(o.get('label'))[:60]!r})"
# The ancestor's name is its own surface: when it differs from the preferred display label (a
# shared leaf label masking distinct row names) AND disagrees across the rows, the disagreeing
# name is the one that converts — gated like the veto itself, on cross-row disagreement.
ancestor_name = (o.get("labels") or [None, None])[1]
ancestor_names = {str((p.get("labels") or [None, None])[1]) for p in rows if (p.get("labels") or [None, None])[1]}
if ancestor_name and str(ancestor_name) != str(o.get("label") or "") and len(ancestor_names) >= 2:
parts += f" (row label {str(ancestor_name)[:60]!r})"
# `vals` entries are attribute-keyed for comparison; render only the value part, and subtract
# what the value surface already showed so a row never prints one value twice.
shown_already = {str(o.get("val")).strip()} if o.get("val") is not None else set()
vals = [bare[:60] for v in (o.get("vals") or []) if (bare := str(v).split("=", 1)[-1]) not in shown_already]
raw_vals = [str(v) for v in (o.get("vals") or [])]
# The caller is being asked to pick between these rows: entries every row carries identically
# cannot be what tells them apart, so the ones that differ print first (the cap must not hide
# the distinguishing value behind agreeing generic ones).
common_to_all = set(raw_vals)
for p in rows:
if p is not o:
common_to_all &= {str(v) for v in (p.get("vals") or [])}
vals = [
bare[:60]
for v in sorted(raw_vals, key=lambda v: v in common_to_all)
if (bare := v.split("=", 1)[-1]) not in shown_already
]
if vals:
shown_vals = vals[:3]
more = "; ..." if len(vals) > 3 else ""
@ -911,7 +948,7 @@ def _identical_text_rows_error(
# and the stale tags may re-land on different rows at the next scan.
return f'[data-tv3-sugg="{o.get("n")}"] ' if tags_live else ""
listing = "; ".join(f"{_sel(o)}{str(o.get('text') or '')[:60]!r}{_row_value_suffix(o)}" for o in shown)
listing = "; ".join(f"{_sel(o)}{str(o.get('text') or '')[:60]!r}{_row_value_suffix(o, rows)}" for o in shown)
more = len(rows) - len(shown)
next_step = (
'click the intended row directly by its [data-tv3-sugg="N"] selector; the typed query was left '
@ -4325,13 +4362,17 @@ _MENU_OPTION_TEXTS_JS = (
// The DOM `.value` IDL is the spec submission value ALREADY -- an explicit `value=""` and an
// absent attribute (which falls back to the option's own text) are genuinely distinct submission
// values even though both can display identical text, so this always reads it, never null.
// NOT trimmed: the DOM preserves whitespace in option submission values, so "x" and "x " are
// genuinely distinct — the trim rule covers attribute-authoring drift, not submission values.
val = String(el.value);
} else {
// Same presence rule as the OPTION branch: an attribute that EXISTS is a present value even
// when empty -- `data-value=""` next to `data-value="x"` is a real disagreement, not absence.
const dv = el.getAttribute('data-value');
const va = el.getAttribute('value');
val = dv !== null ? dv : va;
// data-value is authoring metadata (whitespace drift between copies is presentational); a
// `value` attribute is a submission value and stays byte-exact like the OPTION branch above.
val = dv !== null ? String(dv).trim() : va !== null ? String(va) : null;
}
// The collapse's OWN value read: the verifier's declaredValues drops all-digit and >40-char
// values on purpose (they must not CONFIRM a commit), but for telling two same-text rows apart a
@ -4344,6 +4385,11 @@ _MENU_OPTION_TEXTS_JS = (
const vetoVals = [];
try {
for (const node of new Set([el, opt || el])) {
// Keyed by SURFACE and attribute ("a:" the option row, "l:" a leaf inside it): crossed
// values across surfaces are disagreements a flat set cannot see. The key must come from
// DOM structure, not tagging depth — a self-tagged row and its twin tagged at a leaf must
// read the row's attributes under the same key, or equal renders refuse each other.
const where = node === (opt || el) ? 'a:' : 'l:';
for (const a of node.attributes) {
if (!VETO_ATTR.test(a.name)) continue;
// Trimmed for comparison: whitespace drift between a portal copy and an inline copy is
@ -4354,9 +4400,7 @@ _MENU_OPTION_TEXTS_JS = (
// veto blind to distinct long values, while the fingerprint still tells apart any pair
// differing in prefix or length. Only identical-prefix-same-length pairs read as equal.
if (v.length > 512) v = v.slice(0, 512) + '#len' + v.length;
// Keyed by attribute: crossed values across surfaces (data-code="A" data-key="B" beside
// data-code="B" data-key="A") are a disagreement an unkeyed set cannot see.
vetoVals.push(a.name + '=' + v);
vetoVals.push(where + a.name + '=' + v);
}
}
} catch (e) { /* attributes unreadable: no veto values */ }
@ -4367,7 +4411,8 @@ _MENU_OPTION_TEXTS_JS = (
setsize: Number.isFinite(setsize) && setsize > 0 ? setsize : 0,
pos: pos,
val: val,
vals: vetoVals.slice(0, 12),
// 9 allowlisted attributes x 2 nodes = 18 possible entries; 24 can never truncate.
vals: vetoVals.slice(0, 24),
// Names are FULL veto inputs (truncation happens only where a message renders them): a slice
// here would read two labels diverging past the cut as equal and collapse distinct rows.
label: (accessibleName(el) || accessibleName(opt)) || null,
@ -5180,6 +5225,9 @@ async () => {
// diverge whenever a retained control is dropped later for having no selector that names it.
let hiddenKept = 0;
let hiddenListed = 0;
// Candidates the visibility gates below drop. Those drops are silent, so a page whose whole app
// shell is behind a boot gate renders exactly like an empty one; this is what tells the two apart.
let hiddenDropped = 0;
let phantomDropped = 0;
let truncated = 0;
let truncatedInComponents = 0;
@ -5293,12 +5341,12 @@ async () => {
// like v1 (whose center_x check is only reached for a non-zero rect), so the zero-size
// skinned-proxy carve-out below still runs for an off-screen-positioned skinned control.
const centerX = (gr.left + gr.width) / 2 + window.scrollX;
if (ownGated && gr.width !== 0 && gr.height !== 0 && centerX < 0 && !_hScrolledAncestor(gateEl)) { continue; }
if (ownGated && gr.width !== 0 && gr.height !== 0 && centerX < 0 && !_hScrolledAncestor(gateEl)) { hiddenDropped++; continue; }
// v1's isElementStyleVisibilityVisible (domUtils.js) drops a control whose own computed
// visibility is not 'visible'. Scoped to non-zero-rect elements so the zero-size skinned-proxy
// carve-out below still runs; visibility is read per-element, so a visibility:visible child of a
// hidden ancestor is kept. A native checkbox/radio judges the parent here instead of itself.
if (ownGated && gr.width !== 0 && gr.height !== 0 && window.getComputedStyle(gateEl).visibility !== 'visible') { continue; }
if (ownGated && gr.width !== 0 && gr.height !== 0 && window.getComputedStyle(gateEl).visibility !== 'visible') { hiddenDropped++; continue; }
let hidden = false;
if (r.width === 0 || r.height === 0) {
// Design systems skin a native SELECT/checkbox/radio/file input at zero size behind a styled
@ -5319,6 +5367,7 @@ async () => {
// only when it actually renders visible content, as v1's recursion does -- an empty,
// all-hidden, or all-off-canvas host is a phantom.
} else {
hiddenDropped++;
continue;
}
}
@ -5834,7 +5883,7 @@ async () => {
}
} catch (e) { iframeInfo.total = 0; iframeInfo.inComponents = 0; iframeInfo.entries.length = 0; iframeInfo.unread = 0; iframeInfo.failed = true; }
return JSON.stringify({ url: location.href, title: document.title, text: texts, textFull: texts.map((t) => { const f = fullText.get(t); return f && f !== t ? f : null; }), textTruncated: textFull, textDropped: textDropped, iframes: iframeInfo, dropped: dropped, truncated: truncated, truncatedInComponents: truncatedInComponents, unnamedAnonymous: unnamedAnonymous, unnamedBudget: unnamedBudget, unnamedDuplicated: unnamedDuplicated, unnamedUnverifiable: unnamedUnverifiable, unnamedUnsafe: unnamedUnsafe, unreadableRoot: sawUnreadableRoot, undiscoveredRoots: undiscoveredRoots, rootCount: allRoots.length - 1, hiddenListed: hiddenListed, phantomDropped: phantomDropped, markersMinted: markersWritten, markersReused: markersReused, pageMutated: mutated, elements: out });
return JSON.stringify({ url: location.href, title: document.title, text: texts, textFull: texts.map((t) => { const f = fullText.get(t); return f && f !== t ? f : null; }), textTruncated: textFull, textDropped: textDropped, iframes: iframeInfo, dropped: dropped, truncated: truncated, truncatedInComponents: truncatedInComponents, unnamedAnonymous: unnamedAnonymous, unnamedBudget: unnamedBudget, unnamedDuplicated: unnamedDuplicated, unnamedUnverifiable: unnamedUnverifiable, unnamedUnsafe: unnamedUnsafe, unreadableRoot: sawUnreadableRoot, undiscoveredRoots: undiscoveredRoots, rootCount: allRoots.length - 1, hiddenListed: hiddenListed, hiddenDropped: hiddenDropped, phantomDropped: phantomDropped, markersMinted: markersWritten, markersReused: markersReused, pageMutated: mutated, elements: out });
}
"""
)
@ -6577,6 +6626,20 @@ def build_browser_tools(
lines.append(
f"note: {hidden_kept} native control(s) hidden behind styled proxies are listed with [hidden-native]"
)
hidden_dropped = data.get("hiddenDropped") or 0
# Scoped to total blindness: present-but-hidden chrome (closed menus, inactive tabs) is on
# nearly every page, so an unconditional note would cost the prefix on every call to say
# nothing. With no element listed the count is the whole signal -- it separates an app shell
# still behind its boot gate from a page that genuinely has no controls, which is the one
# distinction the model cannot otherwise make and will poll wait->observe for turns to guess.
# Deliberately time-neutral: a boot gate resolves on its own and a stuck one never does, and
# this cannot tell which, so the wording must not imply that waiting is what fixes it.
if not elements and hidden_dropped:
lines.append(
f"note: the page has {hidden_dropped} control(s) that are present but not visible "
"(CSS-hidden or positioned off-canvas): the DOM is populated but none of it is "
"currently actionable"
)
phantom_dropped = data.get("phantomDropped") or 0
if phantom_dropped:
lines.append(
@ -6753,6 +6816,7 @@ def build_browser_tools(
summary = {
"text_dropped": text_dropped,
"hidden_listed": hidden_kept,
"hidden_dropped": hidden_dropped,
"phantom_dropped": phantom_dropped,
"iframes_in_component_roots": iframe_info.get("inComponents") or 0,
"undiscovered_roots": data.get("undiscoveredRoots") or 0,
@ -8753,7 +8817,8 @@ def build_browser_tools(
def _listed_row(o: dict[str, Any], canon_text: str) -> str:
entry = f'[data-tv3-menu="{o.get("n")}"] {str(o.get("text") or "")[:60]!r}'
if text_counts[canon_text] >= 2:
entry += _row_value_suffix(o)
same_text = [p for p, pt in zip(shown, shown_canon_texts) if pt == canon_text]
entry += _row_value_suffix(o, same_text)
return entry
listing = "; ".join(_listed_row(o, t) for o, t in zip(shown, shown_canon_texts))

View file

@ -15,6 +15,10 @@ sync workflow, and `scripts/sync_openapi_docs.py --check`.
import json
from pathlib import Path
import pytest
pytestmark = pytest.mark.boundary
COMMITTED_SPEC = Path(__file__).resolve().parents[2] / "docs" / "api-reference" / "openapi.json"
# Capitalized resource tags — kept in lockstep with .agents/skills/api-docs-audit/SKILL.md.

View file

@ -18,6 +18,8 @@ import pytest
from skyvern.forge.sdk.core.security import generate_skyvern_webhook_signature
pytestmark = pytest.mark.boundary
REPO_ROOT = Path(__file__).resolve().parents[2]
SNIPPET = REPO_ROOT / "docs/snippets/webhook-signature-verification.mdx"

View file

@ -28,7 +28,9 @@ from skyvern.forge.taskv3 import loop as loop_module
from skyvern.forge.taskv3.loop import (
ACTION_BUDGET_EXTENDED_EVENT,
ACTION_BUDGET_EXTENSION_REFUSED_EVENT,
ACTION_LOOP_NUDGE_AFTER,
ACTION_LOOP_REASON_PREFIX,
ACTION_LOOP_TERMINATE_AFTER,
CANONICAL_SURVIVAL_EVENT,
FAILURE_EVIDENCE_MIN_TOOL_CALLS,
FAILURE_EVIDENCE_MIN_TURNS,
@ -3543,6 +3545,44 @@ async def test_action_loop_catches_varied_probe_evasion() -> None:
assert caller.calls <= 15
def test_action_loop_terminate_threshold_is_pinned_to_its_measured_value() -> None:
# The other action-loop tests read this constant so they track policy instead of drifting, which
# leaves nothing asserting the VALUE — someone could set it to 50 and the suite would stay green.
# Pinning it makes any change deliberate and visible in a diff, and sends the reader to the
# do-not-lower note at the constant: the effective post-clearing counter was measured, showed no
# separation between completed and stuck runs, and its highest observed value fell in a completed
# run — so there is no positive evidence supporting a lower threshold.
assert ACTION_LOOP_TERMINATE_AFTER == 8
assert ACTION_LOOP_NUDGE_AFTER < ACTION_LOOP_TERMINATE_AFTER
@pytest.mark.asyncio
async def test_action_loop_survives_a_page_that_oscillates_between_two_known_states() -> None:
# The production shape the guard was structurally blind to (SKY-14998, tsk in wr_568475173904014164):
# one action key ran 11 times against a terminate threshold of 6, and the repeat nudge fired
# exactly ONCE at repeat_count=3 before the run died on the token cap. A page that CYCLES rather
# than freezes moves on every probe, so `snap.progressed` held every round and wiped the whole
# repeat ledger — the action driving the oscillation reset its own counter forever. Returning to
# a state this probe has already seen is not progress and must not clear the guard.
from skyvern.forge.taskv3.loop import ACTION_LOOP_REASON_PREFIX
panel = ["url=x text: 'filters panel open'", "url=x text: 'filters panel shut'"]
contents = [panel[i % 2] for i in range(16)]
script: list[list[tuple[str, dict[str, Any]]]] = []
for _ in range(16):
script.append([("click", {"selector": "#apply"})])
script.append([("observe", {})])
clicks: list[tuple[str, dict[str, Any]]] = []
tools = [_billable_tool("click", clicks), _perception_tool("observe", contents), make_finish_tool()]
outcome, _ = await _run(script, tools, max_turns=200, max_tool_calls=500)
assert outcome.status == "terminated"
assert outcome.reason.startswith(ACTION_LOOP_REASON_PREFIX)
assert "#apply" in outcome.reason
# Bounded well below the 16 the script offers, and below the 11 the production run reached.
assert len(clicks) <= 10, len(clicks)
@pytest.mark.asyncio
async def test_pagination_with_changing_page_content_never_trips_action_loop() -> None:
# Healthy pagination clicks the same Next selector many times, but each page's observe differs —
@ -3637,7 +3677,8 @@ async def test_action_loop_warn_and_terminate_emit_facetable_logs() -> None:
warned = [entry for entry in logs if entry["event"] == "taskv3 loop action repeat nudged"]
terminated = [entry for entry in logs if entry["event"] == "taskv3 loop action repeated"]
assert len(warned) == 1 and warned[0]["tool"] == "click" and warned[0]["repeat_count"] == 3
assert len(terminated) == 1 and terminated[0]["tool"] == "click" and terminated[0]["repeat_count"] == 6
assert len(terminated) == 1 and terminated[0]["tool"] == "click"
assert terminated[0]["repeat_count"] == ACTION_LOOP_TERMINATE_AFTER
@pytest.mark.asyncio
@ -4071,7 +4112,9 @@ async def test_warn_always_precedes_terminate_even_after_single_batch_burst() ->
from skyvern.forge.taskv3.loop import ACTION_LOOP_REASON_PREFIX
script = [
[("click", {"selector": "#submit"})] * 5,
# The burst must cross the terminate threshold inside ONE turn, or the property under test
# (no verdict before a delivered warning) is never exercised.
[("click", {"selector": "#submit"})] * ACTION_LOOP_TERMINATE_AFTER,
[("click", {"selector": "#submit"})],
[("click", {"selector": "#submit"})],
]
@ -4080,7 +4123,8 @@ async def test_warn_always_precedes_terminate_even_after_single_batch_burst() ->
outcome, _ = await _run(script, tools, max_turns=200, max_tool_calls=500)
assert outcome.status == "terminated"
assert outcome.reason.startswith(ACTION_LOOP_REASON_PREFIX)
assert len(clicks) == 7 # 5 burst + 1 post-warn-queue + 1 post-warn-delivery
# burst + 1 post-warn-queue + 1 post-warn-delivery
assert len(clicks) == ACTION_LOOP_TERMINATE_AFTER + 2
warns = [m for m in outcome.messages if m.get("role") == "user" and "#submit" in str(m.get("content"))]
assert len(warns) == 1
assert outcome.messages.index(warns[0]) < len(outcome.messages) - 1
@ -4095,11 +4139,11 @@ async def test_action_loop_counts_errored_attempts() -> None:
clicks: list[tuple[str, dict[str, Any]]] = []
click = _recording_tool("click", clicks, raises=True)
click.billable = True
script = [[("click", {"selector": "#dead"})] for _ in range(10)]
script = [[("click", {"selector": "#dead"})] for _ in range(ACTION_LOOP_TERMINATE_AFTER + 4)]
outcome, _ = await _run(script, [click, make_finish_tool()], max_turns=200, max_tool_calls=500)
assert outcome.status == "terminated"
assert outcome.reason.startswith(ACTION_LOOP_REASON_PREFIX)
assert len(clicks) == 6
assert len(clicks) == ACTION_LOOP_TERMINATE_AFTER
@pytest.mark.asyncio

View file

@ -1679,6 +1679,7 @@ async def test_observe_result_carries_count_only_summary_for_the_call_record() -
assert set(summary) == {
"text_dropped",
"hidden_listed",
"hidden_dropped",
"phantom_dropped",
"iframes_in_component_roots",
"undiscovered_roots",
@ -16226,12 +16227,15 @@ def _button_listbox_html(
anchor_id: str = "cc",
labels: list[str | None] | None = None,
labelledby: list[str | None] | None = None,
wrap_in_span: bool = False,
wrap_in_span: bool | list[bool] = False,
span_label: str | None = None,
span_attrs: list[dict[str, str]] | None = None,
extra_attrs: list[dict[str, str]] | None = None,
) -> str:
labels = labels or [None] * len(rows)
extra_attrs = extra_attrs or [{} for _ in rows]
span_attrs = span_attrs or [{} for _ in rows]
wraps = wrap_in_span if isinstance(wrap_in_span, list) else [wrap_in_span] * len(rows)
# `labelledby[i]` is the TEXT of a hidden div this row's aria-labelledby points at (not an id) --
# the id is minted from the row's own position so each row gets a distinct label target.
labelledby = labelledby or [None] * len(rows)
@ -16248,7 +16252,9 @@ def _button_listbox_html(
for k, v in extra_attrs[i].items():
attrs += f" {k}={v!r}"
span_attr = f" aria-label={span_label!r}" if span_label else ""
inner = f'<span style="cursor:pointer"{span_attr}>{text}</span>' if wrap_in_span else text
for k, v in span_attrs[i].items():
span_attr += f" {k}={v!r}"
inner = f'<span style="cursor:pointer"{span_attr}>{text}</span>' if wraps[i] else text
return f'<li role="option"{attrs}>{inner}</li>'
li_html = "".join(
@ -16778,6 +16784,95 @@ async def test_select_combobox_refuses_rows_with_crossed_values_across_surfaces(
assert r.status == "error", r.content
def test_lone_duplicate_candidate_refuses_crossed_leaf_and_ancestor_values() -> None:
# The node dimension matters: leaf/ancestor data-code A/B beside A/A share the leaf value but
# disagree on the ancestor's — a flat set would read the second as a subset and collapse.
from skyvern.forge.taskv3.tools import _lone_duplicate_candidate
rows = [
{"n": 1, "text": "Depot", "vals": ["l:data-code=A", "a:data-code=B"]},
{"n": 2, "text": "Depot", "vals": ["l:data-code=A", "a:data-code=A"]},
]
assert _lone_duplicate_candidate(rows) is None
def test_lone_duplicate_candidate_collapses_same_value_at_a_different_depth() -> None:
# A copy carrying the same value on a different node is structural variance, not disagreement:
# only keys present on BOTH rows compare.
from skyvern.forge.taskv3.tools import _lone_duplicate_candidate
rows = [
{"n": 1, "text": "Depot", "vals": ["l:data-code=55"]},
{"n": 2, "text": "Depot", "vals": ["a:data-code=55"]},
]
assert _lone_duplicate_candidate(rows) == 1
@_skip_no_browser
@pytest.mark.asyncio
async def test_select_combobox_commits_duplicate_rows_with_whitespace_drifted_data_value() -> None:
# The primary value surface trims like every other veto surface: data-value "x " beside "x" is
# whitespace drift between copies, not a disagreement.
rows: list[tuple[str, str | None]] = [("Depot", "x "), ("Depot", "x")]
async with _content_page(_duplicate_suggestion_html(rows)) as page:
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "select_combobox").handler({"selector": "#addr", "value": "Depot"})
assert r.status == "ok", r.content
value = await page.eval_on_selector("#addr", "el => el.value")
assert value == "Depot", value
@_skip_no_browser
@pytest.mark.asyncio
async def test_shared_leaf_label_refusal_names_the_disagreeing_ancestor_labels() -> None:
# When a shared leaf action label masks distinct row names, the refusal must surface the names
# that actually disagree, or both rows print identically beside their selectors.
rows: list[tuple[str, str | None]] = [("Depot", None), ("Depot", None)]
async with _content_page(
_button_listbox_html(rows, labels=["Branch A", "Branch B"], wrap_in_span=True, span_label="Choose")
) as page:
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "select_combobox").handler({"selector": "#cc", "value": "Depot"})
assert r.status == "error", r.content
assert "Branch A" in r.content, r.content
assert "Branch B" in r.content, r.content
@_skip_no_browser
@pytest.mark.asyncio
async def test_select_combobox_refuses_rows_declaring_identity_on_disjoint_attributes() -> None:
# Two rows that each declare identity on entirely different surfaces cannot be confirmed the
# same candidate (data-code "A" beside data-key "B" refused on main and must keep refusing).
rows: list[tuple[str, str | None]] = [("Depot", None), ("Depot", None)]
async with _content_page(_duplicate_suggestion_html(rows, attrs=[{"data-code": "A"}, {"data-key": "B"}])) as page:
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "select_combobox").handler({"selector": "#addr", "value": "Depot"})
assert r.status == "error", r.content
def test_lone_duplicate_candidate_refuses_same_attribute_with_disjoint_values_across_depths() -> None:
# The same surface naming disjoint values at different depths is a disagreement even though no
# node+attr key is shared.
from skyvern.forge.taskv3.tools import _lone_duplicate_candidate
rows = [
{"n": 1, "text": "Depot", "vals": ["l:data-code=A"]},
{"n": 2, "text": "Depot", "vals": ["a:data-code=B"]},
]
assert _lone_duplicate_candidate(rows) is None
@_skip_no_browser
@pytest.mark.asyncio
async def test_select_combobox_refuses_option_rows_with_whitespace_distinct_submission_values() -> None:
# The DOM preserves whitespace in option submission values: "x " and "x" are distinct choices.
rows: list[tuple[str, str | None]] = [("Foo", "x "), ("Foo", "x")]
async with _content_page(_duplicate_suggestion_option_html(rows)) as page:
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "select_combobox").handler({"selector": "#addr", "value": "Foo"})
assert r.status == "error", r.content
def _grid_suggestion_html(rows: list[tuple[str, str | None]]) -> str:
# The ARIA grid-combobox pattern: the tagged leaf is the gridcell, and the candidate's identity
# (data-code here) lives on its [role=row] ancestor.
@ -16838,6 +16933,199 @@ async def test_select_combobox_commits_grid_duplicate_rows_sharing_the_row_code(
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "select_combobox").handler({"selector": "#addr", "value": "Depot"})
assert r.status == "ok", r.content
@_skip_no_browser
@pytest.mark.asyncio
async def test_select_combobox_refuses_disjoint_identifiers_despite_shared_generic_attribute() -> None:
# A generic attribute agreeing across rows (a shared tooltip) is ordinary markup, not identity
# evidence: rows declaring identity on disjoint attributes must keep refusing even when a title
# happens to agree on both.
rows: list[tuple[str, str | None]] = [("Depot", None), ("Depot", None)]
async with _content_page(
_duplicate_suggestion_html(
rows, attrs=[{"data-code": "A", "title": "Choose"}, {"data-key": "B", "title": "Choose"}]
)
) as page:
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "select_combobox").handler({"selector": "#addr", "value": "Depot"})
assert r.status == "error", r.content
committed = await page.eval_on_selector("#addr", "el => el.getAttribute('data-committed')")
assert committed is None, committed
def test_lone_duplicate_candidate_refuses_disjoint_identifiers_with_shared_generic_attribute() -> None:
# Any agreeing shared attribute must not stand in for the disjoint identifiers beside it: the
# flattened pair sets do not nest, so there is no affirmative evidence of one candidate.
from skyvern.forge.taskv3.tools import _lone_duplicate_candidate
rows = [
{"n": 1, "text": "Depot", "vals": ["a:data-code=A", "a:title=Choose"]},
{"n": 2, "text": "Depot", "vals": ["a:data-key=B", "a:title=Choose"]},
]
assert _lone_duplicate_candidate(rows) is None
def test_lone_duplicate_candidate_refuses_agreement_on_one_identifier_beside_disjoint_declarations() -> None:
# Deliberately conservative: rows agreeing on their one shared attribute while each declaring
# other, disjoint surfaces still refuse — a false collapse silently commits the wrong row, while
# a refusal only hands the pick back to the caller.
from skyvern.forge.taskv3.tools import _lone_duplicate_candidate
rows = [
{"n": 1, "text": "Depot", "vals": ["a:data-code=A", "a:title=T"]},
{"n": 2, "text": "Depot", "vals": ["a:data-code=A", "l:name=N"]},
]
assert _lone_duplicate_candidate(rows) is None
@_skip_no_browser
@pytest.mark.asyncio
async def test_select_combobox_commits_equal_renders_tagged_at_different_depths() -> None:
# One candidate rendered as a wrapped leaf beside a flattened copy: the tagger lands on the span
# in one row and on the row element itself in the other. Tagging depth is structural variance,
# not identity — the same DOM must read the same way and collapse.
rows: list[tuple[str, str | None]] = [("Depot", None), ("Depot", None)]
async with _content_page(
_button_listbox_html(
rows,
wrap_in_span=[True, False],
extra_attrs=[{"title": "Depot, Main St"}, {"title": "Depot, Main St"}],
span_attrs=[{"title": "Depot"}, {}],
)
) as page:
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "select_combobox").handler({"selector": "#cc", "value": "Depot"})
assert r.status == "ok", r.content
value = await page.eval_on_selector("#cc-value", "el => el.value")
assert value == "Depot", value
@_skip_no_browser
@pytest.mark.asyncio
async def test_select_combobox_refuses_leaf_values_disagreeing_under_agreeing_row_values() -> None:
# The mirror of the crossed-depth case: leaves declare A beside B while both rows declare Z. A
# depth-blind read would let the agreeing row value shadow the leaf disagreement and collapse
# two distinct rows.
rows: list[tuple[str, str | None]] = [("Depot", None), ("Depot", None)]
async with _content_page(
_button_listbox_html(
rows,
wrap_in_span=True,
extra_attrs=[{"data-code": "Z"}, {"data-code": "Z"}],
span_attrs=[{"data-code": "A"}, {"data-code": "B"}],
)
) as page:
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "select_combobox").handler({"selector": "#cc", "value": "Depot"})
assert r.status == "error", r.content
value = await page.eval_on_selector("#cc-value", "el => el.value")
assert value == "", value
@_skip_no_browser
@pytest.mark.asyncio
async def test_select_combobox_commits_duplicate_rows_carrying_the_same_attribute_at_both_depths() -> None:
# The accept direction of the two-depth read: twin renders agreeing at the leaf AND the row.
rows: list[tuple[str, str | None]] = [("Depot", None), ("Depot", None)]
async with _content_page(
_button_listbox_html(
rows,
wrap_in_span=True,
extra_attrs=[{"data-code": "Z"}, {"data-code": "Z"}],
span_attrs=[{"data-code": "A"}, {"data-code": "A"}],
)
) as page:
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "select_combobox").handler({"selector": "#cc", "value": "Depot"})
assert r.status == "ok", r.content
value = await page.eval_on_selector("#cc-value", "el => el.value")
assert value == "Depot", value
@_skip_no_browser
@pytest.mark.asyncio
async def test_select_combobox_refuses_rows_whose_distinguishing_value_sits_past_twelve_entries() -> None:
# The value list must never truncate ahead of the disagreeing entry: nine allowlisted leaf
# attributes plus the row's own push the rows' disagreeing data-code to position 13, which a
# 12-entry cap discarded — both rows then read identical and the first was committed.
nine = {
"value": "v",
"data-value": "v",
"data-val": "v",
"data-v": "v",
"data-code": "C",
"data-key": "k",
"data-option-value": "o",
"name": "n",
"title": "t",
}
rows: list[tuple[str, str | None]] = [("Depot", None), ("Depot", None)]
async with _content_page(
_button_listbox_html(
rows,
wrap_in_span=True,
span_attrs=[dict(nine), dict(nine)],
extra_attrs=[
{"title": "t", "name": "n", "value": "v", "data-code": "ROW-A"},
{"title": "t", "name": "n", "value": "v", "data-code": "ROW-B"},
],
)
) as page:
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "select_combobox").handler({"selector": "#cc", "value": "Depot"})
assert r.status == "error", r.content
value = await page.eval_on_selector("#cc-value", "el => el.value")
assert value == "", value
@_skip_no_browser
@pytest.mark.asyncio
async def test_select_combobox_refuses_rows_with_whitespace_distinct_value_attribute() -> None:
# A `value` attribute is a submission value like OPTION.value, not authoring metadata: "x " and
# "x" are two choices, so byte-exact comparison covers it and only data-value gets the drift trim.
rows: list[tuple[str, str | None]] = [("Depot", None), ("Depot", None)]
async with _content_page(_duplicate_suggestion_html(rows, attrs=[{"value": "x "}, {"value": "x"}])) as page:
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "select_combobox").handler({"selector": "#addr", "value": "Depot"})
assert r.status == "error", r.content
@_skip_no_browser
@pytest.mark.asyncio
async def test_identical_text_refusal_omits_row_labels_that_agree_across_rows() -> None:
# The row-label clause exists to surface the surface that DISAGREES; when every row wears the
# same ancestor name the clause distinguishes nothing and must not print at all.
rows: list[tuple[str, str | None]] = [("Depot", "v1"), ("Depot", "v2")]
async with _content_page(
_button_listbox_html(rows, labels=["Same Branch", "Same Branch"], wrap_in_span=True, span_label="Choose")
) as page:
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "select_combobox").handler({"selector": "#cc", "value": "Depot"})
assert r.status == "error", r.content
assert "row label" not in r.content, r.content
@_skip_no_browser
@pytest.mark.asyncio
async def test_identical_text_refusal_leads_with_the_value_that_disagrees() -> None:
# Agreeing generic values must not crowd the disagreeing identity out of the capped suffix —
# the caller is being asked to pick between the rows, so what differs prints first.
rows: list[tuple[str, str | None]] = [("Depot", None), ("Depot", None)]
async with _content_page(
_duplicate_suggestion_html(
rows,
attrs=[
{"title": "T", "name": "N", "data-val": "V", "data-code": "AAA"},
{"title": "T", "name": "N", "data-val": "V", "data-code": "BBB"},
],
)
) as page:
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "select_combobox").handler({"selector": "#addr", "value": "Depot"})
assert r.status == "error", r.content
assert "AAA" in r.content, r.content
assert "BBB" in r.content, r.content
value = await page.eval_on_selector("#addr", "el => el.value")
assert value == "Depot", value
@ -20849,3 +21137,58 @@ async def test_semantic_verify_kill_switch_keeps_commits_working(monkeypatch: py
{"selector": "#city", "value": "Springfield, Sangamon, IL"}
)
assert picked.status == "ok", picked.content
@_skip_no_browser
@pytest.mark.asyncio
async def test_observe_discloses_that_a_populated_page_was_dropped_whole_by_the_visibility_gates() -> None:
# An app shell that keeps its content in the DOM behind a boot gate renders `(0 interactive
# elements)` -- byte-identical to a genuinely empty page, because the visibility gates drop
# silently. That ambiguity is what makes a model poll wait->observe for turns on end: nothing in
# the payload says the page is full and unreadable rather than still loading. v1 is equally blind
# here (measured), so the gates stay; only the disclosure is new.
html = (
"<!doctype html><html><head><title>Portal</title>"
"<style>#shell { visibility: hidden; }</style></head><body>"
'<div id="shell">'
'<nav><a href="#/home">Home</a><a href="#/logout">Logout</a></nav>'
"<h1>Document Portal</h1><p>Welcome back.</p>"
'<button id="newdoc">New Document</button>'
'<input id="q" type="text" placeholder="Search documents">'
'<div role="status">Signed in successfully</div>'
"</div>"
# Outside the hidden shell, so it is the off-canvas gate that drops this one and not the
# visibility gate: the count then discriminates both gates rather than only the second.
'<button id="ghost" style="position:absolute;left:-9999px">Ghost</button>'
"</body></html>"
)
async with _content_page(html) as page:
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "observe").handler({})
assert r.status == "ok", r.content
assert "(0 interactive elements)" in r.content, r.content
assert "note: the page has 5 control(s) that are present but not visible" in r.content, r.content
assert r.data is not None and r.data["summary"]["hidden_dropped"] == 5
@_skip_no_browser
@pytest.mark.asyncio
async def test_observe_stays_silent_about_hidden_chrome_when_it_can_still_see_the_page() -> None:
# The note is scoped to total blindness on purpose: site chrome is routinely present-but-hidden
# (closed menus, inactive tabs), so a note on every observe would be noise on the one channel
# whose cost is ~linear in the prefix. A page the model can act on says nothing.
html = (
"<!doctype html><html><head><title>Portal</title></head><body>"
'<button id="go">Go</button>'
'<div style="display:none"><button id="hidden-menu">Archive</button></div>'
"</body></html>"
)
async with _content_page(html) as page:
tools = build_browser_tools(_fixed_page_provider(page))
r = await _tool(tools, "observe").handler({})
assert r.status == "ok", r.content
assert "(1 interactive elements)" in r.content, r.content
assert "present but not visible" not in r.content, r.content
assert r.data is not None and r.data["summary"]["hidden_dropped"] == 1