Skyvern/skyvern/forge/taskv3/loop.py

5094 lines
284 KiB
Python

"""Faithful Task V3 agent tool-loop.
A single persistent LLM conversation drives browser tools via native tool-calling:
the model emits ``tool_calls``, we execute them, thread the results back as ``tool``
messages, and repeat until the model calls a terminal tool (``finish``) or a budget
cap is hit. Perception is a tool the model chooses to call — nothing about the page
is injected automatically — which is what distinguishes this from the step engine's
scrape-every-step loop.
The loop itself is transport-agnostic: it depends only on an ``LLMCaller``-shaped
object and a list of ``ToolSpec``. Browser wiring lives in a separate module so this
core can be unit-tested with scripted fakes.
"""
from __future__ import annotations
import asyncio
import hashlib
import json
import re
import secrets
import time
from collections import deque
from contextvars import ContextVar
from dataclasses import dataclass, field
from enum import Enum
from typing import Any, Awaitable, Callable, Collection, Iterator, Literal, NamedTuple, Sequence, TypeVar
from urllib.parse import urlsplit
import structlog
from skyvern.exceptions import SkyvernContextWindowExceededError
from skyvern.forge.sdk.core import skyvern_context
from skyvern.forge.sdk.core.skyvern_context import SkyvernContext
from skyvern.forge.sdk.workflow.context_manager import RANDOM_SECRET_ID_PREFIX
from skyvern.forge.taskv3.goal_check import (
GoalCheckAction,
GoalVerdict,
NonCompletedStatus,
ToolTrail,
TrailEntry,
UnlistedReask,
)
from skyvern.forge.taskv3.handoff_redaction import MAX_HANDOFF_URL_CHARS, sanitize_published_url
from skyvern.forge.taskv3.target_label import describe_target
from skyvern.webeye.navigation import clear_task_nav_error_code, redact_url_secrets
LOG = structlog.get_logger()
ToolStatus = Literal["ok", "error"]
FinishStatus = Literal["completed", "failed", "terminated"]
UnlistedReaskCheck = Callable[[NonCompletedStatus, str], Awaitable[UnlistedReask]]
# Why a failing tool call failed, as one closed vocabulary. Spelled as a `Literal` rather than `str`
# because the whole value of the facet is that it is closed: it is written at ~33 literal sites in
# tools.py, and a typo or a near-synonym added later would split a cohort silently. Enforcement is
# invocation-shaped: `mypy.ini` sets `follow_imports = skip`, so the annotation binds in tools.py only
# while THIS module is in the same mypy run. CI's `pre-commit run mypy --all-files` includes it and
# rejects an undeclared value; checking tools.py alone does not. The AST source census in
# tests/unit/test_taskv3_loop.py backstops the runs that miss it, and it can only read a value spelled
# as a literal at the write site.
ToolErrorClass = Literal[
# The address did not resolve to what it named.
"stale_selector",
"invalid_selector",
"invalid_mark",
"ambiguous_selector",
"ambiguous_frame",
"stale_ref",
"ref_not_in_latest",
"stale_mark",
"mark_not_in_latest",
# The target resolved, but the page will not let the act happen.
"disabled",
"not_editable",
# `type`: the page replaced the typed text with a non-empty value of its own; left in place.
"value_changed_by_page",
# `type`: the field does not hold the typed text afterwards -- an append that is partial or unchanged, or a
# one-character-per-box code field whose boxes did not all keep their character.
"text_not_held",
"covered",
"inert",
"unreachable",
# `file_upload`: the target is not a file input, holds no single one, and opened no file picker;
# or it is one whose click would submit its form, so it was not clicked.
"no_file_input",
"submits_form",
# `file_upload`: the input could not be read back and no upload request was seen.
"attach_unconfirmed",
# The field resolved and the page cooperated, but the requested VALUE named no single option.
# Each of these names only what was READ: an absence claim holds solely over a list read in full,
# which is why a declared-but-truncated list gets `rows_unread` rather than `no_matching_row`.
"ambiguous_rows",
"identical_rows",
"no_matching_row",
"rows_unread",
# The field opened, but no option list rendered to read at all.
"no_option_list",
# A read asked to resume past the end of what it was reading. Not an address failure and not a
# page refusal: the call was well-formed and the page cooperated, the offset simply named nothing.
"offset_past_end",
# The offset itself was unusable — negative, or not a whole number of characters.
"invalid_offset",
# `navigate`: nothing committed — a net error, or the commit budget expired with no response.
"navigation_failed",
# The value was a credential's one-time-code field, and no code could be produced for this call.
"no_one_time_code_source",
# A one-time-code placeholder was combined with other text, or named no credential field.
"unusable_one_time_code_value",
# The handler raised instead of returning; classified by `_raised_error_class`.
"driver_timeout",
"timeout_other",
"handler_raised",
# An erroring call whose construction site named no class. Deliberately a value rather than an
# absence, so "errored, unnamed" is countable and cannot be confused with "did not error".
"other",
]
# The `ok` counterpart to `ToolErrorClass`, for a tool whose success status spans outcomes that are
# not the same event. `tool_status` is the only total outcome field on the record, so a tool that
# returns `ok` for "did the thing", "there was nothing to do", and "declined to try" is indexed
# identically on all three and a blind detector reads as a working one. Naming the branch is
# telemetry, never a behaviour change: like `error_class` this is never serialized into the tool
# message, so the model cannot see it (the loop sends `content` only).
ToolOkClass = Literal[
# `solve_captcha`. Its three `ok` branches are genuinely different events and the `ok` for
# "absent" is correct -- no challenge present is not an error and must not force a retry.
"solved",
"absent",
"attempts_exhausted",
# `solve_captcha` with `image_selector`: characters were read off the image. Not "solved" -- only the
# page can say whether they were right, after the model types them.
"image_text_read",
# `file_upload`. `attached_no_activity` is the file confirmed on the input with no upload request
# seen -- a form that sends the file with the submit lands here, and so does an unwired handler.
"upload_seen",
"consumed_shown",
"attached_no_activity",
# `navigate`, in descending order of how far the landed document got: `loaded` saw the load event
# fire, `document_ready` returned on domcontentloaded with load still outstanding, and
# `committed_not_loaded` got a document that never became ready inside the readiness budget.
"loaded",
"document_ready",
"committed_not_loaded",
]
# Which `covered` message the model actually got. They are one `tool_error_class`, so without this
# the split is only recoverable by pulling step archives and classifying the prose.
CoveredBranch = Literal[
# The layer was named and its controls enumerated into the message.
"named",
# The layer holds a challenge frame, so the message names it and omits the dismissal sentence.
"challenge",
# The walk qualified nothing, the hit was the field's OWN container, AND an ancestor was found
# clipping the point the click would land on. The message names that container and asks for it to
# be opened or scrolled, because there is no third party here to dismiss. Same real-world
# condition as `error_class="unreachable"` -- union them for a reachability census.
"clipped",
# No reading at all: the probe reported occlusion without naming anything and without locating a
# clip. Four producers, and for some of them "something is on top of it" is exactly right: an
# ancestor drawing its own ::before veil (a busy card), a scrim drawn as `body::before` (which
# hit-tests AS body and so never reaches the naming walk at all), a pinned cover the field
# collapsed under, and a skinned control whose own label covers it.
"unnamed",
# The layer intercepts the pointer but paints nothing, so it is absent from the screenshot.
"invisible",
# The layer is the field's own container and a non-interactive element in it takes the pointer;
# the message names that stamped element to click, so no controls are listed.
"own_container",
]
# WHICH ELEMENT the probe named as the layer, which the branch cannot recover. `named` with zero
# controls spans two different events: a real overlay whose controls the enumeration dropped, and a
# walk that qualified nothing and named the raw hit element -- an option row or a value cell, which
# has no actionable child and nothing to dismiss. Only the probe knows which, so it is on the record.
CoveredLayerKind = Literal[
# The walk found an element that qualifies as a layer: pinned, a layer role, aria-modal, <dialog>,
# or view-sized.
"qualified",
# Nothing qualified, so the probe named the hit element itself.
"hit_fallback",
# `clipped` only: the probe named the ancestor it found clipping the click point. Not a layer --
# nothing is over the control -- but it is the one address the message can offer, so this is the
# split between a `clipped` the model can act on and one that can only say "that container".
"clipper",
# The probe named no element at all.
"unnamed",
]
@dataclass
class ToolResult:
status: ToolStatus
content: str
data: dict[str, Any] | None = None
# Transient images the loop must show the model on the NEXT call only (the on-demand `look`
# tool's annotated screenshot). Threaded into one .call()'s ephemeral screenshots= arg and never
# appended to the transcript, so it costs one image on one turn and is gone the turn after.
screenshots: list[bytes] | None = None
# Telemetry only, and deliberately NOT in `data`: callers and tests pin `data` by equality, so a
# measurement riding in it would change an observable contract. Never shown to the model.
error_class: ToolErrorClass | None = None
# Same contract as `error_class`, on the other side of the status. Telemetry only, never shown
# to the model. Each is read only under its own status, so a result carrying the class for the
# OTHER one drops it silently; the classmethods below cannot express that pairing (neither
# accepts the other's kwarg) and the raw constructor is the only route that could.
ok_class: ToolOkClass | None = None
# The requested action was refused and never dispatched (a ToolRefusal). A field, not a raise, so
# every wrapper around the handler still post-processes the call.
refused: bool = False
# A refused call that still dispatched input to the page first, to release a previous field's
# suggestion list. It is charged and the page may have changed, though the requested action never ran.
touched_page: bool = False
@classmethod
def ok(
cls,
content: str,
data: dict[str, Any] | None = None,
screenshots: list[bytes] | None = None,
*,
ok_class: ToolOkClass | None = None,
) -> ToolResult:
return cls("ok", content, data, screenshots, ok_class=ok_class)
@classmethod
def error(
cls, content: str, data: dict[str, Any] | None = None, *, error_class: ToolErrorClass | None = None
) -> ToolResult:
return cls("error", content, data, error_class=error_class)
ToolHandler = Callable[[dict[str, Any]], Awaitable[ToolResult]]
class ToolRefusal(Exception):
"""Refuse the call before it acts on the page (no input dispatched; a marker attribute written to
address an element is not an action). The loop sends the message alone as a refused error, which
does not spend an action step within the grace, so it must carry nothing the model must not see."""
def __init__(self, message: str, *, error_class: ToolErrorClass, data: dict[str, Any] | None = None) -> None:
super().__init__(message)
self.error_class: ToolErrorClass = error_class
self.data = data
@classmethod
def of(cls, result: ToolResult) -> ToolRefusal:
return cls(result.content, error_class=result.error_class or "other", data=result.data)
def as_result(self) -> ToolResult:
return ToolResult("error", str(self), self.data, error_class=self.error_class, refused=True)
# The tool-result `data` keys the target-name/target-kind capture ride on (written by the browser
# tools, read here). Internal to the loop: only `content` is ever shown to the model, so they cost no
# tokens. The label composition itself -- the floor vocabulary, the shape filter, the secret matcher --
# lives in target_label.py; this module only carries the two raw values from probe to `RoundAction`.
TARGET_LABEL_DATA_KEY = "target_label"
TARGET_KIND_DATA_KEY = "target_kind"
# The tool-result `data` key a recorded action's outcome rides on: the machine facts about what the
# call achieved (`requested_url`, `url`, `http_status`, `page_transitioned`, `navigation_dead_end`),
# carried verbatim to `RoundAction.outcome` for the caller to persist on the action row. Internal to
# the loop like the two keys above -- never shown to the model, which reads the tool's own content.
ACTION_OUTCOME_DATA_KEY = "action_outcome"
# An HTTP status at or above this reached no usable page, so the action row reads as failed even
# though the tool honestly returned ok (the model still gets the status and decides what to do).
ACTION_OUTCOME_FAILED_HTTP_STATUS = 400
# How long a call spent turning an address into a target, before the act. A context variable rather
# than a field on the result, because the cohort this exists to price is the one where the handler
# RAISES -- a driver timeout on a resolved target -- and a result the handler never returned cannot
# carry anything. Set before the handler runs, so the value survives whichever way the call ends.
# Accumulated, because the mark wrapper resolves an address and the ref wrapper then runs inside it.
#
# WHAT ONE ROW MEANS, written here because a chart of this by `selector_kind` reads the field and
# cannot see the branch that produced it:
# absent no address was supplied. Its only meaning -- never "measured, and it was zero".
# ref / mark the server-side table lookup and the frame routing it implies: the persistent-ref
# model's own addressing cost, which is what this field exists to price.
# css the frame ROUTING only: which realm a typed selector resolves in.
# Working-page acquisition is excluded on every branch: every design has to get the page, so it is
# not a cost of the addressing model.
_RESOLVE_SECONDS: ContextVar[float | None] = ContextVar("taskv3_resolve_seconds", default=None)
# Which case behind the click reach-probe's boolean applied, for the one click call this context covers.
# A context variable rather than a result field because `click` returns from many places, several of
# them after the probe has already answered -- the same reason `_RESOLVE_SECONDS` lives here.
_HIT_CLASS: ContextVar[dict[str, Any] | None] = ContextVar("taskv3_hit_class", default=None)
# Which `covered` message the refusal rendered, for the one tool call this context covers. A context
# variable for the same reason `_HIT_CLASS` is one: the covered message is built in a shared helper
# five call sites reach, several of them after the probe has already answered.
_COVERED_LAYER: ContextVar[dict[str, Any] | None] = ContextVar("taskv3_covered_layer", default=None)
# The text-delta read's cost and yield for the one tool call this context covers.
_TEXT_DELTA: ContextVar[tuple[float, int | None, int, bool, str | None, int] | None] = ContextVar(
"taskv3_text_delta", default=None
)
# The ordinal of the tool call being dispatched, so a tool can say how many calls a report spans.
_TOOL_CALL_SEQ: ContextVar[int | None] = ContextVar("taskv3_tool_call_seq", default=None)
def record_resolve_seconds(elapsed: float) -> None:
"""Add `elapsed` to this tool call's address-resolution time. Telemetry only."""
_RESOLVE_SECONDS.set((_RESOLVE_SECONDS.get() or 0.0) + elapsed)
def record_hit_class(
hit_class: str,
*,
needed: bool,
probe_seconds: float | None = None,
isolated: bool | None = None,
raised: bool = False,
) -> None:
"""Record what this click's reach probe returned, the decision it produced, and what it cost.
`needed` is on the record because the class alone cannot recover it: a shadow-rooted target
answers `unknown` with needed=TRUE, so `unknown` would otherwise pool the rows that bought the
second probe with the rows that answered nothing. `disabled` is answered before any hit test runs;
every other `hit_class` value names what the hit test RETURNED, never why — nothing in that probe
tells a real occluder from a hit that fell through, since a modal scrim drawn as `body::before`
hit-tests as body and still blocks the click.
`isolated` says which realm answered -- realms, not costs, since its `False` side spans both the
cheapest fallback and the most expensive retry storm. `raised` marks a probe that threw, whose
duration is a blow-up rather than a cost. The RECORDED copies are inert: the click path branches on
its own `needed`, never on anything read back from here.
"""
_HIT_CLASS.set(
{
"hit_class": hit_class,
"needed": needed,
"probe_seconds": probe_seconds,
"isolated": isolated,
"raised": raised,
}
)
def record_text_delta(
seconds: float,
lines: int | None,
chars: int = 0,
*,
over_bound: bool = False,
skipped: str | None = None,
pending: int = 0,
) -> None:
"""Record the text-delta read. Telemetry only. `lines` is None when the read failed, timed out or
was past the char bound (`over_bound`), or when the tool errored after a read before it ran; else
how many lines the result reported before the display cap. `chars` is the section's length.
`skipped` names why a call did not read at all: "over_bound" (this document was already read past
the bound) or "unreadable" (its reads failed twice in a row).
`pending` is how many lines the page showed before the call ran (read before a repeated call): part
of `lines` when the call's own read succeeded, and still reported when it did not (`lines` None)."""
_TEXT_DELTA.set((seconds, lines, chars, over_bound, skipped, pending))
def current_tool_call_seq() -> int | None:
"""The loop's ordinal for the tool call being dispatched; None outside the loop."""
return _TOOL_CALL_SEQ.get()
def record_covered_layer(branch: CoveredBranch, *, controls: int, layer_kind: CoveredLayerKind) -> None:
"""Record which `covered` message this call rendered. Telemetry only, never a behaviour change.
`controls` is how many controls the MESSAGE named, not how many the probe found: a control with
neither a selector nor a label is dropped from the message, and the existing eight-slot
truncation caps what the model is handed. So this is the count the model acted on. WHAT A
DENOMINATOR MEANS HERE: zero is not one event. Under `unnamed` it means there was no layer to
enumerate; under `clipped` it is zero always, because a container is not a layer and its
controls are never the thing to act on; under `named` it means the layer had no control the
enumeration could name. Cut on `tool_error_class:covered`, then group by `covered_branch`, and
read the count within a branch.
"""
_COVERED_LAYER.set({"branch": branch, "controls": int(controls), "layer_kind": layer_kind})
# What observe() prints and the model hands back, BYTE-IDENTICAL in both directions: the digest
# prints `ref=12` and that exact string is the selector argument, so "copy it as printed" has one
# reading. Deliberately not an attribute-selector shape and deliberately not bracketed -- the ref
# namespace is a server-side table, and accepting a `[ref=12]` form would let a page that authors a
# `ref` attribute collide with it, which is the wrong-element class this addressing exists to close.
# It lives here rather than in tools.py because both modules classify against it and a second copy
# would drift.
REF_SELECTOR_RE = re.compile(r"^\s*ref=(\d+)\s*$")
DriverTimeoutPredicate = Callable[[BaseException], bool]
def _no_driver_here(exc: BaseException) -> bool:
return False
# Whether an exception is the browser driver's OWN timeout. Injected rather than imported, because
# the driver package ships only in the `local`/`server` extras: importing it at this module's scope
# would make the loop unimportable in a base install and would break the scripted-fake unit testing
# the module docstring promises. `tools.py` -- the only module that can build a handler capable of
# raising a driver error -- installs the real predicate as it imports, so the driver cohort cannot
# read empty while a browser tool exists to fill it.
_is_driver_timeout: DriverTimeoutPredicate = _no_driver_here
def set_driver_timeout_predicate(predicate: DriverTimeoutPredicate) -> None:
global _is_driver_timeout
_is_driver_timeout = predicate
def _raised_error_class(exc: BaseException) -> ToolErrorClass:
"""Classify an exception that escaped a tool handler, for the failure-cost read.
A bare `TimeoutError` is NOT evidence of a driver timeout: `file_upload` awaits a source fetch
that raises one, and on 3.11 `asyncio.TimeoutError` IS `builtins.TimeoutError`. Folding those
into the driver cohort would corrupt the very number this field exists to produce.
The driver class is decided by the installed predicate -- `is_driver_timeout_error`, which
recognises BOTH Playwright-family packages -- and never by the exception's own module name: the
browser image rewrites this repository's driver imports to the fork, so a module test would
recognise driver timeouts only where the fork is absent -- i.e. nowhere that matters -- and
report the cohort as empty in production. Nothing here reads the exception's MESSAGE: a message
can carry page text and this field is indexed.
"""
if _is_driver_timeout(exc):
return "driver_timeout"
if isinstance(exc, TimeoutError):
return "timeout_other"
return "handler_raised"
def mark_is_filler(mark: Any) -> bool:
# look() numbers marks from 1, so 0 addresses nothing: it is an upstream filling the optional slot,
# not the model addressing by mark. `to_openai_tool` sets the `strict` key that provokes that, so
# this is the tolerance for an upstream that fills it anyway.
return mark == 0 and not isinstance(mark, str)
def _selector_kind(args: dict[str, Any]) -> str:
"""How the model addressed its target on this call, read off the ARGS AS SENT.
Must be taken before dispatch: the ref and act-by-mark wrappers rewrite `args["selector"]` in
place, so the same read afterwards reports the resolved address rather than the one the model
chose. It is the cut for `resolve_seconds`, but the two cohorts are NOT symmetric and the
contract above `_RESOLVE_SECONDS` says how: a `css` row bounds frame routing only.
"""
if args.get("mark") is not None and not mark_is_filler(args.get("mark")):
return "mark"
selector = args.get("selector")
# `not selector`, matching what the wrappers themselves treat as absent. Note this does NOT make
# the record self-consistent: `selector_present` is read after dispatch, and act-by-mark leaves
# the selector it resolved in `args`, so a mark call logs kind=mark WITH selector_present=True.
# The two fields describe different moments on purpose -- read `selector_kind` for what the model
# sent.
if not isinstance(selector, str) or not selector:
return "none"
return "ref" if REF_SELECTOR_RE.match(selector) else "css"
class RoundAction(NamedTuple):
"""One dispatched page action, as the caller persists it."""
tool: str
args: dict[str, Any]
succeeded: bool
# The target's page-visible name, read off the element before the action ran. None when the tool
# names no element, when the target has no readable name, or when the probe could not run.
target_name: str | None = None
# The target's role/type, from a fixed vocabulary (see target_label.py) -- computed the same time
# as target_name and independent of whether a name was found.
target_kind: str | None = None
# Whether this action consumed a budget unit. Carried rather than re-derived from the tool name
# downstream: the loop already read it off the ToolSpec, and a second name list is a second place
# for a new tool to be missing from.
billable: bool = False
# What the action achieved, as the tool itself reported it (see ACTION_OUTCOME_DATA_KEY): machine
# facts only, for the caller to persist alongside the verb. None when the tool reported none --
# a call that errored before it reached a page has only its `error` below to offer.
outcome: dict[str, Any] | None = None
# The tool's own error text when the call failed, which is all a call that never reached a page
# can say about itself -- a refusal the engine issued on purpose reads as an unexplained failed
# row without it. Model-facing prose, so a caller that persists it must redact and cap it.
error: str | None = None
# A probe consulted after a billable/download-signaling tool result; a truthy return ends the run as
# completed with that reason, without the model ever calling finish. A blocker consulted from
# finish(completed) itself; a truthy return rejects that verdict with the message as the reason. Both
# receive the basenames tools staged into the downloads dir this run, to exclude from detection.
CompletionProbe = Callable[[frozenset[str]], Awaitable[str | None]]
CompletionBlocker = Callable[[frozenset[str]], Awaitable[str | None]]
# Consulted from finish for EVERY verdict, unlike CompletionBlocker: it takes the finish status and
# gates on state the caller already tracks (e.g. a verification-code budget), not on staged
# downloads. One callback, because a completed claim made on a blank verification step and a
# non-complete verdict given up with polling budget still unspent are the same concern seen from two
# sides; splitting them into two hooks would scatter it.
VerificationBlocker = Callable[[str], Awaitable[str | None]]
# The status a re-ask conversion asks the verification gate about: a completion the model never claimed.
CONVERSION_VERIFICATION_STATUS = "converted"
@dataclass
class ToolSpec:
name: str
description: str
parameters: dict[str, Any]
handler: ToolHandler
terminal: bool = False
billable: bool = False # a page-mutating browser action that meters like a step-engine action
recordable: bool = False # persisted as an action row (with screenshot) but not billed/budgeted
compactable: bool = False # a large perception result safe to elide from the transcript once superseded
# Its result hands out handles (observe refs, look marks) that the tool's NEXT call disposes and renumbers,
# so an older result is not merely older but wrong to act on. Compaction keeps exactly one of these,
# whatever the page showed; a result without handles stays usable and is superseded only by the same content.
issues_handles: bool = False
# Drives the live page without being billed or recorded. Read ONLY as engagement ("did this run
# touch the page at all"), never as budget or artifact parity -- those are `billable` and
# `recordable`, and widening either to cover this would change action rows and the step budget.
# Exists because `open_verification_link` calls page.goto() while carrying neither flag, so an
# engagement predicate built on those two alone calls a run that navigated "never tried".
engages_page: bool = False
# Whether a call's target carries the toggle state observe prints (checked/pressed). Consulted only
# to exempt a click from the post-failure submit skip; set by the browser tools on click alone.
toggle_probe: Callable[[dict[str, Any]], Awaitable[bool]] | None = None
@property
def touches_page(self) -> bool:
return self.billable or self.recordable or self.engages_page
def to_openai_tool(self) -> dict[str, Any]:
return {
"type": "function",
"function": {
"name": self.name,
"description": self.description,
"parameters": self.parameters,
# Stated rather than left unset because OpenRouter reads an unset `strict` as license
# to fill every declared property with a type default (`mark: 0`, `selector: ""`),
# which the act wrappers then refuse. `True` is not the alternative: these schemas
# are not strict-shaped and it 400s. See GOTCHAS.md "Loop" for the measured matrix.
"strict": False,
},
}
@dataclass
class LoopOutcome:
status: Literal["completed", "failed", "terminated", "budget_exhausted", "loop_error", "canceled"]
reason: str
# Which loop guard authored this verdict (`PERCEPTION_STALL_GUARD` and its siblings), or None for a
# terminal the model or a budget produced. This is the machine-readable class that used to be a
# prefix on `reason`; `reason` itself is customer-facing text (see `_guard_verdict`).
guard: str | None = None
extracted_output: Any = None
# The raw budget-cap string (e.g. "max_turns (40) reached") that granted this run its one final
# observed turn, carried on whatever outcome the final turn produces — a finish verdict or, if
# the model didn't finish, the honest budget_exhausted exit. None for a run that never tripped a cap.
cap_trip: str | None = None
turns: int = 0
tool_calls: int = 0
action_steps: int = 0
# Wall-clock spent inside tool handlers, summed over the run. Serial by construction, so it is
# directly comparable against the run's total duration.
tool_seconds: float = 0.0
# Turns where the model answered with prose instead of a tool call, costing a full round trip
# plus the NO_TOOL_CALL_NUDGE recovery turn.
no_tool_call_turns: int = 0
# Whether tool_choice was still being sent when the run ended. Distinguishes a run that was
# asked to force tool calls from one where the request was degraded away mid-run.
tool_choice_in_effect: bool = False
billable_actions: list[str] = field(default_factory=list)
# Perception snapshots are compacted in place during the run, so superseded observe/get_html
# content is already elided here — treat as lossy if ever persisted for audit.
messages: list[dict[str, Any]] = field(default_factory=list)
# The progress signals for the caller's one terminal record, set on every path out of the loop.
# The None default only covers an outcome built by someone else; the caller still writes its
# record and simply carries no progress fields on it.
telemetry: TerminalTelemetry | None = None
goal_check: dict[str, Any] | None = None
# The finish-gate goal check ended this run; its verdict discards the model's output, so the
# final-turn stamp must not refill it.
goal_check_ended: bool = False
# Set when the unlisted-outcome re-ask turned the model's failed/terminated finish into completed; a
# post-loop veto of that completion restores this status and reason instead of failing the task.
converted_from: NonCompletedStatus | None = None
converted_from_reason: str = ""
unlisted_reask: dict[str, Any] | None = None
def _get(obj: Any, key: str, default: Any = None) -> Any:
# raw_response=True returns a model_dump() dict, but test fakes and some
# providers hand back objects — accept either shape.
if isinstance(obj, dict):
return obj.get(key, default)
return getattr(obj, key, default)
def _extract_message(response: Any) -> Any:
choices = _get(response, "choices") or []
if not choices:
return None
return _get(choices[0], "message")
def _extract_text(response: Any) -> str | None:
message = _extract_message(response)
if message is None:
return None
return _get(message, "content")
def _extract_reasoning_summary(response: Any) -> str | None:
"""The provider's reasoning summary (message.reasoning_content), when litellm's chat->responses
bridge returned one -- surfaces the provider's reasoning summary where available; a tool call
with empty message.content on a continuation turn often has none either."""
message = _extract_message(response)
if message is None:
return None
return _get(message, "reasoning_content")
def _extract_tool_calls(response: Any) -> list[tuple[str, str, dict[str, Any]]]:
message = _extract_message(response)
if message is None:
return []
raw_tool_calls = _get(message, "tool_calls") or []
tool_calls: list[tuple[str, str, dict[str, Any]]] = []
for raw in raw_tool_calls:
function = _get(raw, "function") or {}
name = _get(function, "name")
if not name:
continue
tool_call_id = _get(raw, "id") or f"call_{len(tool_calls)}"
arguments = _get(function, "arguments")
if isinstance(arguments, str):
try:
parsed_args = json.loads(arguments) if arguments else {}
except json.JSONDecodeError:
parsed_args = {}
elif isinstance(arguments, dict):
parsed_args = arguments
else:
parsed_args = {}
tool_calls.append((tool_call_id, name, parsed_args))
return tool_calls
NO_TOOL_CALL_NUDGE = (
"You did not call a tool. Call a browser tool to make progress, or call "
"finish(status, reason, extracted_output) if the goal is complete. Emit a tool call now."
)
# Perception-stall policy: N consecutive identical (marker-canonicalized) snapshots from the same (compactable) tool
# AND from the same probe (that tool with the same arguments) mean the page has stopped changing in
# response to actions, and a page gated by something the run cannot perceive or operate otherwise
# burns the whole budget on identical re-observes. The per-probe term is what keeps two different
# probes returning the same string from reading as one frozen page. Only compactable tools count:
# action tools legitimately return the same string every call ("waited"), so they can never witness
# "the page is unchanged".
PERCEPTION_STALL_NUDGE_AFTER = 6
PERCEPTION_STALL_TERMINATE_AFTER = 15
# SKY-17222. Read bodies (tools without handles) the model has already seen are retained by SIZE, not
# count: once they sum past HIGH, the oldest are elided down to LOW in one rewrite (an epoch). HIGH is
# the previous ceiling (two get_html windows, `tools.HTML_MAX_CHARS` each), so the worst case does not
# move. Between epochs the transcript only grows at the end, which is what a provider prefix cache
# needs: it hits only when the whole previous request is a byte prefix of the next one.
PERCEPTION_RETAIN_CHARS_LOW = 20_000
PERCEPTION_RETAIN_CHARS_HIGH = 2 * PERCEPTION_RETAIN_CHARS_LOW
# However large they are, the newest reads the model has seen that an epoch keeps: two paged windows.
PERCEPTION_RETAIN_MIN_READS = 2
# A same-bytes marker replaces a body only when it saves most of it; on a short body it costs more.
PERCEPTION_MARKER_MIN_RATIO = 4
# The stall verdict's machine class. Stable and facetable — telemetry keys on it to measure how often
# the policy fires — and it travels beside the verdict (`LoopOutcome.guard`, the guard's log line, the
# engine's terminal record), never inside its text: a verdict's `reason` is persisted verbatim as the
# customer-facing `task.failure_reason`, so a dashboard facet in it is a facet the customer reads
# (SKY-16271). Change it only with the dashboards that read it.
PERCEPTION_STALL_GUARD = "perception_stall"
# Action-loop policy: N repeated executions of the same billable action (same tool + same args)
# with no new evidence the page changed mean the run is re-trying against an unchanged outcome —
# the live shape is re-submitting into the same rejection banner, which the stall policy cannot
# see because interleaved actions and varied probes keep the perception stream changing while the
# SITUATION stays the same. Evidence of change is a REPEATED probe (same tool + same args)
# returning different content, or a download landing; a first-time probe has no baseline and is
# evidence of nothing, so varied-selector probing cannot launder repetition into "progress".
ACTION_LOOP_NUDGE_AFTER = 3
# 8, not 6: RAW repeats of one action key peak at 6 across 50 completed prod runs while stuck runs
# reach 36, so 6 sat on the completed population's edge. DO NOT LOWER THIS. The effective
# post-clearing counter — what this constant is actually compared against — was then measured by
# replaying the clearing rule over per-call telemetry, and NO SEPARATION WAS OBSERVED: two completed
# runs peaked at 2 and 4, four stuck runs at 1, 3, 3 and below (upper bounds, computed identically
# on both sides), and the highest observed value fell in a COMPLETED run. n=6 is far too small to
# establish inversion as a property of either population — that spread is also consistent with
# noise — but it is no evidence FOR a lower threshold either, and lowering a safety threshold
# requires positive evidence. 8 is above everything observed, which is why the change is safe; it
# is also why nothing in the sample fires. REVISIT if the effective distribution ever becomes
# measurable at population scale; do not lower it before then. Repeat count may not be a stuck-ness
# signal at all: SKY-15602 tracks the real problem, which is that the loop has no definition of
# goal progress, only of page change.
ACTION_LOOP_TERMINATE_AFTER = 8
# Facetable sibling of PERCEPTION_STALL_GUARD; same dashboard contract.
ACTION_LOOP_GUARD = "action_loop"
# Progress-gated action-step budget extension (SKY-15264): a run that hits its action-step cap while
# the page is still demonstrably changing (a repeated probe returned fresh content, a navigation or
# download landed) earns an extension of half the original cap — the observed failure population is
# genuinely long multi-page forms dying mid-progress, while a stalled run must be refused exactly as
# before. Evidence must be at most this many action rounds old: staler change evidence says nothing
# about the run's current state.
ACTION_BUDGET_EXTENSION_EVIDENCE_WINDOW = 8
# A long form does not stop being long because one extension already fired: 41.4% of long-block
# step-cap deaths die at the ALREADY-EXTENDED cap (SKY-15666). So the grant repeats under the same
# evidence predicate, bounded two ways. Each grant is half the ORIGINAL cap, never half the current
# one: a run that has already spent 1.5x its budget must not draw a proportionally larger handout,
# and constant-size grants keep every extension priced the same and re-earned on fresh evidence.
# Total growth stops at this multiple of the original cap. The wall clock — the one cap that is not
# a function of the action-step budget — is the outer runaway stop and is never re-derived.
ACTION_BUDGET_EXTENSION_MAX_FACTOR = 3
# Refusals left uncharged in a row before a refusal is charged like any failed call. The repeat guard cannot
# bound them (a stale address resets it), so this does.
UNCHARGED_REFUSAL_GRACE = ACTION_LOOP_NUDGE_AFTER
# Facetable event names — both the grant and the refusal are queryable so the gate's decision
# precision is measurable on the canary; change only with the dashboards that read them.
ACTION_BUDGET_EXTENDED_EVENT = "taskv3 loop action budget extended"
ACTION_BUDGET_EXTENSION_REFUSED_EVENT = "taskv3 loop action budget extension refused"
BATCH_SKIP_TOGGLE_EXEMPT_EVENT = "taskv3 loop batch skip exempted toggle click"
CREDENTIAL_RESUBMIT_REFUSED_EVENT = "taskv3 loop credential resubmit refused"
EXTRACTION_ENTRY_REFUSED_EVENT = "taskv3 loop extraction entry refused"
# Every tool that authors input on the page. Defined here rather than in tools.py because tools.py imports
# this module; the extraction-block refusal and tools.py's frame-work ledger both read this one set.
FILL_TOOLS = frozenset({"type", "select_option", "select_combobox", "file_upload"})
# The arguments that carry what a fill tool typed or chose, per the tool schemas in tools.py. Selectors and
# refs are excluded: they name a field, and a field's label on the page is not a value the block entered.
ENTERED_VALUE_ARGS: dict[str, tuple[str, ...]] = {
"type": ("text",),
"select_option": ("value", "label", "values", "labels"),
"select_combobox": ("value", "search"),
}
# A wrap-up turn granted by a guard that a later budget extension raised past its trip; facetable
# so a released latch is distinguishable from one that never fired.
FINAL_TURN_RELEASED_EVENT = "taskv3 loop final turn grant released by budget extension"
# Final-turn grant (budget-exhaustion final turn): a budget cap trip buys one more unconstrained
# model turn instead of ending the run immediately, so a model mid-extraction can still call finish
# with what it has. NOT once per run since SKY-15666: an extension that relieves the latching guard
# releases the latch and re-arms it for a later trip, so this fires up to one more time than the run
# had grants. Facetable — change only with the dashboards that read it.
FINAL_TURN_GRANTED_EVENT = "taskv3 loop final turn granted"
# Page-state stall policy (SKY-15265): rounds of billable batches that left the page fingerprint
# byte-identical, with no page-change flag, mean the run is cycling on a frozen document in shapes
# the per-tool guards cannot see (varied probes never streak; scroll/wait carry no digest at all).
# Completions of the worst-affected canary block finish in <=9 rounds; the nudge re-plans the model
# once before the verdict.
PAGE_STATE_STALL_NUDGE_AFTER = 8
PAGE_STATE_STALL_TERMINATE_AFTER = 12
# The verdict is SHADOW-ONLY for now: the fingerprint is blind to work inside iframes (main-frame
# innerHTML only), so a live termination could kill healthy embedded-widget runs. The nudge ships
# live (benign direction); the shadow event measures the would-terminate precision on the canary,
# and promotion to a live verdict is a separate release decision on that data.
PAGE_STATE_STALL_SHADOW_EVENT = "taskv3 loop page state stall would terminate"
# Facetable sibling of PERCEPTION_STALL_GUARD; reserved for the future live verdict, which owes a
# customer-facing sentence of its own the day it stops being shadow-only.
PAGE_STATE_STALL_GUARD = "page_state_stall"
# Hard "the resource does not exist / is gone" HTTP statuses. A navigation landing on one of these is
# a genuine non-capability dead-end (a dead or removed posting), which v1 routes to `terminated`. Both
# the in-loop `navigate` tool and the pre-loop initial-URL navigation classify against this set. NARROW
# on purpose: auth (401/403), rate-limit (429) and transient server errors (5xx) are recoverable or
# capability failures, not dead-ends, and are left to the model / stay `failed`.
NAVIGATION_DEAD_END_STATUSES = frozenset({404, 410})
class _NavDeadEnd(NamedTuple):
"""A navigation that landed on a hard dead-end status, and the URL it landed on when the tool
reported one."""
status: int
url: str | None
# Defined here (not tools.py) so the batch-dispatch poisoning check below can compare against it
# without an import cycle -- tools.py already imports ToolResult/ToolSpec from this module.
PAGE_UNAVAILABLE_ERROR = "browser page unavailable"
# A navigation landed on a hard dead-end (HTTP 404/410): the target posting does not exist or was
# removed, so the goal cannot be completed there. Ends the run as `terminated`, matching v1's terminate
# verdict for the same condition. Covers both the in-loop `navigate` tool and the pre-loop initial-URL
# navigation. Facetable sibling of the classes above.
NAV_DEAD_END_GUARD = "navigation_dead_end"
# A page-level handler kept asking for the page to be reloaded past the per-run cap: the page cannot
# be stabilized, and acting on it would mean acting on a page declared stale.
PAGE_REFRESH_EXHAUSTED_GUARD = "page_refresh_exhausted"
# A URL long enough to be a payload rather than a place is DROPPED from a verdict, never cut: the
# customer-facing reason is redacted by whole-value match, so a truncated URL would no longer match a
# registered secret it was cut from (the rule `clean_target_label` applies to an element name). The cap
# is the handoff one rather than a second number: both answer "how much URL may leave the run".
VERDICT_URL_MAX_CHARS = MAX_HANDOFF_URL_CHARS
# Emitted, never acted on, when the oscillation rule WOULD have terminated. The step engine's
# tripwires (skyvern/forge/sdk/fail_fast/shadow.py) earn the right to act by publishing this event
# first and deriving a decision precision from it; a rule that ADDS terminations gets the same
# treatment rather than being trusted because it looks right.
PERCEPTION_STALL_SHADOW_EVENT = "taskv3 loop perception stall would fire"
# Emitted once per run, where the argument-blind per-tool counter first reached the terminator and
# the per-probe term did not. It marks a verdict withheld, not a run spared: most such runs end on
# another guard, so read it joined to the final status.
PERCEPTION_STALL_SUPPRESSED_EVENT = "taskv3 loop perception stall suppressed_main_fire"
# Guard-attribution hashes: sha256 over a per-run secret salt, truncated to 64 bits — enough to join a
# firing to the value the guard compared within one run (cardinality ≤ max_tool_calls), while the salt
# keeps a low-entropy input (a short typed text, a selector) non-enumerable, which truncation alone does not.
TELEMETRY_HASH_HEX_LEN = 16
def telemetry_hash(salt: str, *parts: str) -> str:
return hashlib.sha256("\x1f".join((salt, *parts)).encode()).hexdigest()[:TELEMETRY_HASH_HEX_LEN]
# The value shape observe()'s enrichment mints ('t' + monotonic counter, optional '-<n>'
# disambiguator — tools._OBSERVE_JS): identity handles, not page semantics, and a node-replacing
# framework re-mints them on every read, so hashed raw they hide a frozen page from the stall
# guard. Page-authored data-tv3 values (any other shape) are page content and stay significant,
# as is a page-authored data-tv3-ref: observe addresses by a server-held ref, never by an
# attribute, so nothing in the markup with that name is a handle this engine minted.
_TV3_MARKER_VALUE_RE = re.compile(r'data-tv3="t\d+(?:-\d+)?"')
_TV3_PICK_VALUE_RE = re.compile(r'data-tv3-pick="[^"]*"')
# get_html truncates to a fixed budget before the loop ever sees the content, so a marker the cut
# leaves open at the tail has no closing quote for the pattern above and its churning digits would
# be the one leak that survives canonicalization. The lookahead assumes the truncation notice itself
# carries no quote character, and this sub must run AFTER closed markers are rewritten to the
# quote-bearing placeholder — either broken silently brings the leak back.
_TV3_MARKER_CUT_RE = re.compile(r'(?:data-tv3="t\d*(?:-\d*)?|data-tv3-pick="[a-f0-9]*)(?=[^"]*\Z)')
# A read can now start at an offset, so a marker can be cut open at the HEAD of a window too. Same
# leak, same canonicalization, opposite end — but the boundary can land at ANY of the sixteen
# characters of `data-tv3="tN"`, not only inside the digits, so the pattern is every suffix of the
# attribute's fixed prefix (longest first) plus the empty one for a cut inside the value. Anchored at
# the very start of the content, so the only thing it can take is the fragment a window made.
# observe prints its own address as `ref=N` at the head of each element line. The number is
# engine-minted identity, not page semantics: a framework that remounts a control between readings
# gives the replacement a new one, so hashed raw they hide a semantically frozen page from the stall
# guard -- the same reason the marker values above are canonicalized. Anchored at line start, so it
# can only take the address, never a value further along the line; the caller scopes it to
# observe's own payload so page-authored bytes from get_html are never subject to it.
_TV3_REF_ADDRESS_RE = re.compile(r"^ref=\d+", re.MULTILINE)
_PERCEPTION_URL_LINE_RE = re.compile(r"^url=\S+", flags=re.MULTILINE)
def _fold_reported_spans(content: str, spans: Sequence[Sequence[int]] | None, placeholder: str) -> str:
# Spans the TOOL reported, as offsets into `content` as returned. Anything out of range or overlapping
# is not ours to trust and is kept.
pieces: list[str] = []
cursor = 0
for span in sorted(list(s) for s in spans or () if len(s) == 2 and all(type(i) is int for i in s)):
start, end = span
if cursor <= start < end <= len(content):
pieces.append(content[cursor:start])
pieces.append(placeholder)
cursor = end
pieces.append(content[cursor:])
return "".join(pieces)
def _canonical_perception_content(
content: str,
*,
is_observe: bool = False,
head_fragment_len: int = 0,
notice_at: int | None = None,
clip_spans: Sequence[Sequence[int]] | None = None,
delta_at: int | None = None,
delta_end: int | None = None,
) -> str:
# The newly-shown-text section at [delta_at, delta_end) reports what changed since an earlier call,
# not the page's state, so it is cut first; every other reported span lies before it.
if type(delta_at) is int and 0 <= delta_at <= len(content):
end = delta_end if type(delta_end) is int and delta_at <= delta_end <= len(content) else len(content)
content = content[:delta_at] + content[end:]
# observe's clip counts ("…[+N chars]") annotate what the display cut; a page whose clipped tail
# changes has not changed what the model can see, so the count is folded at the offsets observe
# reported. That a field is clipped at all stays: one that now fits whole has changed. First, because
# every other pass here shifts offsets -- which holds only while no tool reports clip spans together
# with notice_at or head_fragment_len.
content = _fold_reported_spans(content, clip_spans, " …[+*]")
# The ref pass is scoped to observe's own payload, not to every compactable result: get_html
# returns page-authored bytes, and a page can write a line that opens `ref=<digits>` there. The
# marker passes below are attribute-shaped and page-authored values in that shape stay significant.
addressed = _TV3_REF_ADDRESS_RE.sub("ref=*", content) if is_observe else content
# Before the marker passes: the notice is what the tail lookahead below scans through, and a
# window's head fragment is what the value pattern cannot close.
# Only for a read the TOOL said it cut, and only the LAST match — which is then provably the
# notice it appended, since our notice comes after all of the window's page content. Folded
# unconditionally this reaches page-authored text that merely LOOKS like a notice, in `observe`
# and `look` as well, and a page whose notice-shaped numbers change then digests identically on
# every read: the stall guard sees frozen and ends a run that was still moving.
# Both folds are applied at boundaries the TOOL reported, never by recognizing a shape. Every
# prefix of the marker attribute is legal page text, and both the page (a forged unterminated
# prefix) and the server (a download filename) can author text wearing the cut notice's shape —
# so no pattern and no match-selection rule can tell ours from theirs. Folding something that is
# not ours makes content that genuinely differs read as frozen, which the perception-stall guard
# TERMINATES on: it ends a run that was still making progress.
#
# The notice first, because its index is into the string as the tool returned it and the head
# fold would shift everything after itself.
noticed = addressed
if notice_at is not None and 0 <= notice_at < len(addressed):
closing = addressed.find("]", notice_at)
if closing != -1:
noticed = addressed[:notice_at] + "…[*]" + addressed[closing + 1 :]
head_folded = '*"' + noticed[head_fragment_len:] if head_fragment_len else noticed
closed = _TV3_MARKER_VALUE_RE.sub(lambda m: m.group(0).partition("=")[0] + '="*"', head_folded)
closed = _TV3_PICK_VALUE_RE.sub('data-tv3-pick="*"', closed)
return _TV3_MARKER_CUT_RE.sub(
lambda m: (
m.group(0)
if notice_at is None and m.group(0).startswith("data-tv3-pick=")
else m.group(0).partition("=")[0] + '="*'
),
closed,
)
def _content_only_perception(
content: str,
*,
is_observe: bool = False,
head_fragment_len: int = 0,
notice_at: int | None = None,
clip_spans: Sequence[Sequence[int]] | None = None,
delta_at: int | None = None,
delta_end: int | None = None,
) -> str:
# The URL is a hint, not content: history.pushState moves it without changing the document. The
# full canonicalization (URL included) keeps clearing the repeat guards — a wizard whose pages
# differ only by URL must survive — but budget-extension evidence hashes THIS, so a URL flip
# alone can never earn budget.
return _PERCEPTION_URL_LINE_RE.sub(
"url=*",
_canonical_perception_content(
content,
is_observe=is_observe,
head_fragment_len=head_fragment_len,
notice_at=notice_at,
clip_spans=clip_spans,
delta_at=delta_at,
delta_end=delta_end,
),
)
# How many recent states a probe remembers. This length IS the longest oscillation period that can
# be recognised, so it is a detection limit and not a memory tuning knob.
PERCEPTION_RING = 8
# Run-scoped revisit memory (SKY-14998). The ring above is a DETECTION LIMIT: a run that cycles with
# a period longer than it returns to states the ring has already evicted, so every ring-bounded guard
# is structurally blind to the loop rather than judging it wrongly. Measured production case: a
# macro-cycle of period ~44 billable touches, against a ring of 8 and a canonical window of 10.
# This remembers perception states for the WHOLE run instead, which is what makes a long cycle
# visible at all. It is a memory, not a verdict: LOG-ONLY, nothing reads it for control flow.
PERCEPTION_REVISIT_EVENT = "taskv3 perception state revisited"
# Two returns to a state is a panel toggling; the third is the earliest a cycle is worth reporting.
PERCEPTION_REVISIT_LOG_AFTER = 3
# Bounded storage: one 64-char digest per DISTINCT perception state, so the worst case is ~32KB of digests per
# run. At the cap the memory stops admitting NEW states and keeps counting the ones it already
# holds — it degrades to a smaller memory rather than growing without bound, and every emission says
# whether it was capped so a truncated run is never read as a complete one.
PERCEPTION_REVISIT_CAP = 512
@dataclass
class _RevisitMemory:
"""How many times each perception state has been seen this run, unbounded in time and bounded in
size. Distinct from _ProbeStreak's ring, which only remembers the last PERCEPTION_RING states and
therefore cannot see a cycle longer than that."""
cap: int = PERCEPTION_REVISIT_CAP
# digest -> (times seen, novel-state count at its last sighting)
_seen: dict[str, tuple[int, int]] = field(default_factory=dict)
capped: bool = False
distinct_states: int = 0
peak_revisits: int = 0
# Revisits that had fresh ground covered since the last sighting, so they were not reported.
# Kept as a count so the healthy population stays measurable in aggregate without a line each.
progressed_revisits: int = 0
_novel_states: int = 0
def record(self, digest: str) -> tuple[int, int]:
"""Returns (times this state has now been seen, distinct NEW states seen since it was last
seen). Zero new states between two sightings is a replay; a drill-down that opens a fresh
page between returns to its list is not, and that difference is what makes the signal
filterable rather than firing on healthy and stuck runs alike.
Novelty is measured against the WHOLE run, not the probe's ring: the ring is bounded at
PERCEPTION_RING, so a cycle longer than it reads as fresh content there — the exact
blindness this memory exists to lift, and reusing that derivation would reintroduce it.
"""
entry = self._seen.get(digest)
if entry is None:
if len(self._seen) >= self.cap:
# Refused, not evicted: dropping a known state to admit a new one would reset a
# streak that is the whole point of the memory.
self.capped = True
return 0, 0
self._novel_states += 1
self._seen[digest] = (1, self._novel_states)
self.distinct_states = len(self._seen)
return 1, 0
seen, novel_at_last_sighting = entry
seen += 1
new_states_since = self._novel_states - novel_at_last_sighting
self._seen[digest] = (seen, self._novel_states)
self.peak_revisits = max(self.peak_revisits, seen)
if new_states_since:
self.progressed_revisits += 1
return seen, new_states_since
@dataclass
class _ProbeStreak:
history: deque[str]
# Consecutive snapshots identical to the previous one; a match means two reads, so it opens at 2.
identical: int = 0
# Consecutive snapshots matching ANY state still in the ring (a superset of ``identical``).
revisits: int = 0
@dataclass
class _Snapshot:
"""Streak readings after one snapshot; the verdict reads ``live``, the warning ``tool_identical``."""
# Per-tool consecutive-identical count, argument-blind: the counter this guard shipped with.
tool_identical: int
probe_identical: int
probe_revisits: int
# A repeated probe returned different content: in-loop evidence that the page changed.
progressed: bool
@property
def live(self) -> int:
return min(self.tool_identical, self.probe_identical)
class _PerceptionLedger:
"""No-progress streaks of perception snapshots, kept per tool AND per probe (tool, canonical args).
The loop acts only where BOTH agree: the tool has returned the same string N times running and
the same probe has too. The per-probe term is what stops distinct probes that happen to return
the same string (a form full of not-yet-chosen dropdowns) from reading as one frozen page; the
per-tool term is what keeps firing a subset of what the argument-blind counter fired, so a
frozen page interleaved with a live sibling probe — which content alone cannot tell from a live
page whose static region is re-read — is never terminated where it was not before.
Streaks are scoped to the probe that produced them. Progress in one probe says nothing about
another: a sibling that ticks every read (a clock, a log tail) must not keep a frozen page
alive to the budget cap, which is exactly what clearing on any probe's progress would do.
Counts per RESULT, not per round: the thresholds are denominated in snapshots, and a turn that
batches five identical probes has taken five of them.
``revisits`` additionally treats a return to any state still in the ring as no progress, which
is what catches a control toggling open and shut. That is NEW firing, so it is only reported.
"""
def __init__(self) -> None:
self._probes: dict[tuple[str, str], _ProbeStreak] = {}
self._tools: dict[str, tuple[str, int]] = {}
self.content_only: dict[tuple[str, str], deque[str]] = {}
self.shadow_reported = False
self.suppressed_reported = False
# Run-level, deliberately NOT cleared by reset(): the worst revisit streak the run ever
# reached, and how many reads could have produced one. A first look at a probe key cannot
# revisit anything, so it is not a chance — counting it would make a run of one-shot probes
# read as "looked and found no looping".
self.peak_probe_revisits = 0
self.revisit_chances = 0
def reset(self) -> None:
"""Forget every streak: the document they described is gone (a reload)."""
self._probes.clear()
self._tools.clear()
self.content_only.clear()
def first_time(self, key: tuple[str, str]) -> bool:
return key not in self._probes
def record(self, key: tuple[str, str], content_digest: str) -> _Snapshot:
tool_name = key[0]
prev = self._tools.get(tool_name)
tool_identical = prev[1] + 1 if prev is not None and prev[0] == content_digest else 1
self._tools[tool_name] = (content_digest, tool_identical)
probe = self._probes.get(key)
if probe is None:
self._probes[key] = _ProbeStreak(deque([content_digest], maxlen=PERCEPTION_RING))
return _Snapshot(tool_identical, 0, 0, False)
self.revisit_chances += 1
progressed = content_digest != probe.history[-1]
if progressed:
probe.identical = 0
else:
probe.identical = probe.identical + 1 if probe.identical else 2
if content_digest in probe.history:
probe.revisits = probe.revisits + 1 if probe.revisits else 2
else:
probe.revisits = 0
probe.history.append(content_digest)
self.peak_probe_revisits = max(self.peak_probe_revisits, probe.revisits)
return _Snapshot(tool_identical, probe.identical, probe.revisits, progressed)
def next_snapshot_can_trip(self, threshold: int) -> bool:
"""Whether ONE more read of some probe, returning what it last returned, reaches ``threshold``
live. This is the trip's exact precondition, not an estimate of it: the next read continues
the tool counter only if that content is also the tool's last content, so a probe that went
dormant while its tool moved on to other content cannot be the one that trips."""
for (tool_name, _), probe in self._probes.items():
tool_last_content, tool_identical = self._tools[tool_name]
if probe.history[-1] != tool_last_content:
continue
next_probe_identical = probe.identical + 1 if probe.identical else 2
if min(tool_identical + 1, next_probe_identical) >= threshold:
return True
return False
# Net-progress ledger (SKY-15020 Lever C). The repetition guards above ask whether the page or
# action REPEATED; this asks whether the run made NET PROGRESS. Real stuck-ness is often VARIED
# actions with zero net progress (SKY-14998: 21 input-timeouts on different selectors — varied
# actions, varied perception — so no repetition guard trips and the run oscillates to the cap). The
# ledger is SHADOW-ONLY and ADDITIVE: it emits a "would fail-fast" event but terminates nothing and
# leaves the three guards untouched. Adding live terminations is a release-posture change gated on
# this event's own decision precision plus operator sign-off.
PROGRESS_LEDGER_WINDOW = 8
# Facetable event names; the offline precision/survival metrics key on these — change only with the
# dashboards that read them.
# Canonical progress signal, Phase A (SKY-15379): ONE target-churn + compound-progress tracker,
# LOG-ONLY. The loop answers "is this run progressing?" five separate ways today; this tracker is
# the candidate replacement signal, emitting the parity/precision data any consolidation needs
# while changing no behavior. Its target key is the canonicalized SELECTOR, not tool+args — the
# incumbent action-loop key provably missed prod repeat-loops whose args varied on a fixed target.
CANONICAL_LOOP_EVENT = "taskv3 canonical progress loop detected"
CANONICAL_EXTEND_DELTA_EVENT = "taskv3 canonical progress extension delta"
# Log-only thresholds; the Phase-B intervention thresholds are tuned FROM this data, so the fire
# points are logged at every rung of the distribution rather than one cliff.
CANONICAL_LOOP_WINDOW = 10
CANONICAL_LOOP_FIRE_COUNTS = (3, 4, 6, 8)
# The rung that counts as "looping" for the extension-delta read — named so reordering the logging
# rungs above can never silently move the threshold Phase B tunes against.
CANONICAL_LOOP_TOUCHES = 4
PROGRESS_LEDGER_SHADOW_EVENT = "taskv3 loop progress ledger would fail-fast"
@dataclass
class _ProgressLedger:
"""Billable actions since the last net-progress signal, plus its per-run peak.
The already-computed distance-to-done metric is the observe summary's ``invalid_fields`` count.
Net progress is any of: a navigation or download landing (hard progress), the count reaching a
NEW LOW (a real form advance), or the count RISING since the last look (a fresh page's required
fields, or a submit that surfaced new errors — either way the context changed, so the prior
no-progress streak is stale). Novelty is deliberately NOT progress: varied thrash produces novel
perception, which is why the streak is counted against the invalid-field trend, not the page
digest.
Two invariants keep the over-termination direction safe:
- The verdict is taken on an OBSERVE that CONFIRMS no progress, never on the raw action count: a
run that batches several fixes before re-observing (markers stay valid until the page
re-renders) has not yet shown a stalled look, so the streak withholds rather than fires.
- ``form_armed`` reflects the CURRENT look, not a sticky earlier one, so a form-less page (a
confirmation/extraction page, or a solved form) can never be judged stuck here.
The design is intentionally biased toward PRECISION over recall — a shadow verdict headed for a
future live terminator must not fire on a progressing run. A real page-transition signal (the click
tool's ``page_transitioned``, from its already-computed url_before/url_after) closes ONE precision
gap the rise heuristic left: a click that moves the URL re-baselines cleanly as hard progress, so a
fresh page whose ``invalid_fields`` count coincidentally equals the prior page's no longer reads as a
stalled look. Only the POSITIVE direction is used — a URL change proves a transition — because URL
equality does NOT prove same-page (a URL-stable SPA multi-step form advances without moving the URL),
so the ledger never suppresses a re-baseline on an unchanged URL. That leaves known false-NEGATIVES
(the SAFE direction) the shadow numbers under-count: a run whose ``invalid_fields`` OSCILLATES or
CREEPS still re-baselines on every up-swing (the rise heuristic cannot separate same-page thrash from
a real advance without a stronger same-page oracle — the Lever C recall follow-up), and a form
silently rejected while showing zero invalid fields never arms.
Reads only what the loop already computed — no extra LLM turn, no re-observe.
"""
window: int = PROGRESS_LEDGER_WINDOW
actions_since_progress: int = 0
peak_actions_since_progress: int = 0
invalid_baseline: int | None = None # floor for the current page/context; rebased on rise or new low
last_invalid: int | None = None
form_armed: bool = False
ever_armed: bool = False # a form was seen at least once — the runs the survival record is emitted for
shadow_reported: bool = False
def observe(self, invalid_fields: int) -> bool:
"""Record an observation; return True the first time one CONFIRMS window-length no-progress."""
prev = self.last_invalid
self.last_invalid = invalid_fields
# Reflects THIS look only: a page with no invalid fields must not be judged stuck here.
self.form_armed = invalid_fields > 0
self.ever_armed = self.ever_armed or self.form_armed
if self.invalid_baseline is None:
self.invalid_baseline = invalid_fields
return False
if invalid_fields < self.invalid_baseline:
# A new low: real net progress on the form. Reset the streak and re-baseline.
self.actions_since_progress = 0
self.invalid_baseline = invalid_fields
return False
if prev is not None and invalid_fields > prev:
# The count rose since the last look: a new page's fresh required fields, or a submit that
# surfaced new errors. Re-baseline to the new floor instead of measuring it against a
# stale, lower one. A real click-driven transition instead arrives as hard_progress()
# upstream (baseline cleared); this rise path is the fallback for same-page count changes
# and for transitions no tool witnessed.
self.actions_since_progress = 0
self.invalid_baseline = invalid_fields
return False
# Flat, with no new low and no rise: this look confirms no net progress since the last one.
if self.actions_since_progress >= self.window and self.form_armed and not self.shadow_reported:
self.shadow_reported = True
return True
return False
def hard_progress(self) -> None:
# Navigation or a download landing: unambiguous progress. The old page's distance metric is
# stale, so drop the baseline; the next observe re-arms from the new page.
self.actions_since_progress = 0
self.form_armed = False
self.invalid_baseline = None
def on_billable(self) -> None:
"""Count one billable action toward the no-progress streak (the verdict is taken on observe)."""
self.actions_since_progress += 1
self.peak_actions_since_progress = max(self.peak_actions_since_progress, self.actions_since_progress)
# Failure-evidence gate: a finish(failed) issued shortly after a submit-class action or a
# solve_captcha attempt is held for ONE evidence turn, because submissions and captcha protocols
# complete asynchronously — the sampled false-negative verdicts fired 2-7s after the model's last
# look while the page went on to show the submission confirmation. Trigger tools are the ones whose
# page effects can land after their tool result; the window is in loop turns so intervening
# perception does NOT disarm it (the state can flip after the last observe while a protocol is in
# flight). The true verdict-to-flip latency is unmeasured in the sampled replays: the quiescence
# wait exits on the first stable fingerprint pair (so honest gated failures pay ~one sample), the
# cap only bounds a still-mutating page, and the effective evidence window is dominated by the
# deferral round-trip itself (one LLM turn + the observe).
# Completed-side settle deferrals. 0 disables that gate while leaving the failure-evidence gate
# (which shares the fingerprint sampler) intact.
DEFAULT_MAX_SETTLE_DEFERRALS = 2
FAILURE_EVIDENCE_WINDOW_TURNS = 5
FAILURE_EVIDENCE_SETTLE_MAX_SECONDS = 8.0
# A deferral needs room for its corrected cycle; without it the gate would convert an honest
# failure verdict into budget_exhausted (a budget cap landing mid-deferral). The worst-case cycle
# is wait + observe + re-finish — the deferral message invites an optional brief wait — so both
# the turn and tool-call reservations are 3, the latter read from a per-call refreshed counter.
# The deadline headroom must additionally fund the settle cap plus the cycle's LLM round trips.
FAILURE_EVIDENCE_MIN_DEADLINE_HEADROOM_SECONDS = 60.0
FAILURE_EVIDENCE_MIN_TOOL_CALLS = 3
FAILURE_EVIDENCE_MIN_TURNS = 3
# A verification give-up deferral asks for a BLOCKING poll slice (auth_tools caps one at 120s) on top
# of that cycle, so the 60s above would let the gate convert an honest failure into the
# budget-exhausted end it exists to prevent. Not imported from auth_tools: that module imports this one.
VERIFICATION_GIVEUP_MIN_DEADLINE_HEADROOM_SECONDS = 180.0
def _is_enter_submit(tool_name: str, args: dict[str, Any]) -> bool:
"""An Enter keypress, or a type or select_combobox that pressed Enter: the submit shapes with no selector to key on."""
if tool_name == "press_key":
raw = str(args.get("key", ""))
# Space on a focused button activates it exactly like a click; a literal " " would strip to "".
if raw == " ":
return True
# Playwright accepts Modifier+Key chords, and Control/Meta+Enter is a real submit.
key = raw.strip().lower().rsplit("+", 1)[-1]
return key in ("enter", "return", "numpadenter", "space")
if tool_name in ("type", "select_combobox"):
return bool(args.get("press_enter"))
return False
def _call_selector(args: dict[str, Any]) -> str | None:
selector = args.get("selector")
if isinstance(selector, str) and selector:
return selector
mark = args.get("mark")
return f"mark={mark}" if isinstance(mark, (int, str)) else None
_PAGE_PROBE_TIMEOUT_SECONDS = 10.0
_T = TypeVar("_T")
async def _bounded_probe(probe: Awaitable[_T], timeout: float | None = None) -> _T:
"""Every awaited page.evaluate probe goes through here: the loop's deadline and cancellation checks
run BETWEEN awaits, so a hung renderer can only be interrupted by a bound on the await itself.
A timeout raises (asyncio.TimeoutError) and each call site's raise handling decides what a
missing reading means there. ``timeout`` defaults to the module-level cap read at CALL time (not
baked in as a parameter default) so tests can still monkeypatch _PAGE_PROBE_TIMEOUT_SECONDS."""
return await asyncio.wait_for(probe, timeout=_PAGE_PROBE_TIMEOUT_SECONDS if timeout is None else timeout)
async def _sample_probe(probe: Callable[[], Awaitable[str | None]], deadline_at: float | None = None) -> str | None:
"""A raising or hung probe is as uninformative as a None one: all mean "no reading," not "unchanged."
Generic over what it samples -- the page_probe (document identity) and page_fingerprint (rendered
content) callables share this exact shape. ``deadline_at`` caps the wait to whatever's left of the
loop's own deadline (never longer than the default), so a hung probe cannot outlive it; a deadline
already passed skips the probe call entirely and reads as a missing sample, same as a timeout."""
timeout = _PAGE_PROBE_TIMEOUT_SECONDS
if deadline_at is not None:
remaining = deadline_at - time.monotonic()
if remaining <= 0:
return None
timeout = min(timeout, remaining)
try:
return await _bounded_probe(probe(), timeout=timeout)
except Exception:
return None
# The code tool drives the page through Playwright directly, so the loop sees one tool call where an
# arbitrary number of clicks and submits may have happened. Named here, beside the detector, because
# the detector is what has to know: it is the one fact about that tool the submit guards need, and a
# guard that learns about a caller from a list somebody remembered to update is the shape of the bug
# this closes.
CODE_TOOL_NAME = "execute_python"
def _may_submit(tool_name: str, args: dict[str, Any]) -> bool:
"""A click, an Enter press, or a type or select_combobox that pressed Enter: the loop cannot tell a submit from any of them.
The code tool counts too: its body is opaque to the loop, so it is treated as possibly having
submitted rather than as certainly not having.
"""
return tool_name in ("click", CODE_TOOL_NAME) or _is_enter_submit(tool_name, args)
# Secret values reach the model only as `placeholder_...` tokens, so the guard below keys on the
# token the model passed back and never on what it resolves to — it cannot handle a plaintext secret.
# The prefix covers every registered secret, not only a login: AWS and Azure parameters and card
# fields are minted the same way, which is why the refusal names a rejected value rather than a
# failed sign-in. This matches greedily where the registry matches longest-known-key, so it shares
# find_embedded_placeholder_tokens' boundary limitation and adds one: two tokens concatenated with no
# separator merge into one, and a token the model reproduces with a different trailing fragment reads
# as a different key and goes uncounted. Both fail toward under-counting. It cannot call the registry,
# because loop.py must not import `app`; thread a token callable from agent.py if either is observed.
_CREDENTIAL_PLACEHOLDER_RE = re.compile(re.escape(RANDOM_SECRET_ID_PREFIX) + r"[A-Za-z0-9_]+")
# file_upload is excluded: its placeholder names a file, not a credential, and re-attaching a file
# after a submit is ordinary recovery rather than a second authentication attempt. The code tool is
# included for the reason _may_submit includes it — its body is opaque, so it is treated as possibly
# having entered the credential rather than as certainly not.
CREDENTIAL_ENTRY_TOOLS = (FILL_TOOLS | frozenset({CODE_TOOL_NAME})) - frozenset({"file_upload"})
# How many times one credential may be submitted within a task, and the whole knob. Replaying the
# rule over every v3 block in a 22h window: budget 1 refuses 38 blocks of which 23 COMPLETED, so it
# would break more runs than it protects; budget 3 reaches the bottom of the range the observed
# lockouts came from (3-5 submissions). 2 is the largest budget that stays under the observed harm
# threshold. It is chosen against today's submit precision — _may_submit counts any click, so a
# click that submits nothing still spends budget — and must be re-derived if that precision improves.
CREDENTIAL_SUBMIT_BUDGET = 2
# A login identifier spends no lockout allowance, but identifier-first flows can send a code or hit lookup
# throttles on every submit, so it gets a higher budget rather than none — 5 is an unmeasured judgment
# with no replay behind it, unlike CREDENTIAL_SUBMIT_BUDGET.
LOGIN_IDENTIFIER_SUBMIT_BUDGET = 5
def _credential_placeholders(args: dict[str, Any]) -> set[str]:
"""Every credential placeholder token this call's arguments carry, across all of them.
Scanning the values rather than naming `text`/`value`/`chosen` keeps the guard from being a
list of argument conventions that a new fill tool can be added without. Nested because the
arguments are the model's JSON: `select_option` carries `values`/`labels` as arrays, so a
top-level-strings-only scan would let a credential through in a list.
"""
def walk(value: Any) -> Iterator[str]:
if isinstance(value, str):
yield from _CREDENTIAL_PLACEHOLDER_RE.findall(value)
elif isinstance(value, dict):
for nested in value.values():
yield from walk(nested)
elif isinstance(value, (list, tuple)):
for nested in value:
yield from walk(nested)
return set(walk(args))
def _trail_url_before(result: ToolResult, ctx: SkyvernContext | None) -> str | None:
url_before = (result.data or {}).get("url_before")
if not isinstance(url_before, str) or not url_before:
return None
return ctx.hide_from_model(url_before) if ctx is not None else url_before
def entered_values(tool_name: str, args: dict[str, Any]) -> tuple[str, ...]:
values: list[str] = []
for key in ENTERED_VALUE_ARGS.get(tool_name, ()):
value = args.get(key)
if isinstance(value, str):
values.append(value)
elif isinstance(value, list):
values.extend(item for item in value if isinstance(item, str))
return tuple(values)
# Bounds the one probe a would-be-skipped click pays; a probe that overruns keeps the skip.
TOGGLE_PROBE_TIMEOUT_SECONDS = 2.0
async def _is_toggle_click(tool_name: str, spec: ToolSpec | None, args: dict[str, Any]) -> bool:
"""A click whose target exposes a toggle state chooses an answer rather than submitting. Fails closed."""
if tool_name != "click" or spec is None or spec.toggle_probe is None:
return False
try:
is_toggle = await asyncio.wait_for(spec.toggle_probe(args), timeout=TOGGLE_PROBE_TIMEOUT_SECONDS)
except Exception:
LOG.info("taskv3 toggle probe failed; keeping the batch skip", exc_info=True)
return False
if is_toggle is True:
LOG.info(BATCH_SKIP_TOGGLE_EXEMPT_EVENT, tool=tool_name)
return True
return False
def _is_finish(tool_name: str) -> bool:
return tool_name == "finish"
def _outcome_reports_failure(outcome: dict[str, Any] | None) -> bool:
"""Whether a tool that returned ok nonetheless reported reaching no usable page. Read off the
machine facts the tool exposed (an HTTP status), so the tool never has to adjudicate its own
success -- and only the persisted row moves: the model still reads the tool's own ok result."""
if not outcome:
return False
status = outcome.get("http_status")
return isinstance(status, int) and status >= ACTION_OUTCOME_FAILED_HTTP_STATUS
def _arms_failure_evidence(tool_name: str, args: dict[str, Any], ok: bool) -> bool:
"""solve_captcha arms on ANY dispatch — its "not solved" error is exactly the verdict the async
protocol can contradict. Other actions arm only when they reached the page AND in their
submit-shaped form."""
if tool_name == "solve_captcha":
return True
if not ok:
return False
return _may_submit(tool_name, args)
@dataclass
class ActivityRecency:
"""Written by the tool loop each turn/action, read by the finish tool's holds and the engine's summary line.
Most fields are recency/headroom signals sampled per turn. `action_attempts` and `perceptions`
are the exception: run-lifetime cumulative counters, never reset.
"""
turn: int = 0
turns_remaining: int | None = None
tool_calls_remaining: int | None = None
tokens_remaining: int | None = None
last_turn_tokens: int = 0
last_trigger_turn: int | None = None
# True while one more read of some probe, returning what it last returned, would trip the stall
# terminator: a deferral-forced observe must never be the snapshot that trips it. KNOWN LIMIT: a
# run that reaches the edge and then stops reading that tool altogether leaves this true for the
# rest of the run, disabling the deferral (same as the argument-blind counter it shipped with).
perception_stall_imminent: bool = False
# True from the moment the loop's granted final turn starts running. A hold asks for a retry
# turn that no longer exists — honoring it would silently convert the model's verdict into
# budget_exhausted, the exact conversion the hold headroom gates exist to prevent.
final_turn_active: bool = False
# Attempts, not successes: every `touches_page` call (billable, recordable or `engages_page`) whatever
# its result, plus each pre-dispatch refusal (mark-stale, extraction entry, credential resubmit).
action_attempts: int = 0
# Successful perceptions (the compactable tools: observe / get_html / look).
perceptions: int = 0
# State at the FIRST failed/terminated finish past the failure-evidence gate, emitted once per run
# by the engine's summary line. Not the run's first verdict: a zero-action `completed` refused by a
# gate above reaches this point later as failed/terminated.
attempts_at_hold_gate: int | None = None
perceptions_at_hold_gate: int | None = None
status_at_hold_gate: str | None = None
# Raised only by the goal-check enforce hold and consumed later in the SAME batch, whose queued calls
# predate the hold. Named `_batch_skip` because `test_canonical_ring_state_is_touched_only_through_the_tracker`
# greps this module for the canonical ring's field-name fragments.
held_verdict_batch_skip: bool = False
def armed(self, window: int = FAILURE_EVIDENCE_WINDOW_TURNS) -> bool:
return self.last_trigger_turn is not None and (self.turn - self.last_trigger_turn) <= window
def _names_submit_control(tool_name: str, args: dict[str, Any], ok: bool) -> str | None:
"""The selector of a control a successful action acted on directly, or None.
Only `click`. An Enter press and a type-that-pressed-Enter submit through a control they do not
name, and a captcha dispatch names none at all, so for all three "is that control still in
flight" has no subject — and the selector they do carry is a text field, whose value is the
model's own typed text."""
if not ok or tool_name != "click":
return None
selector = args.get("selector")
return selector if isinstance(selector, str) and selector else None
def _has_hold_headroom(
activity: ActivityRecency | None,
deadline_at: float | None,
min_deadline_headroom_seconds: float = FAILURE_EVIDENCE_MIN_DEADLINE_HEADROOM_SECONDS,
) -> bool:
"""Whether a deferral has the budget to buy the re-verification turn it asks for.
Without it the run ends budget_exhausted, which is unmapped and lands on failed -- turning an
honest hold into the false failure it exists to avoid. Every axis the failure-evidence gate
reserves, for the same reason: a token exhaustion and a stall streak one short of its terminator
each convert the deferral into a verdict the gate did not choose."""
if activity is not None:
if activity.final_turn_active:
return False
if activity.turns_remaining is not None and activity.turns_remaining < FAILURE_EVIDENCE_MIN_TURNS:
return False
if (
activity.tool_calls_remaining is not None
and activity.tool_calls_remaining < FAILURE_EVIDENCE_MIN_TOOL_CALLS
):
return False
if activity.tokens_remaining is not None and activity.tokens_remaining < FAILURE_EVIDENCE_MIN_TURNS * max(
activity.last_turn_tokens, 1
):
return False
if activity.perception_stall_imminent:
return False
if deadline_at is not None and deadline_at - time.monotonic() < min_deadline_headroom_seconds:
return False
return True
def _cap_trip_relieved(
cap_trip: str | None,
max_tokens: int | None,
total_tokens: int,
final_turn_token_reserve: int,
max_turns: int,
turns: int,
max_tool_calls: int,
total_tool_calls: int,
) -> bool:
"""Whether the guard that produced `cap_trip` has since been raised past its trip.
Only the three guards a budget extension re-derives can be relieved. Mirrors the top-of-turn
checks exactly, reserve included, so a released latch cannot re-trip on the very next turn.
The two exclusions differ in kind. The wall clock is never re-derived, and its branch is
unreachable in practice besides: the gate refuses every grant once the clock is blown
("insufficient_deadline_headroom"), so a deadline latch and a grant cannot coexist in one run.
The action-step trip is excluded on purpose and IS reachable — a refusal can latch the wrap-up
turn and the next turn's refreshed evidence can then earn a grant, which the spent-grant exit
discards. That wastes an earned extension and is worth revisiting, but releasing on it is a
behaviour change beyond raising the guards and is deliberately not made here."""
if cap_trip is None:
return False
if cap_trip.startswith("max_tokens"):
return max_tokens is not None and total_tokens < max(0, max_tokens - final_turn_token_reserve)
if cap_trip.startswith("max_turns"):
return turns < max_turns
if cap_trip.startswith("max_tool_calls"):
return total_tool_calls < max_tool_calls
return False
def _budget_extension_gate(
action_steps: int,
last_change_evidence_step: int | None,
action_warned: set[tuple[str, str]],
progress_stalled: bool,
activity: ActivityRecency | None,
deadline_at: float | None,
extension: int,
seconds_per_step: float | None = None,
headroom_gain: tuple[int, int, int] = (0, 0, 0),
) -> tuple[bool, str]:
"""Whether an exhausted action-step budget may be extended, and the deciding reason.
Progress is the property, not its absence-of-stall proxy: the gate requires POSITIVE recent
evidence the page changed (a repeated probe returning fresh content, a navigation or download —
the same events that clear the action-retry ledger), then vetoes on any live stall signal, and
finally requires the turn/token/deadline headroom to actually fund the extension — granting
budget the runaway guards would immediately revoke converts an honest exhaustion into a worse
one. `headroom_gain` is the extra (turns, tool calls, tokens) that granting would itself create
by re-deriving the runaway backstops from the extended cap, and is added to what the run has
left: the question is whether the extension is fundable AFTER the grant, so judging it on
pre-grant headroom would refuse extensions the grant pays for.
Which of the three checks stay live depends on that gain. Under the standard sizing policy the
turn and tool-call gains exceed anything one extension can consume, so those two become
satisfied by construction — correctly, since a guard re-derived from the new cap provably
cannot revoke it — and they still bind for a caller passing no `backstops_for_cap` (the
default) or its own guards. TOKENS are the live discriminator: the gain per step is that
guard's own per-step allowance, so a run burning more than that per turn — the re-read-the-page
spiral the backstop exists for — still fails, and so does one whose gain is zero because the
guard has clamped at its ceiling.
The checks are a funding FLOOR sized off the run's own observed burn (turns per step so far,
last turn's tokens), not a guarantee the extension completes — the facetable grant/refusal
events measure that on the canary."""
extra_turns, extra_tool_calls, extra_tokens = headroom_gain
if (
last_change_evidence_step is None
or action_steps - last_change_evidence_step > ACTION_BUDGET_EXTENSION_EVIDENCE_WINDOW
):
return False, "no_recent_page_change_evidence"
if action_warned:
return False, "warned_action_retry_streak"
if progress_stalled:
return False, "no_net_progress_window"
if activity is not None and activity.turns_remaining is not None:
turns_remaining = activity.turns_remaining + extra_turns
# Fractional comparison (cross-multiplied): floor division would read 1.9 observed
# turns-per-step as 1 and fund an extension the remaining turns cannot run.
if turns_remaining < extension or turns_remaining * max(action_steps, 1) < extension * activity.turn:
return False, "insufficient_turn_headroom"
if activity is not None and activity.tool_calls_remaining is not None:
# +1 reserves the terminal finish call: funding only the actions trades one budget
# exhaustion for another at the very last call.
if activity.tool_calls_remaining + extra_tool_calls < extension + 1:
return False, "insufficient_tool_call_headroom"
if (
activity is not None
and activity.tokens_remaining is not None
and activity.tokens_remaining + extra_tokens < extension * max(activity.last_turn_tokens, 1)
):
return False, "insufficient_token_headroom"
if activity is not None and activity.perception_stall_imminent:
return False, "perception_stall_imminent"
if deadline_at is not None:
# Fund the extension in wall-clock at the run's own observed pace, never below the flat
# minimum the deferral gates use.
required = FAILURE_EVIDENCE_MIN_DEADLINE_HEADROOM_SECONDS
if seconds_per_step is not None:
required = max(required, extension * seconds_per_step)
if deadline_at - time.monotonic() < required:
return False, "insufficient_deadline_headroom"
return True, "recent_page_change_evidence"
@dataclass
class SubmitWatch:
"""The control a click last acted on, and whether a completed verdict has already been held once
against it.
Deliberately NOT part of `ActivityRecency`. That record arms failure evidence, where a broad
trigger (any click, any captcha dispatch) and a decaying turn window are both correct. Neither is
correct here: a captcha dispatch carries no selector and would erase the control, and a turn
window expires while the run is doing the waiting this gate asked for. There is no window — the
probe is the arbiter, because a control that resolves and still reads as in flight IS the
question, where elapsed turns are only a proxy for it."""
selector: str | None = None
deferred: bool = False
def record(self, selector: str) -> None:
self.selector = selector
self.deferred = False
def clear(self) -> None:
self.selector = None
self.deferred = False
@dataclass
class SemanticCommitStats:
"""Tier-1 semantic commit reads and how many accepted. One instance per run, shared by every
tool call, so a per-run accept rate is a division and not a join."""
opportunities: int = 0
accepts: int = 0
def log_fields(self) -> dict[str, int]:
return {"semantic_commit_opportunities": self.opportunities, "semantic_commit_accepts": self.accepts}
@dataclass
class LedgerTerminalFields:
"""The ledger's end-of-run numbers, present only for a run that ever armed a form."""
peak_actions_since_progress: int
actions_since_progress: int
# The CURRENT look, cleared by progress — not the partition. See `TerminalTelemetry.form_ever_armed`.
form_armed: bool
would_fire: bool
def log_fields(self) -> dict[str, int | bool]:
return {
"peak_actions_since_progress": self.peak_actions_since_progress,
"actions_since_progress": self.actions_since_progress,
"form_armed": self.form_armed,
"would_fire": self.would_fire,
}
@dataclass
class TerminalTelemetry:
"""The run's progress signals, assembled where their state lives and emitted by the caller on the
one terminal record it already writes per run. Every optional member is OMITTED rather than zeroed
when its tier could not have fired, because a 0 would read as "fired nothing" instead of "never
ran" and would drag an offline rate's denominator."""
# Sticky, and the partition. `form_armed` on the same record is the current look and answers a
# different question, so filter populations on this one.
form_ever_armed: bool
survival: dict[str, int]
ledger: LedgerTerminalFields | None = None
peak_page_state_stall_rounds: int | None = None
peak_probe_revisits: int | None = None
semantic_commit: SemanticCommitStats | None = None
perception_reads: dict[str, int] | None = None
def log_fields(self) -> dict[str, Any]:
fields: dict[str, Any] = {"form_ever_armed": self.form_ever_armed, **self.survival}
if self.ledger is not None:
fields.update(self.ledger.log_fields())
if self.peak_page_state_stall_rounds is not None:
fields["peak_page_state_stall_rounds"] = self.peak_page_state_stall_rounds
if self.peak_probe_revisits is not None:
fields["peak_probe_revisits"] = self.peak_probe_revisits
if self.semantic_commit is not None:
fields.update(self.semantic_commit.log_fields())
if self.perception_reads is not None:
fields.update(self.perception_reads)
return fields
def _unblocker_options(available_tools: set[str]) -> list[str]:
options = []
if "solve_captcha" in available_tools:
options.append("if the page may be waiting on a verification widget, call solve_captcha")
if "get_html" in available_tools:
options.append("take ONE targeted get_html look at the region that should be changing")
if "look" in available_tools:
options.append("if you can't tell what's on the page or why an action isn't taking, call look to see it")
options.append("if the goal is already met, call finish(status=completed)")
options.append("if genuinely blocked, call finish(status=terminated) naming the blocker as the reason")
return options
def _page_state_nudge_text(rounds: int) -> str:
return (
f"Your last {rounds} action rounds left the page's rendered content completely unchanged — "
"whatever you are trying is not affecting this page. Stop repeating the current approach: "
"re-plan from a fresh observe, try a genuinely different control or path, or finish honestly "
"with a non-completed status, naming what is blocking you."
)
def _stall_nudge_text(stalled: list[tuple[str, int]], available_tools: set[str]) -> str:
"""One warning naming every stalled perception tool and the unblockers this run actually has —
a model that cannot see the gate won't reach for solve_captcha unless the symptom names it."""
symptoms = "; ".join(f"{name} has returned identical output {count} times in a row" for name, count in stalled)
return (
f"The page is not changing: {symptoms}, despite your actions (transient element-marker ids are "
"ignored when comparing). Do not keep re-observing, "
"waiting, or repeating the same action. Your options: " + "; ".join(_unblocker_options(available_tools)) + "."
)
def _action_target(args: dict[str, Any]) -> str:
return str(args.get("selector") or args.get("url") or args.get("key") or "the same target")
def _action_nudge_text(
repeats: list[tuple[str, dict[str, Any], int]], available_tools: set[str], *, page_moved: bool | None
) -> str:
"""The transcript cannot show the model its own repetition (superseded snapshots are compacted
away), so the warning carries that memory: which action, how many times, and whether the page
fingerprint moved since the streak began. page_moved is None when no pair of samples can answer
that — a missing reading is not evidence of a still page, so neither claim is made."""
symptoms = "; ".join(
f"you have called {name} on {_action_target(args)} {count} times" for name, args, count in repeats
)
read_the_page = "Read the current page before deciding: what you need may already be shown there."
if page_moved:
state = (
f"You are repeating the same action: {symptoms}. The page has changed since before the streak "
f"began, and the repeat attempts land on that same state, so repeating it will not change it "
f"further. {read_the_page}"
)
elif page_moved is None:
state = f"You are repeating the same action: {symptoms}. {read_the_page}"
else:
state = (
f"You are repeating the same action without effect: {symptoms}, and the page state you "
"last observed is unchanged since before the first attempt."
)
return (
f"{state} A message inviting you to retry (e.g. 'please submit again') is not an instruction to "
"loop — at most one retry, then report the outcome honestly. Your options: "
+ "; ".join(_unblocker_options(available_tools))
+ "."
)
def _reload_failed_nudge_text() -> str:
return (
"A page-level handler asked for the page to be reloaded, but the reload failed, so the page may be "
"stale or unresponsive. Re-observe before acting, and do not re-submit a form unless the fresh page "
"shows it was not already submitted."
)
def _refresh_nudge_text() -> str:
"""A page-level handler reloaded the page outside the model's own tool calls, so nothing else
in the transcript tells it the last observation is now stale."""
return (
"The page was refreshed by a page-level handler after the last tool call. Its state may have "
"changed: re-observe before acting, and do not re-submit a form until the fresh page shows it "
"was not already submitted."
)
def _budget_extended_observation(cap: str, recency: ActivityRecency | None) -> str:
"""The retraction of a `_budget_exhausted_observation` whose cap has since been raised.
APPENDED, never popped: `LoopState.reads` is keyed by absolute message index, so deleting the
stale message would silently re-anchor compaction onto the wrong ones. Without this the model
keeps reading "this is the final turn" for the rest of the run and wraps up early — which spends
the extension the release exists to preserve, through the prompt instead of through a counter."""
payload = {
"budget_exhausted": False,
"cap": cap,
"headroom": {
"tool_calls": recency.tool_calls_remaining if recency is not None else None,
"tokens": recency.tokens_remaining if recency is not None else None,
},
}
json_line = json.dumps(payload, separators=(",", ":"))
return (
f"{json_line}\nThat budget was extended: the earlier final-turn notice no longer applies and "
"the run continues. Keep working toward the goal."
)
def _budget_exhausted_observation(cap: str, recency: ActivityRecency | None) -> str:
"""A typed, model-facing fact (not an instruction) that this is the run's last granted turn.
Deliberately silent on what to do next -- finishing with a partial result, retrying, or reporting
failure are all legitimate depending on what the model has."""
payload = {
"budget_exhausted": True,
"cap": cap,
"headroom": {
"tool_calls": recency.tool_calls_remaining if recency is not None else None,
"tokens": recency.tokens_remaining if recency is not None else None,
},
}
json_line = json.dumps(payload, separators=(",", ":"))
return f"{json_line}\nThe run's budget cap has tripped; this is the final turn before the run ends."
def _guard_verdict(guard: str, reason: str) -> LoopOutcome:
"""The one way a loop guard ends a run: the customer-facing sentence in `reason`, the machine class
in `guard`.
They are separate because `reason` leaves the system verbatim as the task's `failure_reason` — a
field on the customer webhook and the run view — so the facet a dashboard groups on, the counter
that tripped, the tool that reported it and the selector it compared all belong on the guard's own
log line instead (SKY-16271). `_budget_exhausted_reason` below holds the same standard for the
budget exits.
"""
return LoopOutcome("terminated", reason, guard=guard)
def _verdict_url(url: Any, caller_known_urls: Collection[str]) -> str | None:
"""A URL fit to name in a customer-facing verdict, or None when nothing can be named.
`sanitize_published_url` owns the invariant this surface rests on -- a verdict publishes only what
the caller already gave us, plus the host -- so the scheme and host are always named and the path
only when the landed URL is one of `caller_known_urls`, elided to `/…` otherwise.
"""
if not isinstance(url, str) or not url.startswith(("http://", "https://")) or any(ch.isspace() for ch in url):
return None
published = sanitize_published_url(url, caller_known_urls)
# Only a caller URL can be this long once the path is elided, and the caller can read it in their
# own task config; cutting it would name a page that does not exist, so it is dropped whole.
if published is not None and len(published) > VERDICT_URL_MAX_CHARS:
return None
return published
def _dead_end_reason(status: int, url: Any, *, page_noun: str, caller_known_urls: Collection[str]) -> str:
"""The HTTP status stays in the sentence: `classify_from_failure_reason` derives NAVIGATION_FAILURE
from it, and it is the one fact separating a page that is gone from a page that is merely wrong."""
where = _verdict_url(url, caller_known_urls)
named = f"{page_noun} ({where})" if where else page_noun
return (
f"{named} returned HTTP {status}: it no longer exists or has been removed, so the task could "
"not be completed there."
)
def _verdict_target(
tool: str,
name: Any,
kind: Any,
label_secret_values: Callable[[], Collection[str]] | None,
ctx: SkyvernContext | None,
) -> str | None:
"""The control a verdict names, or None when the acting call reported neither a name nor a kind.
A page-supplied name is only printable against the run's full drop-check secret set -- its
registered parameters included, unfloored, since a name is dropped whole rather than scrubbed --
and only the caller can see that registry, so it arrives as a resolver read at verdict time. Same
set the persisted timeline label is checked against: a credential-shaped label must not be dropped
from the row and printed here. With no resolver, no context, or a resolver that fails, the verdict
falls back to the kind floor rather than publishing page text it could not check -- a missing
context is a missing HALF of that set (the codes minted this turn), not a reason to skip it.
"""
if not name and not kind:
return None
if label_secret_values is None or ctx is None:
return describe_target(tool, None, kind, ())
try:
secret_values = set(label_secret_values())
except Exception:
LOG.warning("taskv3 loop could not resolve a verdict's drop-check secrets", tool=tool)
return describe_target(tool, None, kind, ())
# Unfloored, and read live: a code minted this turn is not in the resolver's floored view.
secret_values |= set(ctx.runtime_secret_values)
return describe_target(tool, name, kind, secret_values)
def _action_loop_reason(target: str | None) -> str:
"""`target` is the control the run kept acting on, or None when the repeated call reported no name
(a failed call reports none) — then the verdict names no place rather than guessing one."""
where = f" on {target}" if target else ""
return (
f"The run repeated the same action{where} and the page did not change in response, so the task "
"could not make progress — commonly the site rejecting the action and showing the same page again."
)
def _budget_exhausted_reason(cap_trip: str) -> str:
"""Human sentence for a no-finish budget exit. Never includes the raw cap literal (that lives
only in `cap_trip` and the logs) so a status/reason readout doesn't leak an internal counter."""
if "deadline" in cap_trip:
axis = "time"
elif "max_tokens" in cap_trip:
axis = "token"
elif "max_turns" in cap_trip:
axis = "turn"
elif "max_tool_calls" in cap_trip:
axis = "tool-call"
elif "maximum steps" in cap_trip:
axis = "step"
else:
axis = "run"
return f"The run reached its {axis} budget before the model finished; the recorded output may be partial."
_TOOL_CALL_RECORD_FIELDS = frozenset(
{
"tool",
"tool_status",
"duration_seconds",
"result_chars",
"selector_present",
"selector_kind",
"tool_error_class",
"tool_ok_class",
"resolve_seconds",
"billable",
"turn",
"batch_size",
"batch_index",
"action_key_hash",
"snapshot_digest",
"probe_first_time",
"hit_class",
"hit_needed",
"hit_probe_seconds",
"hit_probe_isolated",
"hit_probe_raised",
"covered_branch",
"covered_controls",
"covered_layer_kind",
"requested_url",
"landed_url",
"nav_error_code",
"same_page",
"readiness_read_failed",
"menu_note",
"menu_rows",
"withhold_reason",
"text_delta_seconds",
"text_delta_lines",
"text_delta_chars",
"text_delta_over_bound",
"text_delta_skipped",
"text_delta_pending_lines",
}
)
# The counts observe's `summary` carries onto its record (see `_observe_summary_fields`).
OBSERVE_SUMMARY_FIELDS = frozenset(
{
"text_dropped",
"hidden_listed",
"hidden_dropped",
"hidden_dropped_off_canvas",
"hidden_dropped_visibility",
"hidden_dropped_zero_rect",
"hidden_dropped_off_viewport",
"off_viewport_unreachable_unnamed",
"off_viewport_unnamed_host_exempt",
"phantom_dropped",
"iframes_in_component_roots",
"undiscovered_roots",
"omitted_unnameable",
"invalid_fields",
"markers_minted",
"markers_reused",
"group_texts_found",
"a11y_removed_listed",
"duplicate_digest_lines",
"frames_same_origin",
"frames_cross_origin",
"frames_same_origin_interactive",
"frames_peeked",
"frames_peek_failed",
"frame_scan_failed",
"frame_unreadable_regions",
"elements_listed",
"pointer_roots_listed",
"pointer_capped",
"pointer_truncated",
"pointer_dropped",
"pointer_scan_stopped",
"pointer_scan_failed",
"elements_truncated",
"elements_truncated_in_components",
"elements_dropped",
"elements_truncated_by_page_cap",
}
)
# Every field a "taskv3 tool call finished" record can carry. A grader filtering these records checks its keys
# against this set, so a misspelled or renamed field fails loudly instead of matching nothing.
TOOL_CALL_RECORD_FIELD_NAMES = _TOOL_CALL_RECORD_FIELDS | OBSERVE_SUMMARY_FIELDS | {"charged"}
# A host is bounded in the DNS but not in a string `urlsplit` was handed, and a record field is
# indexed: cap the scrubbed value rather than trust what it was reduced from.
LOGGED_URL_MAX_CHARS = 500
# A hostname is letters (any script, so an IDN still logs its host), digits, dots and hyphens, or an
# IPv6 literal, which `hostname` hands back with the brackets stripped. Anything else in what
# `urlsplit` called the host means it found no host at all, whatever it returned: `urlsplit` does not
# treat a backslash as a path separator, so `https://host\signin\TOKEN` parses the whole run as the
# authority while the browser normalizes it to `/` and navigates to a path — the token would survive
# the scrub as part of the "host" and land in an indexed field.
_LOGGABLE_HOST = re.compile(r"[\w.\-]+|[0-9A-Fa-f.:]+")
class _ProgressEvidence(str, Enum):
"""Why the canonical ring was cleared. Every clear names the positive evidence class that
justified it — an "assume changed" fallback (a missing sample read charitably) may never clear."""
REFRESH_RELOAD = "refresh_reload"
FRESH_DOWNLOAD_OR_NAVIGATION = "fresh_download_or_navigation"
PAGE_TRANSITIONED = "page_transitioned"
INVALID_FIELDS_BASELINE_MOVE = "invalid_fields_baseline_move"
PERCEPTION_DIGEST = "perception_digest"
CROSS_BATCH_MOVEMENT = "cross_batch_movement"
PROBE_MISMATCH = "probe_mismatch"
PAGE_STATE_VERDICT = "page_state_verdict"
TERMINAL_BATCH_FINGERPRINT = "terminal_batch_fingerprint"
@dataclass
class _CanonicalProgressTracker:
"""Touches on canonicalized targets since the last compound-progress event (SKY-15379, Phase A).
A touch is one billable tool dispatch, keyed by its selector (tool name when selector-less).
Progress — any of the ledger's hard-progress events, an invalid-fields new-low, or a confirmed
page change — clears the ring, so a repeat streak can only accumulate across a genuinely
unchanged situation; a progressing run is untrippable by construction. Log-only: the counts
feed telemetry, nothing reads them for control flow.
"""
window: int = CANONICAL_LOOP_WINDOW
_touches: list[tuple[str, bool]] = field(default_factory=list)
# Bumped on every clear: a pending loop event minted under an older generation was completed
# by (or followed by) absorbed progress and must not be emitted.
gen: int = 0
# Rungs already fired this generation. Once the window saturates, same-target counts can PARK
# on a fire rung while churn evicts other keys — a rung is one threshold crossing, not one
# event per touch spent at it.
_fired: set[tuple[str, int]] = field(default_factory=set)
# Run-level high-water marks and per-evidence clear tallies, deliberately NOT cleared by
# progress(): the survival record needs the worst churn the run ever reached and every clear it
# ever made, the same way the ledger keeps peak_actions_since_progress across its own clears.
# Clearing these with the ring would report the last streak, not the peak.
peak_same_touches: int = 0
peak_same_errors: int = 0
_evidence_counts: dict[str, int] = field(default_factory=dict)
def record_touch(self, target_key: str, is_error: bool) -> tuple[int, int]:
"""Record one billable touch; returns (same-target touches, same-target errors) in-window."""
self._touches.append((target_key, is_error))
if len(self._touches) > self.window:
self._touches.pop(0)
same = [err for key, err in self._touches if key == target_key]
# A rung re-arms once the in-window count dips below it: a fresh streak after full eviction
# is a genuine re-crossing, distinct from a saturated window PARKED on a fire count.
self._fired = {entry for entry in self._fired if entry[0] != target_key or entry[1] <= len(same)}
same_errors = sum(1 for err in same if err)
self.peak_same_touches = max(self.peak_same_touches, len(same))
self.peak_same_errors = max(self.peak_same_errors, same_errors)
return len(same), same_errors
def progress(self, evidence: _ProgressEvidence) -> None:
"""The one clear choke point: a caller must name its evidence class, so no site can wipe
the ring on a local, untyped notion of progress."""
self._touches.clear()
self._fired.clear()
self.gen += 1
self._evidence_counts[evidence.value] = self._evidence_counts.get(evidence.value, 0) + 1
LOG.debug("taskv3 canonical progress clear", evidence=evidence.value)
def claim_rung(self, target_key: str, count: int) -> bool:
"""A rung is one threshold crossing per generation: the first claim wins, and a repeat
claim while a saturated window PARKS on a fire count is refused."""
if (target_key, count) in self._fired:
return False
self._fired.add((target_key, count))
return True
def invalidate_marks(self) -> None:
"""A look renumbered the manifest, so every mark=N key now names an arbitrary element: drop
them rather than let four different controls alias into one false streak. A surviving mark
streak is therefore always within a single manifest generation. Selector keys stay — their
identity outlives the renumbering."""
self._touches = [touch for touch in self._touches if not touch[0].startswith("mark=")]
self._fired = {entry for entry in self._fired if not entry[0].startswith("mark=")}
def survival_fields(self) -> dict[str, int]:
"""The per-run survival numbers, read through the class for the same reason progress() is
the one write choke point: the ring's state has no references outside this body. Every
evidence class carries an explicit 0, so a class that never fired stays distinguishable
from one this record forgot to emit."""
return {
"peak_same_touches": self.peak_same_touches,
"peak_same_errors": self.peak_same_errors,
"looping_targets": self.looping_targets(),
**{f"clear_{member.value}": self._evidence_counts.get(member.value, 0) for member in _ProgressEvidence},
}
def looping_targets(self) -> int:
counts: dict[str, list[bool]] = {}
for key, err in self._touches:
counts.setdefault(key, []).append(err)
return sum(
1
for errs in counts.values()
if len(errs) >= CANONICAL_LOOP_TOUCHES and sum(errs) >= CANONICAL_LOOP_TOUCHES - 1
)
def _logged_url(url: str) -> str:
"""A URL reduced to what an indexed field may carry: scheme and host, never path, query, userinfo
or fragment.
Unconditional, the model's own argument included. A signed or sign-in URL is a bearer secret
wherever it came from, and the model types back the one a page just showed it, so "the model typed
this" is no evidence the URL is safe to keep whole.
The exception is a payload ref, which is logged as the token it is: a name for the target that is
not the address. Membership in the run's minted refs, never shape — the same rule the model-facing
boundary masks by.
Never raises. The argument is model-typed, and this is evaluated to build the kwargs of the loop's
`taskv3 tool handler raised` line — inside that `except` block — so a raise here escapes the block,
the per-call try and the batch loop, aborting the run instead of producing a tool error.
"""
ctx = skyvern_context.current()
if ctx is not None and url in ctx.opaque_url_refs:
return url[:LOGGED_URL_MAX_CHARS]
try:
host = urlsplit(url).hostname
if host is None or not _LOGGABLE_HOST.fullmatch(host):
return "<redacted>"
return redact_url_secrets(url)[:LOGGED_URL_MAX_CHARS]
except ValueError:
# urlsplit parses the port and the IPv6 brackets lazily, on attribute access: a non-numeric or
# out-of-range port and an unclosed bracket each raise here, not at the split.
return "<redacted>"
def _navigate_record_fields(tool_name: str, args: dict[str, Any], result: ToolResult | None) -> dict[str, str | bool]:
"""What a `navigate` record is about: the URL asked for, where it landed, the driver's code, and
whether the browser was already on the page it was sent to.
Empty for every other tool, so no facet is added fleet-wide. Both URLs are attribution only —
scheme and host — which is what a fleet read of navigate outcomes needs and all a log line may
hold. The requested one is the ARGUMENT, never the resolved one: an opaque payload ref or a
credential placeholder must stay unresolved here.
"""
if tool_name != "navigate":
return {}
fields: dict[str, str | bool] = {}
requested = args.get("url")
if isinstance(requested, str) and requested:
fields["requested_url"] = _logged_url(requested)
data = (result.data if result is not None else None) or {}
landed = data.get("landed_url")
if isinstance(landed, str) and landed:
# Scrubbed again rather than trusted: the tool already reduced it, and the rule belongs to the
# field, not to whichever caller filled it.
fields["landed_url"] = _logged_url(landed)
# A driver's own net:: code, which is a closed vocabulary carrying no address.
nav_error_code = data.get("nav_error_code")
if isinstance(nav_error_code, str) and nav_error_code:
fields["nav_error_code"] = nav_error_code[:LOGGED_URL_MAX_CHARS]
# Whether the requested URL was the one the browser was already showing. Nothing reads it but a
# post-deploy rate: how often the model sends the browser back to the page it is already on.
same_page = data.get("same_page")
if isinstance(same_page, bool):
fields["same_page"] = same_page
# Which fact the `committed_not_loaded` class is about on this row: a document that never became
# ready, or a readyState read that failed on a wedged renderer. Present only on that class, so a
# rate taken over it can exclude probe failures instead of silently mixing them in.
readiness_read_failed = data.get("readiness_read_failed")
if isinstance(readiness_read_failed, bool):
fields["readiness_read_failed"] = readiness_read_failed
return fields
def _menu_note_record_fields(tool_name: str, result: ToolResult | None) -> dict[str, str | int]:
"""Whether a click's menu-open note listed its rows or withheld them, and why, re-checked against the closed
vocabulary. Empty for every other tool and for clicks with no note."""
if tool_name != "click" or result is None:
return {}
data = result.data or {}
fields: dict[str, str | int] = {}
note = data.get("menu_note")
if note in ("listed", "withheld"):
fields["menu_note"] = note
rows = data.get("menu_rows")
if isinstance(rows, int) and not isinstance(rows, bool):
fields["menu_rows"] = rows
reason = data.get("withhold_reason")
if note == "withheld" and reason in ("declared_row_over_caps", "bare_text_beside", "single_row_pieces"):
fields["withhold_reason"] = reason
return fields
def _observe_summary_fields(result: ToolResult) -> dict[str, int]:
"""Counts only: the summary is built by the tool, but an indexed field is re-checked here."""
summary = (result.data or {}).get("summary")
if not isinstance(summary, dict):
return {}
return {
key: value
for key, value in summary.items()
if key not in _TOOL_CALL_RECORD_FIELDS and isinstance(value, int) and not isinstance(value, bool)
}
def _append_skipped_tool_results(
messages: list[dict[str, Any]], remaining: list[tuple[str, str, dict[str, Any]]], reason: str
) -> None:
"""Answer tool_calls we stopped before executing, so every id in the assistant turn has a
matching tool result. An unanswered tool_call is an invalid transcript for the next call."""
for tool_call_id, tool_name, _args in remaining:
messages.append(
{"role": "tool", "tool_call_id": tool_call_id, "name": tool_name, "content": f"skipped: {reason}"}
)
# The `failed` / `terminated` line, stated once and read by every model-facing surface that mentions
# the finish statuses. It is not a new policy: v1's planner and completion-check prompts never offer
# the model `failed` at all -- `TERMINATE` is its only negative verdict, and every v1 `failed` is
# written by the harness itself (an exception, a retry/step ceiling). `TaskStatus.can_update_to`
# encodes the same split: `terminated` is reachable only from `running`, while `failed` can be set on
# a run that never started. So `failed` is the engine's own breakdown and `terminated` is the verdict
# a run reaches about the task -- which is what v3 never told the model (SKY-16648).
# The closing override is v1's own contract: its validation prompt TERMINATEs when a terminate criterion
# holds, so a criterion-bearing goal outranks the negative-answer-is-completed default.
# Wording is load-bearing: two trims that kept the meaning moved 24/40 negative lookups to terminated.
# Kept under ~1000 characters: no other tool this engine ships has a longer description, and the
# OpenAI function schema has historically capped a description at 1024.
FINISH_STATUS_TAXONOMY = (
"Call this only when the goal is met, or when you have established that it cannot be met. The "
"three statuses are not interchangeable:\n"
"- completed: the goal was achieved. A goal that asked you to find out or report something is "
"achieved once you have a definite answer, including a negative one -- report it in "
"extracted_output.\n"
"- terminated: the TASK cannot be achieved, and you can say why from what you saw -- a record "
"you needed does not exist, a lookup returned no match, a rule excludes it, a wall you cannot "
"pass, a site that is down.\n"
"- failed: YOU could not carry the task out, so you cannot say what the outcome is -- your "
"tools kept erroring, you lost track of the page, or you never got far enough to judge.\n"
"The test is whose side the reason is on, not whether you have one: the site or the task means "
"terminated, you or your tools means failed. The goal's own completion or termination criteria "
"override these defaults."
)
def _non_completed(status: object) -> NonCompletedStatus | None:
if status == "failed":
return "failed"
if status == "terminated":
return "terminated"
return None
def make_finish_tool(
page_fingerprint: Callable[[], Awaitable[str | None]] | None = None,
max_settle_deferrals: int = DEFAULT_MAX_SETTLE_DEFERRALS,
should_cancel: Callable[[], Awaitable[bool]] | None = None,
deadline_at: float | None = None,
settle_wait_seconds: float = 0.7,
activity: ActivityRecency | None = None,
max_failure_deferrals: int = 1,
failure_settle_max_seconds: float = FAILURE_EVIDENCE_SETTLE_MAX_SECONDS,
pending_marker: Callable[[str], Awaitable[str | None]] | None = None,
submit_watch: SubmitWatch | None = None,
completion_blocker: CompletionBlocker | None = None,
staged_downloads: set[str] | None = None,
verification_blocker: VerificationBlocker | None = None,
goal_check: Callable[[], Awaitable[GoalVerdict]] | None = None,
goal_check_enforce: bool = False,
unlisted_reask: UnlistedReaskCheck | None = None,
document_identity: Callable[[], Awaitable[str | None]] | None = None,
) -> ToolSpec:
"""`page_fingerprint` samples an opaque fingerprint of the page's rendered content (None when no
page is available). A finish(completed) is deferred (bounded by `max_settle_deferrals`, then
accepted) unless two samples `settle_wait_seconds` apart match, so the model re-verifies against
the settled state instead of a mid-render shell — delayed loads otherwise produce stochastic
false completions. A sampling error is unknown, not settled: it defers. The wait between samples
is capped at `deadline_at` (time.monotonic clock) and abandoned once `should_cancel` reports
True, so probing cannot outlive the loop's own bounds.
The symmetric failure side: when `activity` reports recent submit-class/captcha activity, a
finish(failed) is held for ONE evidence turn (`max_failure_deferrals`, per run like the
completed-side cap, not per verdict attempt) — a quiescence wait
bounded by `failure_settle_max_seconds`, then a deferral asking the model to re-observe —
because async submissions and captcha protocols otherwise produce false-negative verdicts.
`verification_blocker` is the one gate consulted for EVERY verdict: it refuses a completed claim
once the verification source terminally failed, and holds a failed OR terminated one while the
run is still awaiting a code it has unspent polling budget for -- a give-up at one 120s slice of
a 15-minute budget throws away minutes of waiting the run already owns. The hold is bounded by
the callee (the budget shrinks under every productive hold) and refused here without the deadline
headroom to fund the blocking poll slice it asks for. Apart from that gate, terminated is
ungated on both sides.
`pending_marker` reports the text the page still shows the control in `submit_watch` as in
flight with, or None. A settled page is not a submitted one -- a submit frozen mid-flight is
maximally stable, so the settle probe is satisfied by exactly the state it should refuse. A
completed verdict is held ONCE against that marker, with the re-observe it asks for budgeted; if
the model insists a second time its verdict stands, so a run that declares completion on a
still-pending control remains possible after this gate. Deliberately a POSITIVE observation -- a
probe that fails reports nothing, and nothing is not evidence of pending, so it accepts rather
than holding a run on probe flakiness.
`unlisted_reask` asks once per run whether a failed or terminated verdict that is about to stand only
missed a screen the site skipped. A grounded yes becomes completed unless a completed-side gate vetoes it;
it never gives a turn back, and a completed verdict cannot reach it. Its settle gate samples as many times as
the completed path would defer. When `document_identity` is wired, it is read before the judge and again after
the settle window, and a conversion whose identity changed or could not be read is refused, settled or not."""
deferrals = 0
failure_deferrals = 0
goal_check_held = False
reask_asked = False
async def _bounded_fingerprint() -> str | None:
"""page_fingerprint(), timed out against whatever is left of the run's deadline instead of
the flat default cap -- so a HANGING sampler can no longer run the full default timeout once
the deadline is nearly gone. Mirrors _sample_probe's own convention once the deadline has
fully elapsed: the call is skipped entirely (never even invoked, so a hanging implementation
is never awaited) and reads as a missing sample, same as a genuinely absent page. That is
deliberately NOT the same as an exception -- a real sampling error must still propagate and
defer (see _settled/_quiesced's docstrings), and a fast probe that would have answered
instantly must not be denied the chance just because the deadline's clock already read zero;
only a positive-but-reduced timeout, not an outright skip, can preserve that."""
assert page_fingerprint is not None # gated by each call site's own None check
timeout = _PAGE_PROBE_TIMEOUT_SECONDS
if deadline_at is not None:
remaining = deadline_at - time.monotonic()
if remaining <= 0:
return None
timeout = min(timeout, remaining)
return await _bounded_probe(page_fingerprint(), timeout=timeout)
async def _quiesced() -> bool:
"""Bounded wait for the page to stop mutating before the failure verdict's evidence turn.
Returns False when there is no page to observe (a deferral would burn a turn for nothing)."""
prev = await _bounded_fingerprint()
if prev is None:
return False
cap_at = time.monotonic() + failure_settle_max_seconds
while True:
wait = min(settle_wait_seconds, cap_at - time.monotonic())
if deadline_at is not None:
wait = min(wait, deadline_at - time.monotonic())
if wait <= 0:
return True
await asyncio.sleep(wait)
if should_cancel is not None and await should_cancel():
return True # defer: the loop's cancellation check ends the run before another turn
current = await _bounded_fingerprint()
if current is None:
return False
if current == prev:
return True
prev = current
async def _settled() -> bool:
first = await _bounded_fingerprint()
if first is None:
return True # no page to sample (non-recovering peek): accept the verdict as-is
wait = settle_wait_seconds
if deadline_at is not None:
wait = min(wait, deadline_at - time.monotonic())
if wait > 0:
await asyncio.sleep(wait)
if should_cancel is not None and await should_cancel():
return False # defer: the loop's cancellation check ends the run before another turn
return first == await _bounded_fingerprint()
async def _conversion_veto(result: UnlistedReask, identity_before: str | None) -> str | None:
"""The completed-side gates, as vetoes only: the model never claimed completion, so none may hold."""
try:
if should_cancel is not None and await should_cancel():
return "canceled"
except Exception:
return "canceled"
if pending_marker is not None and submit_watch is not None and submit_watch.selector:
timeout = _PAGE_PROBE_TIMEOUT_SECONDS
if deadline_at is not None:
timeout = min(timeout, deadline_at - time.monotonic())
if timeout <= 0:
return "deadline"
try:
if await _bounded_probe(pending_marker(submit_watch.selector), timeout=timeout):
return "pending_marker"
except Exception:
return "pending_marker"
if verification_blocker is not None:
try:
if await verification_blocker(CONVERSION_VERIFICATION_STATUS):
return "verification_blocker"
except Exception:
return "verification_blocker"
if page_fingerprint is not None:
if document_identity is not None and identity_before is None:
return "identity_unreadable"
settled = False
for _ in range(max_settle_deferrals + 1):
result.settle_rounds += 1
try:
settled = await _settled()
except Exception:
pass # as on the completed path: an unreadable sample is an unsettled one
result.settled = settled
if settled:
break
try:
if should_cancel is not None and await should_cancel():
return "canceled"
except Exception:
return "canceled"
if deadline_at is not None and deadline_at - time.monotonic() <= 0:
return "deadline"
if deadline_at is not None and deadline_at - time.monotonic() <= 0:
return "deadline"
if document_identity is not None:
identity_after = await _sample_probe(document_identity, deadline_at=deadline_at)
if identity_after is None:
return "identity_unreadable"
if identity_before != identity_after:
return "navigating"
elif not settled:
return "unsettled"
if goal_check is not None:
try:
verdict = await goal_check()
except Exception:
LOG.warning("taskv3 goal check of a reask conversion failed", exc_info=True)
return "goal_check" if goal_check_enforce else None
result.goal_check_verdict = verdict.verdict
# One check, no hold: the model never claimed completion, so there is no turn to give back.
if goal_check_enforce and verdict.is_contradiction:
return "goal_check"
return None
async def _reask(original: NonCompletedStatus, args: dict[str, Any]) -> ToolResult | None:
assert unlisted_reask is not None
reason = args.get("reason") or ""
# Before the judge: a navigation that commits while it runs must not become the baseline.
identity_before = (
await _sample_probe(document_identity, deadline_at=deadline_at)
if document_identity is not None and page_fingerprint is not None
else None
)
try:
result: UnlistedReask | None = await unlisted_reask(original, reason)
except Exception:
LOG.warning("taskv3 unlisted reask failed; keeping the verdict", exc_info=True)
result = None
if result is not None and result.converts:
result.veto = await _conversion_veto(result, identity_before)
result.converted = result.veto is None
LOG.info(
"taskv3 finish unlisted reask",
original_status=original,
verdict=result.verdict if result is not None else None,
terminate_criterion_holds=result.terminate_criterion_holds if result is not None else None,
converts=result.converts if result is not None else False,
converted=result.converted if result is not None else False,
veto=result.veto if result is not None else None,
settled=result.settled if result is not None else None,
settle_rounds=result.settle_rounds if result is not None else 0,
goal_check_verdict=result.goal_check_verdict if result is not None else None,
llm_key=result.llm_key if result is not None else None,
reask_llm_key=result.reask_llm_key if result is not None else None,
skipped_reason=result.skipped_reason if result is not None else "reask_error",
# Never the quote itself: it is page text, possibly customer data.
quote_chars=result.quote_chars if result is not None else 0,
latency_s=result.latency_s if result is not None else None,
turn=activity.turn if activity is not None else None,
)
if result is None or not result.converted:
return None
return ToolResult.ok(
content="Task attempt ended. No further actions are permitted.",
data={
"status": "completed",
"reason": result.reason,
"extracted_output": args.get("extracted_output"),
"converted_from": original,
"converted_from_reason": reason,
},
)
async def handler(args: dict[str, Any]) -> ToolResult:
nonlocal deferrals, failure_deferrals, goal_check_held, reask_asked
status = args.get("status")
if status not in ("completed", "failed", "terminated"):
return ToolResult.error(
f"invalid finish status: {status!r}; call finish again with status=completed|failed|terminated"
)
if (
status == "completed"
and pending_marker is not None
and submit_watch is not None
and submit_watch.selector
and not submit_watch.deferred
# Holding costs a turn, the tool calls of the re-observe it asks for, and deadline
# seconds. Without the headroom to spend them the run ends budget_exhausted, which is
# unmapped and lands on failed -- turning an honest hold into a false failure.
and _has_hold_headroom(activity, deadline_at)
):
try:
marker = await _bounded_probe(pending_marker(submit_watch.selector))
except Exception:
# Warning, not debug: the only way here is a broken probe, and a silently disabled
# gate reads exactly like a page that was never pending.
LOG.warning("taskv3 pending-marker probe failed; not treating it as pending", exc_info=True)
marker = None
if marker:
submit_watch.deferred = True
LOG.info("taskv3 completed verdict held: submission still in flight", marker=marker)
return ToolResult.error(
f"the page still shows a submission in flight: {marker}. That is not a "
"settled failure OR a confirmation -- wait for it to resolve and re-observe. "
"Finish with status=completed only once the page shows the submission "
"landed; if it never resolves, say so with status=terminated."
)
if status == "completed" and completion_blocker is not None:
try:
blocker_message = await completion_blocker(frozenset(staged_downloads or ()))
except Exception:
# Fail closed: a download-gated task must not complete on a storage hiccup with no file.
LOG.warning("taskv3 completion_blocker failed; failing closed", exc_info=True)
return ToolResult.error(
"Could not verify that a file download has finished; retry finish(status=completed) "
"once the download has landed, or finish with the non-completed status the "
"finish tool's own rule gives this outcome."
)
if blocker_message:
return ToolResult.error(blocker_message)
if verification_blocker is not None and (
status == "completed"
# A non-complete verdict is only ever HELD, never refused, so unlike the completed side it
# must fund the retry it asks for -- and that retry is a blocking poll slice, not just a
# re-observe. Absent the accounting to check that (no `activity`), the hold is not
# justifiable and the verdict stands.
or (
activity is not None
and _has_hold_headroom(
activity,
deadline_at,
min_deadline_headroom_seconds=VERIFICATION_GIVEUP_MIN_DEADLINE_HEADROOM_SECONDS,
)
)
):
try:
verification_message = await verification_blocker(status)
except Exception:
if status == "completed":
# Fail closed: an exception here must not let a blank verification step read as done.
LOG.warning("taskv3 verification_blocker failed; failing closed", exc_info=True)
return ToolResult.error(
"Could not verify that the verification-code step completed cleanly; retry "
"finish(status=completed) once verified, or finish with the non-completed "
"status the finish tool's own rule gives this outcome."
)
# Fail open on the give-up side: the opposite verdict. A broken gate must not trap a
# run that wants to end.
LOG.warning("taskv3 verification_blocker failed; honoring the verdict", exc_info=True)
verification_message = None
if verification_message:
return ToolResult.error(verification_message)
if (
status == "completed"
and page_fingerprint is not None
and deferrals < max_settle_deferrals
# On the granted final turn the re-verification turn a deferral asks for no longer
# exists: holding would convert the verdict into budget_exhausted instead.
and not (activity is not None and activity.final_turn_active)
):
try:
settled = await _settled()
except Exception:
# Fail closed: an exception while probing is evidence of nothing, so the verdict is
# deferred for re-verification rather than validated. The deferral cap still bounds it.
settled = False
if not settled:
deferrals += 1
return ToolResult.error(
"the page was still rendering, or could not be verified as settled, when you "
"called finish. Wait for it to settle, re-observe, confirm the goal's effect is "
"present in the loaded content (not a loading indicator or empty container), "
"then finish again."
)
if status == "completed" and goal_check is not None:
try:
verdict: GoalVerdict | None = await goal_check()
except Exception:
LOG.warning("taskv3 finish goal check failed; accepting the verdict", exc_info=True)
verdict = None
if verdict is not None:
# Every first contradiction is held, `impossible` included: a transient error page or
# a page still loading reads as the site refusing, and one re-check absorbs it. Only a
# verdict AFTER a hold can end the run, and the second verdict decides how.
action: GoalCheckAction
if verdict.verdict == "achieved":
action = "accept"
elif not goal_check_held:
action = "hold"
verdict.no_headroom = not _has_hold_headroom(activity, deadline_at)
elif verdict.verdict == "impossible":
action = "terminate"
else:
action = "fail"
verdict.action = action
second: GoalVerdict | None = None
second_skipped_reason: str | None = None
if not goal_check_enforce and action == "hold" and not verdict.no_headroom:
# Shadow accepts here and the run ends, so the second check enforce would make after
# the held turn is asked now instead, once. An upper bound on enforce's false
# failures: enforce's agent gets a turn to scroll or re-observe first.
try:
cancelled = should_cancel is not None and await should_cancel()
except Exception:
cancelled = True
if cancelled:
second_skipped_reason = "cancelled"
else:
try:
second = await goal_check()
except Exception:
LOG.warning("taskv3 finish goal check recheck failed", exc_info=True)
if second is not None:
second.recheck = True
second_skipped_reason = second.skipped_reason
second.action = (
("terminate" if second.verdict == "impossible" else "fail")
if second.is_contradiction
else "accept"
)
# A skipped or failed recheck is no answer; an ungrounded contradiction is one
# (enforce would accept it).
if second.skipped_reason in (None, "ungrounded_quote"):
verdict.would_fail = second.is_contradiction
LOG.info(
"taskv3 finish goal check",
mode="enforce" if goal_check_enforce else "shadow",
verdict=verdict.verdict,
would_be_action=action,
no_headroom=verdict.no_headroom,
# Never the quote itself: it is page text, possibly customer data, and this index is
# searchable. The judge's full response is stored as the step's LLM response artifact.
quote_chars=len(verdict.quote),
quote_source=verdict.quote_source,
skipped_reason=verdict.skipped_reason,
latency_s=verdict.latency_s,
second_verdict=second.verdict if second is not None else None,
second_skipped_reason=second_skipped_reason,
second_latency_s=second.latency_s if second is not None else None,
would_fail=verdict.would_fail,
turn=activity.turn if activity is not None else None,
)
try:
canceled_during_check = should_cancel is not None and await should_cancel()
except Exception:
canceled_during_check = False
if canceled_during_check:
# Defer, like the settle probe: the loop's cancellation check ends the run before another
# turn, instead of this finish persisting a completion for a canceled run.
return ToolResult.error("the run was canceled while the completion was being checked.")
if goal_check_enforce and action == "hold" and not verdict.no_headroom:
goal_check_held = True
if activity is not None:
activity.held_verdict_batch_skip = True
found = (
"the site preventing the goal"
if verdict.verdict == "impossible"
else "them contradicting the goal"
)
return ToolResult.error(
"completed verdict held once: an independent check of the page and of your recent "
f"tool results found {found}: {verdict.bounded_missing.rstrip('.')}. Look at the page "
"once more and confirm which verdict is right before finishing again."
)
if goal_check_enforce and action in ("fail", "terminate"):
return ToolResult.ok(
content="Task attempt ended. No further actions are permitted.",
data={
"status": "failed" if action == "fail" else "terminated",
"reason": f"goal check: {verdict.bounded_missing}",
"extracted_output": None,
"goal_check_ended": True,
},
)
if (
status == "failed"
and activity is not None
and page_fingerprint is not None
and failure_deferrals < max_failure_deferrals
# Same rule as the completed-side holds: no retry turn exists on the granted final turn.
and not activity.final_turn_active
and activity.armed()
# The corrected cycle needs budget for its worst case (wait + observe + re-finish);
# without headroom on every budget axis a deferral would convert an honest failure
# into budget_exhausted (or, for a stall-streak one short of the terminator, into a
# generic stall termination that replaces the model's accurate reason).
and (activity.turns_remaining is None or activity.turns_remaining >= FAILURE_EVIDENCE_MIN_TURNS)
and (
activity.tool_calls_remaining is None
or activity.tool_calls_remaining >= FAILURE_EVIDENCE_MIN_TOOL_CALLS
)
# The token margin is deliberately approximate: sized off the triggering turn, while
# the deferral turns carry a slightly larger transcript.
and (
activity.tokens_remaining is None
or activity.tokens_remaining >= FAILURE_EVIDENCE_MIN_TURNS * max(activity.last_turn_tokens, 1)
)
and not activity.perception_stall_imminent
and (
deadline_at is None or deadline_at - time.monotonic() >= FAILURE_EVIDENCE_MIN_DEADLINE_HEADROOM_SECONDS
)
):
should_defer = True
try:
# False only when there is no page to observe; cancellation mid-wait still defers
# (the loop's own cancel check ends the run first).
should_defer = await _quiesced()
except Exception:
pass # unknown page state still defers: the model's re-observe is the evidence step
if should_defer:
failure_deferrals += 1
LOG.info("taskv3 finish failure deferred for evidence", turn=activity.turn)
return ToolResult.error(
"failure verdict held for one evidence check: it follows recent page actions or "
"a captcha attempt whose effects can land after your last look — submissions and "
"captcha protocols often complete asynchronously, so the page may no longer show "
"the state this verdict was based on. Re-observe the page once (waiting briefly "
"first if it may still be processing): only a positive confirmation of the goal "
"(e.g. a submission confirmation banner) justifies finishing with "
"status=completed; if it still shows the blocked or failed state, or shows no "
"positive confirmation at all, finish again -- status=terminated if the page "
"itself is what stopped you, status=failed if the breakdown was yours -- and "
"the verdict will stand."
)
# Sampled here, after the failure-evidence gate, rather than at the run's first finish: a zero-action
# `finish(completed)` refused by a gate above and followed by `finish(failed)` records the failed verdict.
if activity is not None and status in ("failed", "terminated") and activity.status_at_hold_gate is None:
activity.attempts_at_hold_gate = activity.action_attempts
activity.perceptions_at_hold_gate = activity.perceptions
activity.status_at_hold_gate = status
original = _non_completed(status)
if original is not None and unlisted_reask is not None and not reask_asked and not goal_check_held:
reask_asked = True
converted = await _reask(original, args)
try:
canceled_during_reask = should_cancel is not None and await should_cancel()
except Exception:
canceled_during_reask = False
if canceled_during_reask:
# Defer, like the goal check: the loop's cancellation check persists `canceled` before another turn.
return ToolResult.error("the run was canceled while the verdict was being checked.")
if converted is not None:
return converted
return ToolResult.ok(
content="Task attempt ended. No further actions are permitted.",
data={
"status": status,
"reason": args.get("reason") or "",
"extracted_output": args.get("extracted_output"),
},
)
return ToolSpec(
name="finish",
description=("End the task and report whether the browser goal was completed. " + FINISH_STATUS_TAXONOMY),
parameters={
"type": "object",
"properties": {
"status": {"type": "string", "enum": ["completed", "failed", "terminated"]},
"reason": {"type": "string", "maxLength": 2000},
"extracted_output": {"description": "Structured output requested by the goal, if any."},
},
"required": ["status", "reason"],
},
handler=handler,
terminal=True,
)
# Caps the arguments an elision placeholder echoes: a selector is model-authored and unbounded.
_READ_LABEL_MAX_CHARS = 120
def _declared_args_key(spec: ToolSpec, args: dict[str, Any]) -> str:
"""The argument part of a read's key (`_PerceptionEntry.key`): only the arguments the tool DECLARES.
An argument the tool does not declare cannot change what it reads, so it cannot make two calls different
reads (the page it reads can, which is the page part of the key). The specs are not emitted strict, so a
provider is free to add one — and for an argumentless tool that would split the key and retain two
snapshots where the tool only ever describes the page as it is NOW.
For `look` that is not merely wasted context. Every call disposes the previous handles, clears
`_look_manifest` and renumbers the marks (`tools.py`), so a retained older legend describes numbers
that now address different controls, and `click(mark=N)` following it acts on the wrong one. It
fails open: the stale legend looks perfectly valid. `look` and `observe` declare none and carry no
page part (`issues_handles`), so all their calls collapse to one key and exactly one survives.
"""
declared = (spec.parameters or {}).get("properties") or {}
return json.dumps({k: v for k, v in args.items() if k in declared}, sort_keys=True, default=str)
def _read_label(tool_name: str, args_key: str) -> str:
"""How an elided snapshot names the read it dropped, e.g. `get_html(selector=#rows, offset=20000)`.
A tool whose reads cannot differ is named bare, exactly as before: observe and look take no
arguments, so decorating them would add a token to every elision and distinguish nothing. The
arguments are the model's own, echoed from the assistant message that already carries them, so
this discloses nothing the transcript did not already hold — but it is capped anyway, because a
selector has no length the model cannot choose.
"""
try:
args = json.loads(args_key)
except (TypeError, ValueError):
return tool_name
if not isinstance(args, dict) or not args:
return tool_name
rendered = ", ".join(f"{k}={args[k]}" for k in sorted(args))
if len(rendered) > _READ_LABEL_MAX_CHARS:
rendered = rendered[:_READ_LABEL_MAX_CHARS] + "…"
return f"{tool_name}({rendered})"
_COMPACTED_PREFIX = "[superseded "
@dataclass(slots=True)
class _PerceptionEntry:
tool: str
args: str # `_declared_args_key`; the only part of the key a placeholder or marker may name
page: str # the page probe sampled before the call; "" for a tool that issues handles
# sha256 of the model-facing bytes before `delta`, not the canonical digest: a re-minted marker is new bytes.
raw: str
turn: int
handles: bool
# The tool-reported tail at `delta_at` (a text delta of what an action changed). Outside `raw`, so it never
# makes an unchanged read look changed, and carried verbatim through every placeholder and marker.
delta: str = ""
shown: bool = True
@property
def key(self) -> tuple[str, str, str]:
return (self.tool, self.page, self.args)
@dataclass
class _PerceptionStore:
"""Every successful perception read, by message index, and the rewrites that bound them.
Read bodies are budgeted by size (`PERCEPTION_RETAIN_CHARS_HIGH`/`_LOW`) and rewritten only in
epochs, so between epochs each request is a byte prefix of the next. Tools that issue handles
(`observe`, `look`) keep exactly one body and elide the older one at once: the next call renumbers
refs/marks, so a stale legend addresses the wrong control and fails open.
A re-read whose bytes equal a body still shown becomes a marker citing it. Nothing is predicted: the
read ran, and this compares what it returned. A marker is not an entry, and an epoch moves a cited
body into the newest citing marker's slot before it elides anything, so a marker never outlives its body.
"""
entries: dict[int, _PerceptionEntry] = field(default_factory=dict)
# marker message index -> (cited body index, turn the marker's read was taken, the marker's own delta)
markers: dict[int, tuple[int, int, str]] = field(default_factory=dict)
unchanged_marks: int = 0
epoch_rewrites: int = 0
retained_chars_peak: int = 0
def admit(
self,
index: int,
*,
tool: str,
args: str,
page: str,
handles: bool,
content: str,
turn: int,
delta_at: int | None = None,
) -> str:
"""Record the read about to be appended at `index`; returns what the model is shown for it."""
cut = delta_at if delta_at is not None and 0 <= delta_at <= len(content) else len(content)
head, delta = content[:cut], content[cut:]
entry = _PerceptionEntry(
tool=tool,
args=args,
page="" if handles else page,
raw=hashlib.sha256(head.encode()).hexdigest(),
turn=turn,
handles=handles,
delta=delta,
)
if not handles:
prior_index = max((i for i, e in self.entries.items() if e.key == entry.key), default=None)
prior = None if prior_index is None else self.entries[prior_index]
if prior_index is not None and prior is not None and prior.shown:
label = _read_label(tool, args)
if prior.raw == entry.raw:
# The marker locates its body only by label (the turn number is not in the transcript), so a
# newer body with the same label from another page would be the one the model looks at.
latest_same_label = max(
i for i, e in self.entries.items() if e.shown and e.tool == tool and e.args == args
)
marker = (
f"[{label} returned the same {len(head)} chars as your read at turn {prior.turn}, "
"still shown above]"
)
if latest_same_label == prior_index and len(head) >= PERCEPTION_MARKER_MIN_RATIO * len(marker):
self.markers[index] = (prior_index, turn, delta)
self.unchanged_marks += 1
return marker + delta
else:
# On the NEW read: labelling the old one would rewrite a message the provider has cached.
content = (
f"[{label} returned different content from your read at turn {prior.turn}; that one "
f"predates your later actions, this is the current one]\n{content}"
)
self.entries[index] = entry
return content
def compact(self, messages: list[dict[str, Any]]) -> None:
"""Run before every LLM call. The unread round (after the last assistant message) is never elided
and never charged: one batched turn must not evict every earlier read."""
last_assistant_idx = -1
for i in range(len(messages) - 1, -1, -1):
if messages[i].get("role") == "assistant":
last_assistant_idx = i
break
newest: dict[tuple[str, str, str], int] = {}
for i in sorted(self.entries):
if self.entries[i].shown:
newest[self.entries[i].key] = i
for i, e in list(self.entries.items()):
if e.shown and e.handles and i < last_assistant_idx and newest[e.key] != i:
self._elide(messages, i)
read = self._read_bodies(last_assistant_idx)
charged = sum(len(messages[i]["content"]) for i in read)
# Two full-size reads (window plus its continuation notice) exceed HIGH, so without a floor an epoch
# would break paging, which the count window kept at two. The floor never triggers an epoch on its own.
if charged > PERCEPTION_RETAIN_CHARS_HIGH and len(read) > len(self._floor(read, newest)):
self.epoch_rewrites += 1
self._move_cited_bodies(messages)
newest = {}
for i in sorted(self.entries):
if self.entries[i].shown:
newest[self.entries[i].key] = i
read = self._read_bodies(last_assistant_idx)
charged = sum(len(messages[i]["content"]) for i in read)
floor = self._floor(read, newest)
older = [i for i in read if newest[self.entries[i].key] != i]
current = [i for i in read if newest[self.entries[i].key] == i]
for i in older + current:
if charged <= PERCEPTION_RETAIN_CHARS_LOW:
break
if i in floor:
continue
charged -= len(messages[i]["content"])
self._elide(messages, i)
self.retained_chars_peak = max(self.retained_chars_peak, charged)
def _floor(self, read: list[int], newest: dict[tuple[str, str, str], int]) -> set[int]:
"""The newest reads the model has seen, one per key, that an epoch never elides."""
return set([i for i in read if newest[self.entries[i].key] == i][-PERCEPTION_RETAIN_MIN_READS:])
def _read_bodies(self, last_assistant_idx: int) -> list[int]:
return sorted(i for i, e in self.entries.items() if e.shown and not e.handles and i < last_assistant_idx)
def _placeholder(self, entry: _PerceptionEntry) -> str:
return f"{_COMPACTED_PREFIX}{_read_label(entry.tool, entry.args)} output elided to bound context]"
def _elide(self, messages: list[dict[str, Any]], index: int) -> None:
entry = self.entries[index]
messages[index]["content"] = self._placeholder(entry) + entry.delta
entry.shown = False
def _move_cited_bodies(self, messages: list[dict[str, Any]]) -> None:
citing: dict[int, list[int]] = {}
for marker_index, (body_index, _, _) in self.markers.items():
citing.setdefault(body_index, []).append(marker_index)
for body_index, marker_indices in citing.items():
target = max(marker_indices)
body = self.entries[body_index]
shown = messages[body_index]["content"]
_, turn, target_delta = self.markers[target]
messages[target]["content"] = shown[: len(shown) - len(body.delta)] + target_delta
self._elide(messages, body_index)
for i in marker_indices:
if i != target:
messages[i]["content"] = self._placeholder(body) + self.markers[i][2]
self.entries[target] = _PerceptionEntry(
tool=body.tool,
args=body.args,
page=body.page,
raw=body.raw,
turn=turn,
handles=False,
delta=target_delta,
)
self.markers.clear()
def log_fields(self) -> dict[str, int]:
return {
"perception_unchanged_marks": self.unchanged_marks,
"perception_epoch_rewrites": self.epoch_rewrites,
"perception_retained_chars_peak": self.retained_chars_peak,
}
@dataclass(kw_only=True, slots=True)
class LoopState:
"""Run-scoped state of run_agent_tool_loop, held as one typed object rather than threaded
through the closure as nonlocals.
Several fields encode an ordering invariant that no type can express, so the invariant is
documented at the field it constrains.
"""
outcome: LoopOutcome | None = None
pending_nav_dead_end: _NavDeadEnd | None = None
stall_nudges_due: list[tuple[str, int]] = field(default_factory=list)
# Page-state stall detector (SKY-15265): consecutive billable rounds on a byte-identical
# fingerprint, whether the one re-plan nudge went out, and whether one is due this turn.
trailing_page_state_stall_rounds: int = 0
peak_page_state_stall_rounds: int = 0
# Whether the detector ever reached a verdict. A wired sampler is not a taken sample: it is only
# read behind a billable call, so a run that lands none never judges the page at all.
page_state_ever_judged: bool = False
page_state_nudge_delivered: bool = False
page_state_nudge_due: bool = False
# The download-completion probe refused by the verification gate. The model was told the run
# auto-completes on download, so a silent refusal leaves it acting blindly until a stall guard
# terminates it; this hands it the reason once so it can finish failed instead.
verification_refusal_nudge: str | None = None
verification_refusal_nudged: bool = False
# The last fingerprint sample from the PREVIOUS batch: a delayed render can land between one
# batch's after-sample and the next batch's before-sample, so movement is checked across
# batches, not only within them.
page_state_prev_fp: str | None = None
refresh_cycles: int = 0
refresh_nudge_due: bool = False
reload_failed_nudge_due: bool = False
pending_screenshots: list[bytes] = field(default_factory=list)
# The action round of the latest positive page-change evidence (SKY-15264, SKY-15666); the
# budget-extension gate reads it, so what counts as evidence is load-bearing, not cosmetic.
last_change_evidence_step: int | None = None
# Budget caps. The four are re-derived and applied ATOMICALLY on an extension grant: a partial
# update converts a step-cap death into a token-cap death, so they move together or not at all.
max_turns: int
max_tool_calls: int
max_action_steps: int | None = None
max_tokens: int | None = None
turns: int = 0
no_tool_call_turns: int = 0
total_tool_calls: int = 0
tool_seconds: float = 0.0
total_tokens: int = 0
action_steps: int = 0
# Refused billable calls left uncharged since the last charged, non-refused one.
uncharged_refusals: int = 0
budget_extended_notice: str | None = None
# Progress-gated budget extension (SKY-15264, SKY-15666): how many extensions have been granted,
# and the cap they are all sized and bounded against — captured before the first grant so growth
# stays linear. The evidence input is last_change_evidence_step above.
budget_extensions_granted: int = 0
original_action_steps: int | None = None
token_clamp_reported: bool = False
# Final-turn grant (budget-exhaustion final turn): mirrors budget_extension_granted's shape. A
# budget-cap trip anywhere in the loop sets this and buys one more unconstrained model turn. One
# grant per latch, but an extension that relieves the latching guard releases it and re-arms the
# latch for a later trip; cap_trip_pending remembers which cap granted it, so a finish() outcome
# produced on that turn can carry the fact forward even though finish is a different code path.
# final_turn_started distinguishes "granted, turn not run yet" from "granted turn already ran": a
# mid-batch grant (max_tool_calls, the action-step gate) leaves the SAME counter tripped for the
# very next top-of-turn check, one code site removed from the grant with no turn boundary between
# them -- without this flag that check would end the run before the granted turn ever started.
final_turn_granted: bool = False
final_turn_started: bool = False
cap_trip_pending: str | None = None
# Extraction carried by a finish the granted turn staged but never got honored (skipped behind a
# failed or refused call, or refused by a fail-closed blocker). The verdict itself stays
# unhonored — its premise didn't hold — but the data rides the spent-grant exit.
final_turn_staged_output: Any = None
# We own the message array and assign it to the caller's message_history before
# each call, passing prompt=None: LLMCaller.use_message_history never appends the
# assistant reply or tool results itself, so multi-turn tool use must be threaded here.
messages: list[dict[str, Any]] = field(default_factory=list)
# Successful perception results by index into `messages`, recorded as they are appended, so
# compaction never infers "real snapshot" from content size and a skip/error result is never one.
reads: _PerceptionStore = field(default_factory=_PerceptionStore)
perception: _PerceptionLedger = field(default_factory=_PerceptionLedger)
# Net-progress ledger (additive shadow); None disables it, mirroring the guard's *_after knobs.
progress: _ProgressLedger | None = None
canonical: _CanonicalProgressTracker = field(default_factory=_CanonicalProgressTracker)
revisit_memory: _RevisitMemory = field(default_factory=_RevisitMemory)
# The action-loop counter: (repeat count, first turn of the streak) per billable action
# identity, cleared whenever evidence of page change arrives. action_warned holds the streaks
# whose warning was actually DELIVERED — termination is gated on it, so the model always gets
# the warning (and a chance to self-correct) at least one turn before the verdict.
# (repeats, first turn, page fingerprint before the first attempt, moved since then): the nudge
# may only claim the page is unchanged when no sample in the streak left that fingerprint.
action_counts: dict[tuple[str, str], tuple[int, int, str | None, bool]] = field(default_factory=dict)
action_warned: set[tuple[str, str]] = field(default_factory=set)
billable_actions: list[str] = field(default_factory=list)
# Credential re-submit guard (SKY-16594), keyed on the placeholder token rather than the field:
# the page re-renders between attempts, so an element-keyed rule is evaded by the very re-render
# that precedes the second submit. Tokens entered since the last submit, and how many times each
# has been submitted; entering one that has spent CREDENTIAL_SUBMIT_BUDGET is refused.
credentials_entered: set[str] = field(default_factory=set)
credential_submits: dict[str, int] = field(default_factory=dict)
# A single-action block's completion is offered through the finish tool at most once: a guard that
# holds a verdict only once would pass a second offer the model never saw it hold.
block_completion_offered: bool = False
async def run_agent_tool_loop(
*,
llm_caller: Any,
system_prompt: str,
user_prompt: str,
tools: list[ToolSpec],
max_turns: int,
max_tool_calls: int,
max_action_steps: int | None = None,
max_action_steps_ceiling: int | None = None,
prompt_name: str = "taskv3-agent-loop",
organization_id: str | None = None,
call_kwargs: dict[str, Any] | None = None,
should_cancel: Callable[[], Awaitable[bool]] | None = None,
on_action_round: Callable[[list[RoundAction], str | None], Awaitable[None]] | None = None,
on_pre_action: Callable[[str, dict[str, Any]], Awaitable[None]] | None = None,
max_tokens: int | None = None,
deadline_seconds: float | None = None,
retryable_call_exceptions: tuple[type[BaseException], ...] = (),
max_call_retries: int = 0,
call_retry_base_delay: float = 1.0,
# Makes this loop the owner of the LLM exhaustion receipt: calls are told not to emit one, and this
# fires once when the loop gives up, so a failure its own retry recovers is never counted.
on_llm_call_exhausted: Callable[[BaseException], None] | None = None,
stall_nudge_after: int | None = PERCEPTION_STALL_NUDGE_AFTER,
stall_terminate_after: int | None = PERCEPTION_STALL_TERMINATE_AFTER,
action_nudge_after: int | None = ACTION_LOOP_NUDGE_AFTER,
action_terminate_after: int | None = ACTION_LOOP_TERMINATE_AFTER,
progress_window: int | None = PROGRESS_LEDGER_WINDOW,
activity: ActivityRecency | None = None,
submit_watch: SubmitWatch | None = None,
telemetry_salt: str | None = None,
completion_probe: CompletionProbe | None = None,
verification_blocker: VerificationBlocker | None = None,
staged_downloads: set[str] | None = None,
initial_navigation_status: int | None = None,
# Only ever named in the dead-end verdict's text, so a caller that has the status but not the URL
# (or whose URL is unfit to print) still gets the same verdict, minus the place.
initial_navigation_url: str | None = None,
# Every URL the CALLER gave this run, normalized by `caller_known_published_urls`. A guard verdict
# publishes a landed URL's path only when it is one of these; every other path is elided to its
# host. Computed once by the caller (it owns the task config) and never read by a tool.
caller_known_urls: frozenset[str] = frozenset(),
# Resolves the run's drop-check secret values when a verdict is about to name a page-supplied
# element. Read at verdict time, not loop start: the registry grows as a run resolves credentials.
label_secret_values: Callable[[], Collection[str]] | None = None,
# Placeholder tokens minted for a login credential's username slot, held to
# LOGIN_IDENTIFIER_SUBMIT_BUDGET instead of CREDENTIAL_SUBMIT_BUDGET. Read per refusal.
login_identifier_tokens: Callable[[], Collection[str]] | None = None,
page_probe: Callable[[], Awaitable[str | None]] | None = None,
reload_page: Callable[[], Awaitable[None]] | None = None,
max_refresh_cycles: int = 3,
page_fingerprint: Callable[[], Awaitable[str | None]] | None = None,
# One model call's worth of token headroom, reserved so the granted final turn (below) is
# actually fundable rather than being immediately re-tripped by the same max_tokens check it
# was granted under. 0 (the default) reproduces today's unreserved check.
final_turn_token_reserve: int = 0,
# The caller's policy for sizing the runaway guards off an action-step budget — the same
# function that produced the max_turns/max_tool_calls/max_tokens above. Supplied so a granted
# budget extension can RE-DERIVE them from the extended cap instead of running on the leftover
# headroom of the pre-extension one; without it a bigger action budget merely converts a
# step-cap death into a token-cap death. None keeps the guards fixed for the whole run.
backstops_for_cap: Callable[[int], tuple[int, int, int]] | None = None,
semantic_commit_stats: SemanticCommitStats | None = None,
# Set for an extraction block: it reads, and may click to reveal what it reads, but it does not
# author input, so every FILL_TOOLS call is refused at dispatch.
refuse_input_entry: bool = False,
tool_trail: ToolTrail | None = None,
# A block whose contract is one action: once one succeeded, a follow-up the step cap refuses offers
# finish(completed) instead of failing the block, as the step engine completes it after that step.
single_action_block: bool = False,
) -> LoopOutcome:
tool_by_name = {tool.name: tool for tool in tools}
st = LoopState(
max_turns=max_turns,
max_tool_calls=max_tool_calls,
max_action_steps=max_action_steps,
max_tokens=max_tokens,
original_action_steps=max_action_steps,
messages=[
{"role": "system", "content": system_prompt},
{"role": "user", "content": user_prompt},
],
progress=_ProgressLedger(window=progress_window) if progress_window is not None else None,
)
# Per run, never logged: the hashes it keys are stable within this run (the only scope any guard
# decision spans) and uncorrelatable across runs, so page content and arguments cannot be
# fingerprinted across tenants from telemetry.
if telemetry_salt is None:
telemetry_salt = secrets.token_hex(16)
openai_tools = [tool.to_openai_tool() for tool in tools]
def _clear_action_state() -> None:
st.action_counts.clear()
st.action_warned.clear()
def _note_page_change_evidence() -> None:
st.last_change_evidence_step = st.action_steps
async def _consume_refresh_signal(
ctx: SkyvernContext, tool_name: str, remaining: list[Any], round_actions: list[Any], *, drop: bool
) -> bool:
"""Clear the page-refresh signal and, unless dropped or past the cap, reload and void `remaining`."""
ctx.refresh_working_page = False
st.refresh_cycles += 1
if drop:
LOG.info("taskv3 loop refresh signal dropped", tool=tool_name, turn=st.turns)
return False
if st.refresh_cycles > max_refresh_cycles:
# The queued calls were chosen on a page declared stale, so they are voided rather than
# run; a page that keeps demanding a reload cannot be stabilized, and the run ends there.
LOG.warning(
"taskv3 loop refresh signal past cap",
tool=tool_name,
turn=st.turns,
guard=PAGE_REFRESH_EXHAUSTED_GUARD,
refresh_cycles=st.refresh_cycles,
)
_append_skipped_tool_results(st.messages, remaining, "the page could not be stabilized")
st.outcome = _guard_verdict(
PAGE_REFRESH_EXHAUSTED_GUARD,
"The page kept having to be reloaded and never settled into a state the run could act "
"on, so the task could not continue there.",
)
return True
if reload_page is not None:
reload_record = ("reload_page", {"reason": "a page-level handler requested a refresh"})
try:
await reload_page()
except Exception:
# The page did not change, so nothing is re-baselined; the queued calls are still
# voided (they were chosen on a page declared stale), the signal is re-armed for
# another attempt (bounded by the cap), and the model is told the reload failed.
LOG.warning("taskv3 loop page reload failed after refresh signal", tool=tool_name, exc_info=True)
round_actions.append(RoundAction(*reload_record, False))
ctx.refresh_working_page = True
_append_skipped_tool_results(st.messages, remaining, "a page reload was requested but failed")
st.reload_failed_nudge_due = True
return True
round_actions.append(RoundAction(*reload_record, True))
LOG.info("taskv3 loop honored page refresh signal", tool=tool_name, turn=st.turns)
# The reloaded document is a new baseline for every ledger that described the old one, and a
# look taken before it would hand the model marks that no longer exist. That includes the
# budget-extension evidence stamp: pre-reload progress says nothing about the fresh document,
# so the run must re-demonstrate progress before it can earn an extension.
_clear_action_state()
st.last_change_evidence_step = None
st.trailing_page_state_stall_rounds = 0
st.page_state_nudge_delivered = False
st.page_state_prev_fp = None
st.perception.reset()
if activity is not None:
activity.perception_stall_imminent = False
st.pending_screenshots = []
st.pending_nav_dead_end = None
st.stall_nudges_due = []
if st.progress is not None:
st.progress.hard_progress()
st.canonical.progress(_ProgressEvidence.REFRESH_RELOAD)
if submit_watch is not None:
submit_watch.clear()
_append_skipped_tool_results(st.messages, remaining, "the page was refreshed")
st.refresh_nudge_due = True
return True
def _progress_observe_shadow(
observe_summary: dict[str, int], tool_name: str, attribution: dict[str, Any], baseline_before: int | None
) -> None:
"""Feeds a model-issued no-arg observe's invalid-fields count into the net-progress ledger."""
if st.progress is None or not observe_summary:
return
invalid_fields = observe_summary.get("invalid_fields")
# The ledger's True return is the SHADOW STALL verdict, not progress — the canonical clear
# keys on the ledger re-baselining (a new low, or a rise re-baseline: fresh context), which
# is visible as its baseline moving under an already-set baseline. The baseline is captured
# BEFORE _absorb_result_data runs: a replayed download notice hard-progresses the shadow
# ledger (nulling the baseline) on the same result whose summary carries the new low, and
# reading the post-absorb value would leave this clear dead for the rest of the page.
stalled = invalid_fields is not None and st.progress.observe(invalid_fields)
if baseline_before is not None and st.progress.invalid_baseline != baseline_before:
st.canonical.progress(_ProgressEvidence.INVALID_FIELDS_BASELINE_MOVE)
if stalled:
LOG.info(
PROGRESS_LEDGER_SHADOW_EVENT,
actions=st.progress.actions_since_progress,
invalid_fields=st.progress.last_invalid,
form_armed=st.progress.form_armed,
tool=tool_name,
turn=st.turns,
**attribution,
)
def _absorb_result_data(tool_name: str, spec: ToolSpec | None, result_data: dict[str, Any]) -> bool:
"""Absorbs a download/page-change signal in a tool's result.data (staged_downloads,
action-state clear, progress hard_progress). Returns whether marks were renumbered."""
if staged_downloads is not None and result_data.get("staged_download"):
staged_downloads.add(result_data["staged_download"])
if spec is not None and (result_data.get("download_notice") or result_data.get("page_state_changed")):
# A download landing or a navigation is progress no matter which tool witnessed it or
# whether that call itself errored: re-clicking the button that produces a file (a
# "download next" flow), or re-trying after navigating to a fresh page, is a healthy
# loop, not a stuck one. A same-URL reload is the exception: it resets the retry ledger
# like any reload but is a state WIPE, not progress — it clears the extension evidence
# exactly like the refresh-signal path.
_clear_action_state()
if result_data.get("same_url_reload") or result_data.get("nav_revisit"):
# A reload destroys the observed document and a revisit replaces it with a fresh
# instance of known territory: re-baseline the perception ledger (as the refresh
# path does) so the first post-navigation look cannot diff against a pre-navigation
# digest and read as progress — and clear the evidence stamp in both cases, since
# navigation is non-billable and a surviving stamp would stay maximally recent
# through any amount of oscillation.
st.perception.reset()
st.last_change_evidence_step = None
if activity is not None:
activity.perception_stall_imminent = False
elif result_data.get("download_new") or result_data.get("page_state_changed"):
# Only a download detected on THIS call is evidence — a compactable tool replaying
# a retained notice re-clears the retry ledger but earns no budget.
_note_page_change_evidence()
# The old document's perception streak cannot speak for the fresh page: clear the
# imminent flag exactly as the refresh path does.
if activity is not None:
activity.perception_stall_imminent = False
if st.progress is not None:
st.progress.hard_progress()
if (
result_data.get("download_new")
or result_data.get("page_state_changed")
or result_data.get("same_url_reload")
or result_data.get("nav_revisit")
):
# A REPLAYED notice (download_notice without download_new) re-clears the retry
# ledger but is not fresh progress: it must not keep wiping the canonical ring, or
# a post-download loop could never accumulate enough touches to emit telemetry.
st.canonical.progress(_ProgressEvidence.FRESH_DOWNLOAD_OR_NAVIGATION)
# A click that moved the URL is a real page transition (H1 hard progress) for the shadow
# ledger, but URL equality does NOT prove same-page (a URL-stable SPA form advance) — so only
# the positive direction is acted on, and kept OUT of the branch above so it never clears the
# action-loop guard's state; this signal is shadow-only and additive. It is a URL-only HINT
# (history.pushState moves the URL without changing the document), so it never stamps
# budget-extension evidence either — the content-confirmed signals are that bar.
if result_data.get("page_transitioned") is True:
if st.progress is not None:
st.progress.hard_progress()
st.canonical.progress(_ProgressEvidence.PAGE_TRANSITIONED)
return tool_name == "look" and bool(result_data.get("marks_renumbered"))
async def _completion_probe_outcome(
tool_name: str, spec: ToolSpec | None, result_data: dict[str, Any]
) -> LoopOutcome | None:
"""Consults the completion probe after a billable or download-signaling tool result."""
if not (
completion_probe is not None
and spec is not None
and (spec.billable or result_data.get("download_notice"))
# file_upload stages an http(s) source file into the same downloads dir; that landed
# file is not the run's OWN download unless the wrapper also flagged download_notice.
and not (result_data.get("staged_download") and not result_data.get("download_notice"))
):
return None
try:
completion_reason = await completion_probe(frozenset(staged_downloads or ()))
except Exception:
LOG.warning("taskv3 completion_probe failed; not treating it as complete", exc_info=True)
return None
if not completion_reason:
return None
# The probe is the second path to a completed outcome, and it never reaches the finish tool.
# Without this the verification gate would hold only one of the two, so a run whose code
# never arrived could still end `completed` on a file that happened to land.
if verification_blocker is not None:
try:
verification_message = await verification_blocker("completed")
except Exception:
# Fail closed, as the finish tool's completed side does: a broken gate must not let
# a blank verification step read as done.
LOG.warning("taskv3 completion probe verification gate failed; failing closed", exc_info=True)
return None
if verification_message:
LOG.info(
"taskv3 loop completion probe refused by the verification gate",
tool=tool_name,
turn=st.turns,
)
if not st.verification_refusal_nudged:
st.verification_refusal_nudge = verification_message
return None
LOG.info("taskv3 loop completion probe fired", tool=tool_name, turn=st.turns)
# No tracker-wide clear here: the probe firing is download progress for the COMPLETING
# touch only, and its call site drops that one target's pending rungs — a whole-generation
# bump would erase a sibling target's true-positive rung along with it.
return LoopOutcome("completed", completion_reason)
def _perception_stall_check(
ledger: _PerceptionLedger,
content_digest: str,
action_key: tuple[str, str],
tool_name: str,
attribution: dict[str, Any],
*,
content_only_digest: str | None = None,
refresh_pending: bool = False,
) -> tuple[LoopOutcome | None, list[tuple[str, int]]]:
"""Trips the nudge/terminate thresholds off ``ledger``'s own streak."""
stall_nudges: list[tuple[str, int]] = []
snap = ledger.record(action_key, content_digest)
if snap.progressed:
# This probe saw the page change since it last looked — fresh evidence of progress, so
# repeat counts for actions taken against the old state are stale. A first-time probe has
# no baseline and proves nothing, which is what keeps varied-selector probing from
# laundering repetition into progress. A multi-page wizard that clicks the same selector
# (e.g. "next") on every page relies on THIS clear to survive — page_transitioned alone
# deliberately does not clear the action-loop guard (see below), so only a progressed
# snapshot does.
ring = ledger.content_only.get(action_key) if content_only_digest is not None else None
# INVARIANT: this test and the budget-evidence test below must both refuse a return to
# the ring. They answer one question — did the run reach ground it has not already
# covered? — and relaxing either alone silently reopens SKY-14998: a page that CYCLES
# moves on every probe, so `progressed` holds every round, and an unconditional clear
# let the action DRIVING the oscillation reset its own counter forever (one production
# key ran 11 times against a threshold of 6 while the nudge fired once).
returned_to_known_ground = ring is not None and content_only_digest in ring
if not returned_to_known_ground:
_clear_action_state()
# Two landed digests that differ are positive evidence, exactly like a fingerprint
# mismatch — and the only movement evidence there is when page_fingerprint is absent.
st.canonical.progress(_ProgressEvidence.PERCEPTION_DIGEST)
# Evidence requires NEW content: a URL-only flip (history.pushState) still clears the
# repeat guards above but earns no budget, and neither does a return to a content state
# in the probe's recent ring (a panel toggling open and shut).
if ring and content_only_digest not in ring:
_note_page_change_evidence()
if content_only_digest is not None:
ring = ledger.content_only.get(action_key)
if ring is None:
ring = ledger.content_only.setdefault(action_key, deque(maxlen=PERCEPTION_RING))
ring.append(content_only_digest)
if activity is not None and stall_terminate_after is not None:
activity.perception_stall_imminent = ledger.next_snapshot_can_trip(stall_terminate_after)
# A refresh about to be honored re-baselines this ledger anyway, so a stall verdict raised on
# the stale page it is replacing would be wrong the instant the reload lands.
if stall_terminate_after is not None and snap.live >= stall_terminate_after and not refresh_pending:
LOG.info(
"taskv3 loop perception stalled",
tool=tool_name,
identical_count=snap.live,
turn=st.turns,
guard=PERCEPTION_STALL_GUARD,
**attribution,
)
return (
_guard_verdict(
PERCEPTION_STALL_GUARD,
"The page stopped changing in response to the run's actions — every fresh read of it "
"came back with the same content — so the task could not make progress there, "
"commonly because something on the page is in the way that the run cannot see or "
"operate, such as a prompt inside an embedded frame.",
),
stall_nudges,
)
if (
stall_terminate_after is not None
and snap.tool_identical == stall_terminate_after
and not ledger.suppressed_reported
):
ledger.suppressed_reported = True
LOG.info(
PERCEPTION_STALL_SUPPRESSED_EVENT,
tool=tool_name,
identical_count=snap.tool_identical,
turn=st.turns,
**attribution,
)
if (
stall_terminate_after is not None
and snap.probe_revisits >= stall_terminate_after
and not ledger.shadow_reported
):
ledger.shadow_reported = True
LOG.info(
PERCEPTION_STALL_SHADOW_EVENT,
snapshots=snap.probe_revisits,
tool=tool_name,
turn=st.turns,
**attribution,
)
if stall_nudge_after is not None and snap.tool_identical == stall_nudge_after:
# Warn off the per-tool counter, not ``live``: it moves by one per read, so it
# crosses the threshold exactly once per streak and before any live verdict.
stall_nudges.append((tool_name, snap.tool_identical))
return None, stall_nudges
# Mutable for the run: a provider that rejects tool_choice rejects it every turn, so a drop
# made once must stick.
active_call_kwargs = dict(call_kwargs or {})
if on_llm_call_exhausted is not None:
active_call_kwargs["caller_owns_exhaustion_receipt"] = True
def _degrade_tool_choice(exc: BaseException) -> bool:
"""Drop the optional call parameters (tool_choice, the reasoning-summary dict) and report
whether the turn is worth re-issuing.
Called only when the turn is otherwise about to end the run, so the cost is one extra call
on a run that was already failing. A context-window overflow is excluded because dropping a
parameter provably cannot fix it. Dropping reasoning_effort reverts to the config's own
value, so a provider that rejects the dict form cannot end the run on turn 1.
"""
if isinstance(exc, SkyvernContextWindowExceededError):
return False
# One parameter per degrade, least-proven first: a provider that rejects only the summary
# dict keeps its independently supported tool_choice; a second rejection drops that too.
for key in ("reasoning_effort", "tool_choice"):
if active_call_kwargs.pop(key, None) is not None:
LOG.warning(
"taskv3 loop retrying without optional call param", dropped=key, turn=st.turns, exc_info=True
)
return True
return False
# Images produced by an on-demand `look` this turn, to show the model on the NEXT call only. Passed
# as the transient screenshots= arg once, then cleared, so a look costs one image on one turn and
# never enters `messages` (the transcript re-seeds message_history each turn, so it's structurally
# gone the turn after).
started_at = time.monotonic()
deadline_at = started_at + deadline_seconds if deadline_seconds is not None else None
# The task's starting URL is navigated during browser setup, before this loop runs, so a dead/removed
# starting posting never routes through the in-loop `navigate` tool — the model just observes the dead
# page and finishes (defaulting to failed). Classify that pre-loop navigation here so the dominant
# dead-posting case ends `terminated`, matching v1, without waiting on the model's finish discretion.
# Cancellation is checked first, exactly as the first loop turn would: a run canceled during setup must
# persist as `canceled` (and stay unbilled), not be pre-empted into `terminated` by this fast path.
if st.outcome is None and initial_navigation_status in NAVIGATION_DEAD_END_STATUSES:
if should_cancel is not None and await should_cancel():
st.outcome = LoopOutcome("canceled", "run canceled")
else:
LOG.info(
"taskv3 loop initial navigation dead end",
http_status=initial_navigation_status,
guard=NAV_DEAD_END_GUARD,
)
st.outcome = _guard_verdict(
NAV_DEAD_END_GUARD,
_dead_end_reason(
initial_navigation_status,
initial_navigation_url,
page_noun="The task's starting page",
caller_known_urls=caller_known_urls,
),
)
while st.outcome is None:
if should_cancel is not None and await should_cancel():
st.outcome = LoopOutcome("canceled", "run canceled")
break
# The final-turn grant (see final_turn_granted above): a budget trip here buys one more
# turn instead of ending the run, so the checks below are elif'd
# (only the first tripped cap matters this iteration). A spent grant ends the run BEFORE any
# counter is consulted: the granted turn may have been granted at a gate no top-of-turn
# check re-reads (the action-step gate), so waiting for a counter to re-trip would let a
# finish-less granted turn keep looping on other budgets — and the cap that granted the
# turn, not whichever counter happens to re-trip first, is the honest fact to report.
# final_turn_started (set below, right before the turn actually runs) distinguishes a spent
# grant from a mid-batch grant whose turn has not run yet. Cancellation above is exempt: it
# is not a budget cap.
if st.final_turn_granted and st.final_turn_started:
st.outcome = LoopOutcome(
"budget_exhausted",
_budget_exhausted_reason(st.cap_trip_pending or "budget"),
cap_trip=st.cap_trip_pending,
extracted_output=st.final_turn_staged_output,
)
break
# Once the grant fired no top-of-turn cap can trip again, so the checks only run pre-grant.
top_of_turn_trip: str | None = None
if not st.final_turn_granted:
if deadline_seconds is not None and time.monotonic() - started_at > deadline_seconds:
top_of_turn_trip = f"deadline ({deadline_seconds:.0f}s) reached"
elif st.max_tokens is not None and st.total_tokens >= max(0, st.max_tokens - final_turn_token_reserve):
top_of_turn_trip = f"max_tokens ({st.max_tokens}) reached"
elif st.turns >= st.max_turns:
top_of_turn_trip = f"max_turns ({st.max_turns}) reached"
elif st.total_tool_calls >= st.max_tool_calls:
top_of_turn_trip = f"max_tool_calls ({st.max_tool_calls}) reached"
if top_of_turn_trip is not None:
st.final_turn_granted = True
st.cap_trip_pending = top_of_turn_trip
LOG.info(
FINAL_TURN_GRANTED_EVENT,
cap=top_of_turn_trip,
turn=st.turns,
tool_calls_remaining=None if activity is None else activity.tool_calls_remaining,
tokens_remaining=None if activity is None else activity.tokens_remaining,
)
st.messages.append({"role": "user", "content": _budget_exhausted_observation(top_of_turn_trip, activity)})
if st.final_turn_granted:
st.final_turn_started = True
if activity is not None:
# Read by the finish tool's hold gates: a hold's retry turn no longer exists.
activity.final_turn_active = True
st.turns += 1
if activity is not None:
activity.turn = st.turns
activity.turns_remaining = st.max_turns - st.turns
activity.tool_calls_remaining = st.max_tool_calls - st.total_tool_calls
# Elide superseded perception results before re-sending the transcript, so a perception-heavy
# run can't balloon the context to the token backstop (the pre-compaction runaway mode).
st.reads.compact(st.messages)
llm_caller.message_history = list(st.messages)
# Consume any pending look image into THIS call only, then clear: the image rides one request
# and is never appended to `messages`, so the turn after carries zero image blocks.
screenshots_for_call = st.pending_screenshots or None
st.pending_screenshots = []
# Retry only the LLM call on transient provider errors. No browser tool has run this
# turn, so re-issuing the same call is side-effect-free — unlike a whole-task retry,
# which would re-execute prior clicks/types. This restores the step engine's transient
# resilience, which v3 otherwise loses by running as one non-retried unit.
response = None
call_attempt = 0
while True:
try:
response = await llm_caller.call(
prompt=None,
prompt_name=prompt_name,
organization_id=organization_id,
tools=openai_tools,
use_message_history=True,
raw_response=True,
screenshots=screenshots_for_call,
**active_call_kwargs,
)
break
except retryable_call_exceptions as exc:
call_attempt += 1
if call_attempt > max_call_retries:
# A provider rejecting the parameter surfaces here, not in the generic handler
# below: litellm's 400s subclass openai.APIError, which the LLM layer maps to
# the retryable type. Degrading only after the transient budget is spent keeps
# a passing blip from disabling the lever for the rest of the run.
if _degrade_tool_choice(exc):
# Spend the transient budget once, not once per parameter set: the degraded
# turn gets a single shot, which is what "last resort" is worth.
call_attempt = max_call_retries
continue
LOG.warning(
"taskv3 loop LLM call failed after retries", turn=st.turns, attempts=call_attempt, exc_info=True
)
if on_llm_call_exhausted is not None:
on_llm_call_exhausted(exc)
st.outcome = LoopOutcome("loop_error", f"llm_call_failed: {type(exc).__name__}: {exc}")
break
LOG.info("taskv3 loop retrying transient LLM error", turn=st.turns, attempt=call_attempt)
await asyncio.sleep(call_retry_base_delay * (2 ** (call_attempt - 1)))
except Exception as exc:
if _degrade_tool_choice(exc):
continue
LOG.warning("taskv3 loop LLM call failed", turn=st.turns, exc_info=True)
if on_llm_call_exhausted is not None:
on_llm_call_exhausted(exc)
st.outcome = LoopOutcome("loop_error", f"llm_call_failed: {type(exc).__name__}: {exc}")
break
if st.outcome is not None:
break
usage = _get(response, "usage") or {}
turn_tokens = _get(usage, "total_tokens")
if not turn_tokens:
turn_tokens = (_get(usage, "prompt_tokens") or 0) + (_get(usage, "completion_tokens") or 0)
st.total_tokens += int(turn_tokens or 0)
if activity is not None:
activity.last_turn_tokens = int(turn_tokens or 0)
activity.tokens_remaining = None if st.max_tokens is None else st.max_tokens - st.total_tokens
text = _extract_text(response)
reasoning_summary = _extract_reasoning_summary(response)
tool_calls = _extract_tool_calls(response)
assistant_message: dict[str, Any] = {"role": "assistant", "content": text or None}
if tool_calls:
assistant_message["tool_calls"] = [
{"id": tool_call_id, "type": "function", "function": {"name": name, "arguments": json.dumps(args)}}
for tool_call_id, name, args in tool_calls
]
st.messages.append(assistant_message)
if not tool_calls:
st.no_tool_call_turns += 1
LOG.info("taskv3 loop turn produced no tool call", turn=st.turns)
st.messages.append({"role": "user", "content": NO_TOOL_CALL_NUDGE})
continue
# Dispatched feeds the page-state stall detector; charged feeds the action-step budget and the
# rows' billable flag. They differ only by calls the tool refused before acting on the page.
turn_dispatched_billable = False
turn_charged = False
st.stall_nudges_due = []
st.refresh_nudge_due = False
st.budget_extended_notice = None
st.reload_failed_nudge_due = False
action_nudges_due: list[tuple[str, dict[str, Any], int]] = []
round_actions: list[RoundAction] = []
# A hard 404/410 from an in-loop navigate, applied only AFTER the batch so a same-turn fallback
# navigate can clear it — the model is told to batch aggressively, and terminating on the first
# of a batched [navigate(dead), navigate(live)] would discard the recovery it planned.
st.pending_nav_dead_end = None
# A same-selector dependent of a failed page-action call is skipped; any later click, Enter-shaped
# submit, or finish in the batch is skipped too -- the loop cannot tell a submit from the first two,
# and a verdict written before the failure was seen may be wrong or mis-reasoned.
failed_selectors: set[str] = set()
batch_had_failure = False
# A refusal leaves its value in the field, so no click after it -- toggle or not -- may run.
batch_had_refusal = False
marks_stale = False
# Loop events minted this batch, emitted only after every progress signal the batch can
# produce has been absorbed (see the end-of-batch emission below).
pending_canonical_fires: list[dict[str, Any]] = []
batch_page_change_reason: str | None = None
batch_fp_before: str | None = None
# Sample only when the batch can actually land a billable action -- the end-of-batch check
# below gates on turn_dispatched_billable, so a finish-only or perception-only batch has no use for
# this baseline and shouldn't pay its round-trip.
batch_has_billable_call = any(
tool_by_name.get(tool_name) is not None and tool_by_name[tool_name].billable
for _, tool_name, _ in tool_calls
)
# A cancellation that already landed makes this batch's baseline dead work: the per-call
# check below (before the first dispatch) ends the batch before anything it'd inform runs.
batch_will_sample_baseline = batch_has_billable_call and page_fingerprint is not None
batch_cancelled = batch_will_sample_baseline and should_cancel is not None and await should_cancel()
# The page-state stall detector (SKY-15265) reads the before/after fingerprint pair for
# every billable batch.
if page_fingerprint is not None and batch_has_billable_call and not batch_cancelled:
batch_fp_before = await _sample_probe(page_fingerprint, deadline_at=deadline_at)
if (
st.page_state_prev_fp is not None
and batch_fp_before is not None
and batch_fp_before != st.page_state_prev_fp
):
# The page moved BETWEEN batches (a delayed render landing after the prior
# after-sample): the touches the old samples described are stale, and this batch's
# dispatch and extension decisions must not read them. Canonical-only — the
# incumbent stall counters keep their end-of-batch turn_dispatched_billable gate.
st.canonical.progress(_ProgressEvidence.CROSS_BATCH_MOVEMENT)
batch_fp_after: str | None = None
if activity is not None:
# Scoped to THIS batch, and cleared here rather than only on the consume path: several
# branches between the hold and that path can `break` out of the batch first (the
# refresh signal is the reachable one -- a held finish makes `verdict_stands` false, so a
# pending signal takes that exit). A flag surviving into the next batch would drop
# everything queued behind its first tool, citing a hold that did not happen there.
activity.held_verdict_batch_skip = False
for idx, (tool_call_id, tool_name, args) in enumerate(tool_calls):
# Enforce the cap per tool call so one batched turn cannot overrun it, and honor a
# cancellation that arrives mid-batch before the next click/type/submit runs. Neither
# this call nor the rest of the batch executes, so answer them as skipped.
# Gated on `not final_turn_granted`: once the final turn is granted this same counter is
# already at (or past) the cap by construction, so re-enforcing it here would block the
# granted turn's very first call (including finish) before it ever ran. The top-of-turn
# check is what actually ends the run if this granted turn doesn't finish either.
if not st.final_turn_granted and st.total_tool_calls >= st.max_tool_calls:
mid_batch_trip = f"max_tool_calls ({st.max_tool_calls}) reached"
st.final_turn_granted = True
st.cap_trip_pending = mid_batch_trip
LOG.info(
FINAL_TURN_GRANTED_EVENT,
cap=mid_batch_trip,
turn=st.turns,
tool_calls_remaining=None if activity is None else activity.tool_calls_remaining,
tokens_remaining=None if activity is None else activity.tokens_remaining,
)
_append_skipped_tool_results(st.messages, tool_calls[idx:], "tool-call budget reached")
st.messages.append({"role": "user", "content": _budget_exhausted_observation(mid_batch_trip, activity)})
break
if should_cancel is not None and await should_cancel():
st.outcome = LoopOutcome("canceled", "run canceled")
_append_skipped_tool_results(st.messages, tool_calls[idx:], "run canceled")
break
spec = tool_by_name.get(tool_name)
call_selector = _call_selector(args)
if marks_stale and call_selector is not None and call_selector.startswith("mark="):
if activity is not None:
# Same predicate as the extraction and credential refusals below: the model
# emitted a well-formed action call and the harness declined to dispatch it over
# its OWN bookkeeping.
activity.action_attempts += 1
st.messages.append(
{
"role": "tool",
"tool_call_id": tool_call_id,
"name": tool_name,
"content": (
"skipped: an earlier look in this batch renumbered the marks, so this mark was chosen "
"from an old screenshot; pick it again from the new one"
),
}
)
continue
if call_selector is not None and call_selector in failed_selectors:
st.messages.append(
{
"role": "tool",
"tool_call_id": tool_call_id,
"name": tool_name,
"content": f"skipped: depends on an earlier call in this batch on {call_selector} that failed",
}
)
continue
if batch_had_failure and _is_finish(tool_name):
# Any verdict queued behind the failure was written before the model saw it: a completed
# one may be false, and a failed/terminated one carries a reason that predates the error.
st.messages.append(
{
"role": "tool",
"tool_call_id": tool_call_id,
"name": tool_name,
"content": (
"skipped: a field in this batch failed before this verdict was reached; "
"re-observe, then finish with a status that reflects the failure"
),
}
)
continue
if refuse_input_entry and tool_name in FILL_TOOLS:
LOG.info(EXTRACTION_ENTRY_REFUSED_EVENT, tool=tool_name, turn=st.turns)
if activity is not None:
# The model attempted; the harness declined. A refusal must not read as a run
# that never tried.
activity.action_attempts += 1
# A refused call did not do what the rest of the batch was planned around, so it marks the
# batch failed: a later click, Enter-shaped submit, or finish in the same batch is skipped.
batch_had_failure = True
batch_had_refusal = True
st.messages.append(
{
"role": "tool",
"tool_call_id": tool_call_id,
"name": tool_name,
"content": (
"refused: an extraction block does not type, select, or upload input. Extract what the "
"page shows now (clicking to reveal content is allowed), or finish with a status "
"that reflects it"
),
}
)
continue
spent_credentials = (
{
token
for token in _credential_placeholders(args)
if st.credential_submits.get(token, 0) >= CREDENTIAL_SUBMIT_BUDGET
}
if tool_name in CREDENTIAL_ENTRY_TOOLS
else set()
)
if spent_credentials and login_identifier_tokens is not None:
try:
identifiers = set(login_identifier_tokens())
except Exception:
LOG.warning("taskv3 loop could not resolve the run's login identifier tokens", tool=tool_name)
identifiers = set()
spent_credentials = {
token
for token in spent_credentials
if token not in identifiers or st.credential_submits.get(token, 0) >= LOGIN_IDENTIFIER_SUBMIT_BUDGET
}
if spent_credentials:
LOG.info(CREDENTIAL_RESUBMIT_REFUSED_EVENT, tool=tool_name, turn=st.turns)
if activity is not None:
activity.action_attempts += 1
# Same reason the extraction refusal marks the batch: a click queued behind this call
# would submit the value still sitting in the field, which is the act being refused.
batch_had_failure = True
batch_had_refusal = True
st.messages.append(
{
"role": "tool",
"tool_call_id": tool_call_id,
"name": tool_name,
"content": (
"refused: this secret value has already been entered and submitted in this task. The "
"page asking for it again means it was not accepted, and every further attempt spends a "
"real allowance on the account behind it — a sign-in lockout budget, a card's retry "
"limit — so it will not be entered again: retyping it, rewording it or putting it in a "
"different field will all be refused the same way. Finish with a status that reflects "
"the rejected value, or continue with whatever does not need it"
),
}
)
continue
toggle_exempted = False
if batch_had_failure and not batch_had_refusal and _may_submit(tool_name, args):
toggle_exempted = await _is_toggle_click(tool_name, spec, args)
if batch_had_failure and _may_submit(tool_name, args) and not toggle_exempted:
st.messages.append(
{
"role": "tool",
"tool_call_id": tool_call_id,
"name": tool_name,
"content": (
"skipped: a field in this batch failed and this call may submit the form; "
"fix the field, then re-queue it"
),
}
)
continue
# Once the action-step budget is spent, refuse a further page action — terminate, mirroring
# the step engine's max-steps stop — but let perception/finish through, since the cap bounds
# new action rounds, not the separate re-observe/finish turn the system prompt asks for.
# A run with recent positive page-change evidence earns an extension of half the original
# cap first (SKY-15264): the observed exhaustion population splits into genuinely long
# multi-page forms dying mid-progress (the extension's target) and stalled runs the gate
# refuses so they fail exactly as before. The grant repeats under that same predicate
# while the evidence keeps arriving (SKY-15666) — a form that was long enough to need one
# extension is routinely long enough to need another — up to a hard multiple of the
# original cap.
if (
spec is not None
and spec.billable
and st.max_action_steps is not None
and st.action_steps >= st.max_action_steps
):
base_cap = st.original_action_steps if st.original_action_steps else st.max_action_steps
raw_extension = base_cap // 2
extension_limit = base_cap * ACTION_BUDGET_EXTENSION_MAX_FACTOR
extension = min(raw_extension, max(0, extension_limit - st.max_action_steps))
if max_action_steps_ceiling is not None:
# The org's workflow-run-wide step pool is a HARD ceiling the extension must
# never breach: truncate the grant to what the pool can fund.
extension = min(extension, max_action_steps_ceiling - st.max_action_steps)
# Re-derive the runaway guards from the cap this grant would produce, and never
# below their live values. This is the whole difference between converting a
# step-cap death into a completion and merely relocating it onto the token guard:
# the guards were sized as functions of the action-step budget, so moving that
# budget without moving them leaves the extension running on leftovers.
grant_max_turns, grant_max_tool_calls, grant_max_tokens = st.max_turns, st.max_tool_calls, st.max_tokens
headroom_gain = (0, 0, 0)
token_clamped = False
if backstops_for_cap is not None and extension > 0:
next_turns, next_tool_calls, next_tokens = backstops_for_cap(st.max_action_steps + extension)
grant_max_turns = max(st.max_turns, next_turns)
grant_max_tool_calls = max(st.max_tool_calls, next_tool_calls)
token_gain = 0
if st.max_tokens is not None:
grant_max_tokens = max(st.max_tokens, next_tokens)
token_gain = grant_max_tokens - st.max_tokens
headroom_gain = (
grant_max_turns - st.max_turns,
grant_max_tool_calls - st.max_tool_calls,
token_gain,
)
# Turns still scale with the cap but tokens do not: the token guard is at its
# ceiling, and a cap raised past that point silently stops buying tokens.
# Reported from the grant branch below, never here — a refused extension leaves
# the cap where it was, so reporting on it would name a cap never in effect and
# would spend the once-per-run latch on a non-event.
token_clamped = st.max_tokens is not None and token_gain == 0 and grant_max_turns > st.max_turns
refresh_ctx = skyvern_context.current()
if st.max_action_steps >= extension_limit:
allowed, gate_reason = False, "extension_limit_reached"
elif raw_extension <= 0:
allowed, gate_reason = False, "cap_too_small"
elif extension <= 0:
allowed, gate_reason = False, "hard_step_ceiling"
elif refresh_ctx is not None and refresh_ctx.refresh_working_page:
# A pending refresh voids this very action and re-baselines the page: the grant
# must not race it and spend the extension on pre-reload evidence.
allowed, gate_reason = False, "refresh_pending"
else:
allowed, gate_reason = _budget_extension_gate(
st.action_steps,
st.last_change_evidence_step,
st.action_warned,
# CURRENT confirmed stalled-ness by the ledger's own rules: form_armed
# (the latest look showed a form) plus a window of fruitless actions. Not
# the one-shot telemetry latch, and never a bare counter — a form-less page
# increments the counter but must not be judged stuck by it.
st.progress is not None
and st.progress.form_armed
and st.progress.actions_since_progress >= st.progress.window,
activity,
deadline_at,
extension,
seconds_per_step=(time.monotonic() - started_at) / max(st.action_steps, 1),
headroom_gain=headroom_gain,
)
# This read cannot see unabsorbed in-batch movement: action_steps is frozen during
# a batch, so the cap check trips on the batch's FIRST billable call — before any
# page action dispatches — and between-batch movement was absorbed when
# batch_fp_before was sampled.
looping_now = st.canonical.looping_targets()
LOG.info(
CANONICAL_EXTEND_DELTA_EVENT,
current_decision=allowed,
gate_reason=gate_reason,
canonical_looping_targets=looping_now,
would_block_extension=bool(allowed and looping_now),
)
if allowed:
original_cap = st.max_action_steps
st.budget_extensions_granted += 1
st.max_action_steps += extension
st.max_turns, st.max_tool_calls, st.max_tokens = (
grant_max_turns,
grant_max_tool_calls,
grant_max_tokens,
)
if activity is not None:
# The guards moved mid-turn; the failure-evidence gates read these same
# fields later in this turn and would otherwise judge the run on the
# headroom it had before the grant.
activity.turns_remaining = st.max_turns - st.turns
activity.tool_calls_remaining = st.max_tool_calls - st.total_tool_calls
activity.tokens_remaining = None if st.max_tokens is None else st.max_tokens - st.total_tokens
if token_clamped and not st.token_clamp_reported:
st.token_clamp_reported = True
LOG.info(
"task_v3 token backstop clamped at its ceiling",
log_code="taskv3_token_backstop_clamped",
step_cap=st.max_action_steps,
max_tokens=st.max_tokens,
extensions_granted=st.budget_extensions_granted,
)
if st.final_turn_granted and _cap_trip_relieved(
st.cap_trip_pending,
st.max_tokens,
st.total_tokens,
final_turn_token_reserve,
st.max_turns,
st.turns,
st.max_tool_calls,
st.total_tool_calls,
):
# This run was granted its single wrap-up turn by a guard the grant above just
# raised past the trip. Before SKY-15666 no cap could be relieved mid-run, so
# the latch was permanent and reporting the remembered cap was honest; now it
# would end the run naming a budget it is nowhere near, with the extension it
# just earned unspent. Release it and let the top-of-turn checks resume.
LOG.info(
FINAL_TURN_RELEASED_EVENT,
cap=st.cap_trip_pending,
turn=st.turns,
extensions_granted=st.budget_extensions_granted,
)
released_cap = st.cap_trip_pending
st.final_turn_granted = False
st.final_turn_started = False
st.cap_trip_pending = None
# Staged under a latch that no longer holds: leaving it would stamp a stale
# extraction onto a terminal produced by some LATER cap trip.
st.final_turn_staged_output = None
if activity is not None:
activity.final_turn_active = False
# Staged, not appended: a user message may not sit between an assistant
# turn's tool calls and their results, and this fires mid-batch. It rides
# out with the other end-of-batch notes.
st.budget_extended_notice = _budget_extended_observation(released_cap or "budget", activity)
LOG.info(
ACTION_BUDGET_EXTENDED_EVENT,
original_cap=original_cap,
extension=extension,
extensions_granted=st.budget_extensions_granted,
base_cap=base_cap,
max_turns=st.max_turns,
max_tool_calls=st.max_tool_calls,
max_tokens=st.max_tokens,
action_steps=st.action_steps,
turn=st.turns,
tool=tool_name,
)
else:
LOG.info(
ACTION_BUDGET_EXTENSION_REFUSED_EVENT,
gate_reason=gate_reason,
max_action_steps=st.max_action_steps,
extensions_granted=st.budget_extensions_granted,
base_cap=base_cap,
action_steps=st.action_steps,
turn=st.turns,
tool=tool_name,
)
step_cap_trip = f"Reached the maximum steps ({st.max_action_steps})"
_append_skipped_tool_results(st.messages, tool_calls[idx:], "action-step budget reached")
finish_spec = tool_by_name.get("finish")
block_refusal: str | None = None
if (
single_action_block
and st.billable_actions
and finish_spec is not None
and not st.block_completion_offered
):
st.block_completion_offered = True
block_reason = (
f"performed the block's action ({st.billable_actions[0]}); "
"a further action was past the block's step limit"
)
# Through the real handler, so every guard on a completed verdict still applies.
try:
block_finish = await finish_spec.handler({"status": "completed", "reason": block_reason})
except Exception:
LOG.warning("taskv3 block completion finish raised", exc_info=True)
block_finish = ToolResult.error("")
if activity is not None:
activity.held_verdict_batch_skip = False
if block_finish.status == "ok" and (block_finish.data or {}).get("status") == "completed":
st.outcome = LoopOutcome("completed", block_reason)
break
if block_finish.status == "error":
block_refusal = block_finish.content
# Unlike the mid-batch max_tool_calls check above, the step gate is NOT special-cased
# away once the final turn is granted: a billable dispatch on the granted turn still
# hits it, which is the honest exit the grant exists to produce.
if st.final_turn_granted:
# A finish staged later in this skipped batch: its VERDICT is not honored
# (it presumed the refused action would run — accepting a completed there
# could claim a success that never happened), but the extraction it was
# already carrying is salvaged onto the honest exit.
staged_finish_output = next(
(
t_args.get("extracted_output")
for _t_id, t_name, t_args in tool_calls[idx:]
if t_name == "finish" and isinstance(t_args, dict)
),
None,
)
# The cap that granted the final turn is the honest fact to report — the
# step gate merely happens to be where the spent grant gets caught.
st.outcome = LoopOutcome(
"budget_exhausted",
_budget_exhausted_reason(st.cap_trip_pending or step_cap_trip),
cap_trip=st.cap_trip_pending or step_cap_trip,
extracted_output=staged_finish_output,
)
else:
st.final_turn_granted = True
st.cap_trip_pending = step_cap_trip
LOG.info(
FINAL_TURN_GRANTED_EVENT,
cap=step_cap_trip,
turn=st.turns,
tool_calls_remaining=None if activity is None else activity.tool_calls_remaining,
tokens_remaining=None if activity is None else activity.tokens_remaining,
)
observation = _budget_exhausted_observation(step_cap_trip, activity)
if block_refusal:
observation += f"\n\nThe block's completion was not accepted: {block_refusal}"
st.messages.append({"role": "user", "content": observation})
break
# Submit-shaped actions (the failure-evidence predicate, minus captcha) are reported BEFORE
# dispatch, since after it the page may be the confirmation page. A failure here never fails
# the action, and the time is not billed to the tool.
# Consumed before the side-effecting pre-submit capture, so a call chosen on a page that is
# gone leaves no artifacts; the recheck right before the handler covers the awaits below.
hook_ctx = skyvern_context.current()
if hook_ctx is not None and hook_ctx.refresh_working_page:
if await _consume_refresh_signal(
hook_ctx, tool_name, tool_calls[idx:], round_actions, drop=reload_page is None
):
break
if (
spec is not None
and on_pre_action is not None
and tool_name != "solve_captcha"
and _arms_failure_evidence(tool_name, args, True)
):
try:
await on_pre_action(tool_name, args)
except Exception:
LOG.warning("taskv3 on_pre_action callback failed", tool=tool_name, exc_info=True)
# Sampled before dispatch so an error below can be checked for having moved the page even
# when the tool set no flag (a raised exception carries none). Not billed to the tool's timing.
probe_before: str | None = None
if spec is not None and page_probe is not None:
probe_before = await _sample_probe(page_probe, deadline_at=deadline_at)
# A signal raised between calls (a route handler finishing during the model's turn, or
# during the pre-dispatch probes above) is honored right before this call runs, as the
# legacy per-action check does; the model chose it on a page that is gone.
pre_ctx = skyvern_context.current()
if pre_ctx is not None and pre_ctx.refresh_working_page:
if await _consume_refresh_signal(
pre_ctx, tool_name, tool_calls[idx:], round_actions, drop=reload_page is None
):
break
# Charged only once the call is really dispatched: a call voided by a refresh spent nothing.
st.total_tool_calls += 1
if activity is not None:
# Refreshed per call, not per turn: a batched action+finish turn must not defer on
# a stale turn-start snapshot (the conversion the headroom guard exists to prevent).
activity.tool_calls_remaining = st.max_tool_calls - st.total_tool_calls
selector_kind = _selector_kind(args)
# Cleared per call, so a value can never carry over from the previous one in the batch.
_RESOLVE_SECONDS.set(None)
_HIT_CLASS.set(None)
_COVERED_LAYER.set(None)
_TEXT_DELTA.set(None)
_TOOL_CALL_SEQ.set(st.total_tool_calls)
dispatch_ctx = skyvern_context.current()
runtime_secrets_before = len(dispatch_ctx.runtime_secret_values) if dispatch_ctx is not None else 0
# As in V1/V2, acting on the page again means an earlier navigation failure is no longer what the
# task ends on; reads and finish keep it, since a run that ends there ended on that navigation.
if spec is not None and spec.touches_page and dispatch_ctx is not None and dispatch_ctx.task_id:
clear_task_nav_error_code(dispatch_ctx.task_id)
tool_started_at = time.monotonic()
if spec is None:
result = ToolResult.error(f"unknown_tool: {tool_name}")
else:
if spec.billable:
turn_dispatched_billable = True
try:
result = await spec.handler(args)
except ToolRefusal as refusal:
result = refusal.as_result()
except Exception as exc:
LOG.warning(
"taskv3 tool handler raised",
tool=tool_name,
exc_info=True,
**_navigate_record_fields(tool_name, args, None),
)
raised_class = _raised_error_class(exc)
result = ToolResult.error(f"tool_error: {type(exc).__name__}: {exc}", error_class=raised_class)
tool_duration_seconds = time.monotonic() - tool_started_at
st.tool_seconds += tool_duration_seconds
# A dispatched page action consumes a step even if it errors (it may mutate before failing),
# unless the tool refused it before acting on the page, within the grace. Billing counts
# successes only.
call_charged = spec is not None and spec.billable
if call_charged and result.status == "error" and result.refused and not result.touched_page:
call_charged = st.uncharged_refusals >= UNCHARGED_REFUSAL_GRACE
st.uncharged_refusals += 0 if call_charged else 1
elif call_charged:
st.uncharged_refusals = 0
turn_charged = turn_charged or call_charged
# Entry is recorded only when the tool succeeded (a stale-ref failure put nothing in the
# field), but the submit is recorded on DISPATCH whatever the verdict: the loop ran the
# click itself, so nothing here asks the page whether a submission completed. Entry first,
# so a type that pressed Enter submits the value it just entered.
if spec is not None:
if tool_name in CREDENTIAL_ENTRY_TOOLS and result.status == "ok":
st.credentials_entered |= _credential_placeholders(args)
if _may_submit(tool_name, args):
# Cleared, so a second click on an untouched form is not a second submission of
# a credential that was only ever entered once.
for token in st.credentials_entered:
st.credential_submits[token] = st.credential_submits.get(token, 0) + 1
st.credentials_entered.clear()
# Observe's summary counters are the only trace a perception change leaves on this
# record; its content is deliberately never logged. Gated on the tool, not the payload,
# so every other tool's record keeps exactly today's fields.
observe_summary = _observe_summary_fields(result) if tool_name == "observe" else {}
# Which URL the call was about, so a navigate row is attributable to a target without the
# arguments themselves being logged.
navigate_fields = _navigate_record_fields(tool_name, args, result)
menu_note_fields = _menu_note_record_fields(tool_name, result)
# Conditional for the same reason observe's counters are: a record only carries a field
# the call actually produced, so an ok call's record keeps exactly the fields it has
# today and `resolve_seconds` is absent (not null) on tools with no address to resolve.
cost_fields: dict[str, Any] = {}
if result.status == "error":
# `tool_error_class` on the record, `error_class` on the result: the log key lands in a
# FLAT index where `error_class` is already taken -- cloud/webeye logs it as
# `type(exc).__name__` at ~17 sites, so an unprefixed key here would mix this closed
# vocabulary with Python exception names under one facet, and taskv3 emits on every
# erroring tool call so it would dominate the values.
cost_fields["tool_error_class"] = result.error_class or "other"
# Only on the rows it describes, like the fields below: a billable call that did not
# spend an action step because the tool refused it before acting on the page.
if spec is not None and spec.billable and not call_charged:
cost_fields["charged"] = False
# Only on the class they describe, so every other erroring row keeps exactly the
# fields it has today. Total over `covered` rows by construction: the helper that
# builds all three messages records before it returns any of them, so a covered row
# missing these means the class was emitted somewhere that is not that helper.
covered = _COVERED_LAYER.get()
if result.error_class == "covered" and covered is not None:
cost_fields["covered_branch"] = covered["branch"]
cost_fields["covered_controls"] = covered["controls"]
cost_fields["covered_layer_kind"] = covered["layer_kind"]
elif result.ok_class is not None:
# Prefixed for the same flat-index reason as `tool_error_class` above. NOT defaulted
# the way that field is: it is emitted only by tools whose `ok` spans distinct
# outcomes, so a fleet-wide default would put a field on every successful call to
# say nothing. WHAT A DENOMINATOR MEANS HERE, because a groupBy drops rows missing a
# facet: grouping a tool by this facet alone shows its `ok` calls ONLY and silently
# excludes its errors. `tool_status` is the total partition -- cut on it first, then
# read this within `ok` and `tool_error_class` within `error`. A tool that emits
# this at all must emit it on EVERY `ok` branch, or the buckets don't sum to `ok`.
cost_fields["tool_ok_class"] = result.ok_class
# Read off the context variable, not the result: on the raise path the loop built the
# result itself and the handler's own resolution time would otherwise be lost.
resolve_seconds = _RESOLVE_SECONDS.get()
if resolve_seconds is not None:
cost_fields["resolve_seconds"] = resolve_seconds
text_delta = _TEXT_DELTA.get()
if text_delta is not None:
cost_fields["text_delta_seconds"] = text_delta[0]
if text_delta[1] is not None:
cost_fields["text_delta_lines"] = text_delta[1]
cost_fields["text_delta_chars"] = text_delta[2]
if text_delta[3]:
cost_fields["text_delta_over_bound"] = True
if text_delta[4]:
cost_fields["text_delta_skipped"] = text_delta[4]
if text_delta[5]:
cost_fields["text_delta_pending_lines"] = text_delta[5]
# Every click row carries this, defaulting to `unknown`, because a groupBy DROPS rows
# missing a facet -- a partial field would read as a clean result rather than a gap.
# Gated on `spec is not None` for the same reason the `tool` field is: an unregistered
# tool logs as `unknown_tool`, and such a row must not also carry a click-only facet.
if spec is not None and tool_name == "click":
hit = _HIT_CLASS.get() or {}
cost_fields["hit_class"] = hit.get("hit_class") or "unknown"
# Known for every click whether or not the probe ran, so it stays total alongside the
# class. The two qualifiers below are NOT total by design: they describe a reading
# that happened, and a fabricated `false` would be indistinguishable from a measured
# one -- so group them WITHIN a hit_class bucket, never alone.
# What the probe CALL cost at the production call site. Not the marginal round trip:
# the first reading per realm also pays isolated-world construction, which is why the
# median within one `hit_probe_isolated` bucket is the readable figure.
# Present together, iff the probe was attempted: `hit_class` is the only total one.
probe_seconds = hit.get("probe_seconds")
if probe_seconds is not None:
cost_fields["hit_probe_seconds"] = probe_seconds
cost_fields["hit_needed"] = bool(hit.get("needed"))
cost_fields["hit_probe_raised"] = bool(hit.get("raised"))
if hit.get("isolated") is not None:
cost_fields["hit_probe_isolated"] = hit["isolated"]
# The action-loop guard's key and the perception ledger's digest, computed here (pure) so
# their hashes ride the record below; the ledger itself is updated further down, unchanged.
action_key = (tool_name, json.dumps(args, sort_keys=True, default=str))
attribution: dict[str, Any] = {"action_key_hash": telemetry_hash(telemetry_salt, *action_key)}
content_digest: str | None = None
# Whether this result is markup a window cut open — the only thing that can carry a
# marker fragment at its head. `rendered_text` is the tool's own statement that its
# content holds no start tags of the page's own; asking the ARGUMENTS instead would make
# the loop re-derive from a tool's argument conventions something the tool already said.
# Exact spans the TOOL reported: where it cut a marker open at the window's head, and
# where it appended its own notice. Neither is re-derived here, because page- and
# server-authored text can both wear those shapes and only the tool knows what it wrote.
reported = result.data or {}
head_fragment_len = int(reported.get("head_fragment_len") or 0)
notice_at = reported.get("notice_at")
clip_spans = reported.get("clip_spans")
# The text-delta section's span in `result.content`, read once for this result: the stall
# digest drops exactly that span, and the perception store below re-finds it after masking.
reported_delta_at = reported.get("delta_at")
reported_delta_end = reported.get("delta_end")
if spec is not None and spec.compactable and result.status == "ok":
content_digest = hashlib.sha256(
_canonical_perception_content(
result.content,
is_observe=tool_name == "observe",
head_fragment_len=head_fragment_len,
notice_at=notice_at,
clip_spans=clip_spans,
delta_at=reported_delta_at,
delta_end=reported_delta_end,
).encode()
).hexdigest()
attribution["snapshot_digest"] = telemetry_hash(telemetry_salt, content_digest)
attribution["probe_first_time"] = st.perception.first_time(action_key)
# Emitted on its own record, never folded into the one above: the tool-call record's
# fields are a stable contract and this signal must not perturb them.
seen_count, new_states_since = st.revisit_memory.record(content_digest)
# A revisit with fresh ground covered in between is a drill-down returning to its
# list, not a replay. Reporting those too would fire the signal identically on
# healthy and stuck runs and leave the precision read it exists to feed unable to
# tell them apart.
if seen_count >= PERCEPTION_REVISIT_LOG_AFTER and not new_states_since:
LOG.info(
PERCEPTION_REVISIT_EVENT,
revisit_count=seen_count,
distinct_states=st.revisit_memory.distinct_states,
peak_revisits=st.revisit_memory.peak_revisits,
progressed_revisits=st.revisit_memory.progressed_revisits,
capped=st.revisit_memory.capped,
tool=tool_name,
turn=st.turns,
**attribution,
)
# The only per-tool-call timing the engine has: tool execution is the majority of a v3
# run's wall-clock and otherwise emits nothing at all. Names, sizes and booleans only —
# argument values and result content carry end-user data and must not be logged.
LOG.info(
"taskv3 tool call finished",
# A hallucinated name would otherwise put unbounded model output into an indexed
# field on every call; the name itself stays in the tool result the model reads.
tool=tool_name if spec is not None else "unknown_tool",
tool_status=result.status,
duration_seconds=tool_duration_seconds,
result_chars=len(result.content),
# Truthiness, not presence: the tools treat a null or empty selector as absent and
# fall back to scanning the whole page, which is the case this field exists to find.
selector_present=bool(args.get("selector")),
selector_kind=selector_kind,
billable=bool(spec is not None and spec.billable),
turn=st.turns,
batch_size=len(tool_calls),
batch_index=idx,
**cost_fields,
**observe_summary,
**navigate_fields,
**menu_note_fields,
**attribution,
)
if spec is not None and spec.billable:
target_key = _call_selector(args) or f"tool:{tool_name}"
same_count, same_errors = st.canonical.record_touch(target_key, result.status == "error")
# Every rung of the distribution is logged (not one cliff): the Phase-B block
# threshold is tuned from this data.
if (
same_count in CANONICAL_LOOP_FIRE_COUNTS
and same_errors >= same_count - 1
and st.canonical.claim_rung(target_key, same_count)
):
pending_canonical_fires.append(
{
"gen": st.canonical.gen,
"target_key_hash": telemetry_hash(telemetry_salt, target_key),
"tool": tool_name,
"repeat_count": same_count,
"repeat_errors": same_errors,
"turn": st.turns,
}
)
model_facing_content = result.content
skyvern_ctx = skyvern_context.current()
if skyvern_ctx is not None:
model_facing_content = skyvern_ctx.hide_from_model(model_facing_content)
# Masking can change lengths, so the tool's offset is re-found as the masked delta's suffix
# position; if masking broke the suffix, the whole result counts as the read.
delta_at: int | None = None
if type(reported_delta_at) is int and 0 <= reported_delta_at <= len(result.content):
delta = result.content[reported_delta_at:]
if skyvern_ctx is not None:
delta = skyvern_ctx.hide_from_model(delta)
if model_facing_content.endswith(delta):
delta_at = len(model_facing_content) - len(delta)
transcript_content = model_facing_content
if spec is not None and spec.compactable and result.status == "ok":
transcript_content = st.reads.admit(
len(st.messages),
tool=tool_name,
args=_declared_args_key(spec, args),
page=probe_before or "",
handles=spec.issues_handles,
content=model_facing_content,
turn=st.turns,
delta_at=delta_at,
)
# A refresh raised by this call re-baselines the stall and repeat ledgers below, so neither
# guard may end the run on it first.
refresh_pending = skyvern_ctx is not None and skyvern_ctx.refresh_working_page
st.messages.append(
{"role": "tool", "tool_call_id": tool_call_id, "name": tool_name, "content": transcript_content}
)
if tool_trail is not None and spec is not None and not spec.terminal:
tool_trail.record(
TrailEntry(
tool=tool_name,
status=result.status,
content=model_facing_content,
perception=spec.compactable,
page_changing=spec.touches_page and (result.touched_page or not result.refused),
# The two ways a secret reaches the page: a credential placeholder typed by an
# entry tool, and a value a tool registered as secret (a delivered verification
# code) for the model to type next.
# `navigate` resolves placeholders in a URL too.
secret_entered=(
(tool_name in CREDENTIAL_ENTRY_TOOLS or tool_name == "navigate")
and bool(_credential_placeholders(args))
)
or (
dispatch_ctx is not None
and len(dispatch_ctx.runtime_secret_values) > runtime_secrets_before
),
entered=entered_values(tool_name, args),
# Masked like the content, so an unchanged URL compares equal to the one observe prints.
url_before=_trail_url_before(result, skyvern_ctx),
)
)
# A look's annotated screenshot is shown to the model on the next call only, never stored in
# the transcript. Consumed and cleared at the top of the next turn. Only the LATEST snapshot
# survives: a second look in the same turn supersedes the first (its marks replace the prior
# ones), so re-sending the stale image would just hand the model a dead numbering.
if result.screenshots:
st.pending_screenshots = list(result.screenshots)
result_data = result.data or {}
# The page-state stall detector (SKY-15265) re-baselines on these flags. Keeps the FIRST
# qualifying signal this batch -- the reason field is diagnostic (telemetry), not the
# decision itself, so a later call's signal never overwrites it.
if batch_page_change_reason is None or batch_page_change_reason == "page_transitioned":
# page_transitioned checks LAST and can be superseded: it is a URL-only hint the
# stall detector must not reset on, so a stronger same-batch signal outranks it.
if result_data.get("page_state_changed"):
batch_page_change_reason = "page_state_changed"
elif result_data.get("download_new"):
# A freshly detected download is progress even when the DOM never moves (a
# download-next flow); a replayed notice deliberately does not qualify.
batch_page_change_reason = "download_new"
elif tool_name == "hover" and result.status == "ok":
# A hover's only purpose is to reveal state (submenus, tooltips) that a
# CSS-only change leaves invisible to the innerHTML fingerprint and the
# document-identity probe alike -- always treat it as a change signal.
batch_page_change_reason = "hover"
elif tool_name == "navigate" and result.status == "ok":
batch_page_change_reason = "navigate"
elif batch_page_change_reason is None and result_data.get("page_transitioned") is True:
batch_page_change_reason = "page_transitioned"
shadow_baseline_before = st.progress.invalid_baseline if st.progress is not None else None
if _absorb_result_data(tool_name, spec, result_data):
# Every mark=N still queued in this batch was chosen before this look renumbered the
# marks, so it now names an arbitrary element; a look refused before rebuilding its
# manifest leaves the old marks (and their failed keys) live.
marks_stale = True
st.canonical.invalidate_marks()
_progress_observe_shadow(observe_summary, tool_name, attribution, shadow_baseline_before)
if content_digest is not None:
stall_outcome, nudges_due = _perception_stall_check(
st.perception,
content_digest,
action_key,
tool_name,
attribution,
content_only_digest=hashlib.sha256(
_content_only_perception(
result.content,
is_observe=tool_name == "observe",
head_fragment_len=head_fragment_len,
notice_at=notice_at,
clip_spans=clip_spans,
delta_at=reported_delta_at,
delta_end=reported_delta_end,
).encode()
).hexdigest(),
refresh_pending=refresh_pending,
)
st.stall_nudges_due.extend(nudges_due)
if stall_outcome is not None:
st.outcome = stall_outcome
_append_skipped_tool_results(st.messages, tool_calls[idx + 1 :], "perception stalled")
break
if spec is not None and spec.billable:
if st.progress is not None:
st.progress.on_billable()
# Errored and refused dispatches count too: a repeat-failing action is the same
# no-progress pathology. It does not bound refusals (a stale one resets it);
# UNCHARGED_REFUSAL_GRACE does.
repeat_count, first_turn, streak_fp, streak_moved = st.action_counts.get(
action_key, (0, st.turns, batch_fp_before, False)
)
repeat_count += 1
streak_moved = streak_moved or (
streak_fp is not None and batch_fp_before is not None and batch_fp_before != streak_fp
)
st.action_counts[action_key] = (repeat_count, first_turn, streak_fp, streak_moved)
# Terminate only when the streak spans more than one turn AND its warning was
# delivered: the system prompt commands batching identical clicks (steppers,
# arrows), so a single-batch streak has had no chance to see feedback yet, and a
# verdict must never arrive before the model saw the warning it could have acted on.
if (
action_terminate_after is not None
and repeat_count >= action_terminate_after
and first_turn < st.turns
and (action_nudge_after is None or action_key in st.action_warned)
and not refresh_pending
):
LOG.info(
"taskv3 loop action repeated",
tool=tool_name,
repeat_count=repeat_count,
turn=st.turns,
guard=ACTION_LOOP_GUARD,
**attribution,
)
# The customer-facing half names the control by the page's own accessible name, never
# the selector the model's nudge uses: that selector is usually v3's own `data-tv3`
# marker, which matches nothing in the customer's markup (SKY-16271).
named_target = _verdict_target(
tool_name,
result_data.get(TARGET_LABEL_DATA_KEY),
result_data.get(TARGET_KIND_DATA_KEY),
label_secret_values,
skyvern_ctx,
)
st.outcome = _guard_verdict(ACTION_LOOP_GUARD, _action_loop_reason(named_target))
_append_skipped_tool_results(st.messages, tool_calls[idx + 1 :], "action loop")
break
if (
action_nudge_after is not None
and repeat_count >= action_nudge_after
and action_key not in st.action_warned
):
action_nudges_due.append((tool_name, args, repeat_count))
if submit_watch is not None and tool_name == "navigate" and result.status == "ok":
# Outside the billable/recordable branch on purpose: the run left the page and the
# control it clicked went too, whatever navigate's spec flags happen to say.
submit_watch.clear()
if activity is not None and spec is not None:
# The round_actions branch below is deliberately NARROWER: this one adds
# `engages_page`, which is engagement without billing or an action row. Perception is
# counted separately off `compactable`, the flag observe / get_html / look already
# share, rather than a tool name -- `look` does not even exist when vision is off.
if spec.touches_page:
activity.action_attempts += 1
elif spec.compactable and result.status == "ok":
activity.perceptions += 1
if spec is not None and (spec.billable or spec.recordable):
# Dispatched page actions enter the round with their outcome: a failed charged round
# still consumed budget and must persist (else later blocks undercount the run
# budget); refused calls and recordable tools persist without claiming a budget unit.
round_outcome = result_data.get(ACTION_OUTCOME_DATA_KEY)
round_outcome = round_outcome if isinstance(round_outcome, dict) else None
round_actions.append(
RoundAction(
tool_name,
args,
result.status == "ok" and not _outcome_reports_failure(round_outcome),
result_data.get(TARGET_LABEL_DATA_KEY) or None,
result_data.get(TARGET_KIND_DATA_KEY) or None,
call_charged,
round_outcome,
result.content if result.status == "error" else None,
)
)
if spec.billable and result.status == "ok":
st.billable_actions.append(tool_name)
if activity is not None and _arms_failure_evidence(tool_name, args, result.status == "ok"):
activity.last_trigger_turn = st.turns
if submit_watch is not None:
submit_selector = _names_submit_control(tool_name, args, result.status == "ok")
if submit_selector is not None:
submit_watch.record(submit_selector)
# A page-level handler (e.g. an anti-bot bypass that exhausted its retries) can reload the
# page out from under any tool call, billable or not; honored after the call's own bookkeeping,
# never over a verdict the call itself just produced, mirroring the legacy per-action check.
if skyvern_ctx is not None and skyvern_ctx.refresh_working_page:
# Consumed whether or not it can be acted on: the context outlives this run, and a
# signal left set would fire on the first action of the next block.
verdict_stands = spec is not None and spec.terminal and result.status == "ok"
if await _consume_refresh_signal(
skyvern_ctx,
tool_name,
tool_calls[idx + 1 :],
round_actions,
drop=reload_page is None or verdict_stands,
):
break
completion_outcome = await _completion_probe_outcome(tool_name, spec, result_data)
if completion_outcome is not None:
st.outcome = completion_outcome
if spec is not None and spec.billable:
# The download landed while this call's probe waited, so this call's touch is
# the progressing one: drop ITS pending rungs only, never a sibling target's.
completing_hash = telemetry_hash(telemetry_salt, _call_selector(args) or f"tool:{tool_name}")
pending_canonical_fires[:] = [
fire for fire in pending_canonical_fires if fire["target_key_hash"] != completing_hash
]
_append_skipped_tool_results(st.messages, tool_calls[idx + 1 :], "completion probe fired")
break
dead_end_status = result_data.get("navigation_dead_end")
if dead_end_status is not None:
# A hard 404/410 landing is a non-capability dead-end (a dead/removed posting). Remember
# it but do NOT break the batch: a later navigate in the same turn can land the run on a
# live page and clear it below. Applied once the batch settles (after this for-loop).
st.pending_nav_dead_end = _NavDeadEnd(dead_end_status, result_data.get("navigation_dead_end_url"))
elif result_data.get("page_state_changed"):
# A successful navigate moved the run off any dead page seen earlier this batch.
st.pending_nav_dead_end = None
if activity is not None and activity.held_verdict_batch_skip:
# Raised only by the goal-check enforce hold. Its held finish is neither billable nor recordable,
# so the generic error branch below would let a click or type batched behind it dispatch.
activity.held_verdict_batch_skip = False
_append_skipped_tool_results(
st.messages,
tool_calls[idx + 1 :],
"queued behind a verdict that was held; re-observe the page before choosing again",
)
break
if spec is not None and spec.terminal and result.status == "ok":
data = result.data or {}
st.outcome = LoopOutcome(
status=data.get("status", "completed"),
reason=data.get("reason", ""),
extracted_output=data.get("extracted_output"),
goal_check_ended=bool(data.get("goal_check_ended")),
converted_from=_non_completed(data.get("converted_from")),
converted_from_reason=data.get("converted_from_reason", ""),
# The model's own verdict wins whether or not it landed on the granted final turn;
# cap_trip just records the fact that a cap forced this to be the last turn.
cap_trip=st.cap_trip_pending if st.final_turn_granted else None,
)
break
if result.status == "ok" and result_data.get("readiness_incomplete"):
# A navigate whose document committed but did not finish loading. The rest of the batch
# was queued against a loaded page, and before SKY-16278 the tool raised here and the
# error branch below skipped it; keep that, so the readiness the tool reported reaches
# the model before it chooses its next action.
_append_skipped_tool_results(
st.messages,
tool_calls[idx + 1 :],
"the page this batch navigated to had not finished loading — observe it before re-queuing these",
)
break
# A toggle can auto-advance a wizard, leaving the batch's failed field on the step behind it.
# Read off the click's own before/after URL comparison, so the check costs nothing extra.
if toggle_exempted and result.status == "ok" and result_data.get("page_transitioned") is True:
_append_skipped_tool_results(
st.messages,
tool_calls[idx + 1 :],
"an earlier field in this batch failed and was left on the previous step; the page moved "
"on — re-observe",
)
break
if result.status == "error":
# A live refusal needs a new model choice, so stop_batch skips pending calls without claiming progress.
poisoned = (
tool_name == "navigate"
or result.content == PAGE_UNAVAILABLE_ERROR
or bool(
result_data.get("page_transitioned")
or result_data.get("page_state_changed")
or result_data.get("navigation_dead_end")
or result_data.get("stop_batch")
)
)
# Whether the page moved is independent of the tool's kind: a wait that timed out because
# the site navigated poisons the batch just as a failed click would.
if not poisoned and spec is not None and page_probe is not None:
probe_after = await _sample_probe(page_probe, deadline_at=deadline_at)
poisoned = probe_before is None or probe_after is None or probe_after != probe_before
if probe_before is not None and probe_after is not None and probe_after != probe_before:
# Two landed identity samples that differ: the failed call still moved the
# document, and the fingerprint may render identically on the new one (a
# same-template step) — absorb it here or the rung survives that blindness.
st.canonical.progress(_ProgressEvidence.PROBE_MISMATCH)
if poisoned:
_append_skipped_tool_results(
st.messages,
tool_calls[idx + 1 :],
"earlier tool call in this batch failed and changed the page — re-observe before "
"re-queuing these",
)
break
if spec is not None and (spec.billable or spec.recordable):
batch_had_failure = True
if call_selector is not None:
failed_selectors.add(call_selector)
# Whatever skipped or refused a finish once a cap tripped — behind the very call that
# tripped it in the GRANTING batch, behind a failed call or fail-closed blocker on the
# granted turn, or under a guard that terminated the batch — the extraction it was already
# carrying is remembered; the post-loop stamp rides it out on whatever terminal the run
# ends with. Last write wins, so a granted-turn restatement supersedes the older capture.
if st.final_turn_granted:
staged = next(
(
t_args.get("extracted_output")
for _t_id, t_name, t_args in tool_calls
if t_name == "finish" and isinstance(t_args, dict) and t_args.get("extracted_output") is not None
),
None,
)
if staged is not None:
st.final_turn_staged_output = staged
# The batch settled on a dead page (an in-loop navigate hit a hard 404/410 and no later navigate
# recovered): end the run as terminated deterministically, matching v1, rather than leaving the
# failed/terminated choice to the model's finish tool (which does not converge on this class).
if st.outcome is None and st.pending_nav_dead_end is not None:
LOG.info(
"taskv3 loop navigation dead end",
http_status=st.pending_nav_dead_end.status,
turn=st.turns,
guard=NAV_DEAD_END_GUARD,
)
st.outcome = _guard_verdict(
NAV_DEAD_END_GUARD,
_dead_end_reason(
st.pending_nav_dead_end.status,
st.pending_nav_dead_end.url,
page_noun="The page the run opened",
caller_known_urls=caller_known_urls,
),
)
# Page-state stall detector (SKY-15265): tool-independent — any batch of billable work that
# leaves the rendered document byte-identical ticks the counter, whatever tools produced it.
# A missing sample is no evidence either way; any page-change flag or fingerprint movement
# re-baselines.
if (
st.outcome is None
and turn_dispatched_billable
and page_fingerprint is not None
and batch_fp_before is not None
):
if st.page_state_prev_fp is not None and batch_fp_before != st.page_state_prev_fp:
# The page moved BETWEEN batches (a delayed render landing after the prior
# after-sample): the streak the old samples described is stale.
st.trailing_page_state_stall_rounds = 0
st.page_state_nudge_delivered = False
page_state_changed: bool | None
if batch_page_change_reason is not None and batch_page_change_reason != "page_transitioned":
page_state_changed = True
else:
if batch_fp_after is None and not (deadline_at is not None and deadline_at - time.monotonic() <= 0):
batch_fp_after = await _sample_probe(page_fingerprint, deadline_at=deadline_at)
page_state_changed = None if batch_fp_after is None else batch_fp_after != batch_fp_before
st.page_state_prev_fp = batch_fp_after if batch_fp_after is not None else batch_fp_before
if batch_fp_after is not None:
for key, (count, first_turn, streak_fp, moved) in st.action_counts.items():
if not moved and streak_fp is not None and streak_fp != batch_fp_after:
st.action_counts[key] = (count, first_turn, streak_fp, True)
st.page_state_ever_judged = st.page_state_ever_judged or page_state_changed is not None
if page_state_changed is True:
st.trailing_page_state_stall_rounds = 0
st.page_state_nudge_delivered = False
st.canonical.progress(_ProgressEvidence.PAGE_STATE_VERDICT)
elif page_state_changed is False:
st.trailing_page_state_stall_rounds += 1
st.peak_page_state_stall_rounds = max(
st.peak_page_state_stall_rounds, st.trailing_page_state_stall_rounds
)
if (
st.trailing_page_state_stall_rounds == PAGE_STATE_STALL_TERMINATE_AFTER
and st.page_state_nudge_delivered
):
# Shadow-only verdict: measured, not enforced (see PAGE_STATE_STALL_SHADOW_EVENT).
LOG.info(
PAGE_STATE_STALL_SHADOW_EVENT,
rounds=st.trailing_page_state_stall_rounds,
turn=st.turns,
)
elif (
st.trailing_page_state_stall_rounds >= PAGE_STATE_STALL_NUDGE_AFTER
and not st.page_state_nudge_delivered
):
st.page_state_nudge_due = True
if (
st.outcome is not None
# An acknowledged cancellation must not wait on a possibly-hung renderer just to
# decide telemetry; and a batch whose pending fires are all generation-stale already
# has nothing left to absorb for.
and st.outcome.status != "canceled"
and page_fingerprint is not None
and batch_fp_before is not None
and any(fire["gen"] == st.canonical.gen for fire in pending_canonical_fires)
):
# A terminal outcome mid-batch (a finish, a fired completion probe) skips the detector
# above, so the batch's own movement is unabsorbed here: take the after-sample now and
# clear on a positive mismatch only, before deciding the pending events below.
if batch_fp_after is None:
batch_fp_after = await _sample_probe(page_fingerprint, deadline_at=deadline_at)
if batch_fp_after is not None and batch_fp_after != batch_fp_before:
st.canonical.progress(_ProgressEvidence.TERMINAL_BATCH_FINGERPRINT)
# A rung completed by — or followed in this batch by — absorbed progress describes a
# progressing run, not a loop: only events whose generation survived every clear above
# (same-call result data, invalid-fields baseline, batch fingerprint verdict) are emitted.
for pending_fire in pending_canonical_fires:
if pending_fire.pop("gen") == st.canonical.gen:
LOG.info(CANONICAL_LOOP_EVENT, **pending_fire)
pending_canonical_fires.clear()
# Warn only after the batch completes: a user message may not sit between an assistant
# turn's tool results, and the model reads it with the snapshot that tripped it. Every note
# due this turn shares ONE user message so the transcript keeps alternating roles.
nudge_parts: list[str] = []
if st.budget_extended_notice is not None:
# First: it retracts a standing claim that the run is ending, which every other note
# this turn is written as if untrue.
nudge_parts.append(st.budget_extended_notice)
if st.outcome is None and st.refresh_nudge_due:
nudge_parts.append(_refresh_nudge_text())
elif st.outcome is None and st.reload_failed_nudge_due:
nudge_parts.append(_reload_failed_nudge_text())
if st.outcome is None and st.stall_nudges_due:
nudge_parts.append(_stall_nudge_text(st.stall_nudges_due, set(tool_by_name)))
if st.outcome is None and st.page_state_nudge_due:
st.page_state_nudge_due = False
st.page_state_nudge_delivered = True
LOG.info("taskv3 loop page state stall nudged", rounds=st.trailing_page_state_stall_rounds, turn=st.turns)
nudge_parts.append(_page_state_nudge_text(st.trailing_page_state_stall_rounds))
if st.outcome is None and st.verification_refusal_nudge is not None:
nudge_parts.append(st.verification_refusal_nudge)
st.verification_refusal_nudge = None
st.verification_refusal_nudged = True
if st.outcome is None and action_nudges_due:
# Deliver only warnings whose streak survived the batch AND spans turns: a later call in
# the same batch (an observe showing the page changed, a download) may have cleared it,
# and a streak born entirely this turn has had no feedback yet — the message's "the
# state you last observed is unchanged" would be false for it. An undelivered warning
# stays unmarked, so it re-queues (and termination stays blocked) until the model has
# actually seen it. Counts read live, not the threshold-crossing snapshot, and logged
# here so the warn-then-recovered metric counts only warnings the model saw.
still_stuck = []
movements: list[bool | None] = []
for name, warn_args, _count in action_nudges_due:
key = (name, json.dumps(warn_args, sort_keys=True, default=str))
entry = st.action_counts.get(key)
if entry is not None and entry[1] < st.turns and key not in st.action_warned:
st.action_warned.add(key)
still_stuck.append((name, warn_args, entry[0]))
streak_fp, latest_fp = entry[2], st.page_state_prev_fp
comparable = streak_fp is not None and latest_fp is not None
if entry[3] or (comparable and latest_fp != streak_fp):
movements.append(True)
else:
# No pair to compare means no reading, which _sample_probe defines as
# evidence of nothing — never a claim that the page held still.
movements.append(False if comparable else None)
if still_stuck:
page_moved = True if any(movements) else (None if None in movements else False)
for name, _warn_args, count in still_stuck:
LOG.info(
"taskv3 loop action repeat nudged",
tool=name,
repeat_count=count,
turn=st.turns,
page_moved=page_moved,
)
nudge_parts.append(_action_nudge_text(still_stuck, set(tool_by_name), page_moved=page_moved))
if nudge_parts:
st.messages.append({"role": "user", "content": "\n\n".join(nudge_parts)})
# A "step" is one action round: a turn that ran >=1 page-mutating action. Perception-only
# turns (observe/get_html) don't consume the caller's step budget — the step engine bundles
# perception into each step, so counting v3's perception rounds against the same budget
# under-counts equivalent work.
if turn_charged:
st.action_steps += 1
# Hand the round's executed actions to the caller so it can persist per-action artifacts
# (screenshot, DB rows) — kept out of this transport-agnostic core, like should_cancel. A
# persistence hiccup must not abort an otherwise-good run, so failures are contained here.
if round_actions and on_action_round is not None:
try:
await on_action_round(round_actions, text or reasoning_summary or None)
except Exception:
LOG.warning("taskv3 on_action_round callback failed", turn=st.turns, exc_info=True)
if st.outcome is None:
st.outcome = LoopOutcome("loop_error", "loop exited without an outcome")
# Every model- or loop-produced terminal once the cap tripped — a finish verdict, a guard
# termination (even in the granting batch itself), a stall exit, the spent-grant exit —
# happened UNDER the cap: it carries the cap fact, and a missing extraction is filled from
# what a skipped finish had staged (the model's own earlier data), so no exit path can
# re-discard the partial output. A completed verdict that didn't restate its output would
# otherwise be demoted for missing extraction.
if st.final_turn_granted and st.outcome.status in (
"completed",
"failed",
"terminated",
"budget_exhausted",
"loop_error",
):
if st.outcome.cap_trip is None:
st.outcome.cap_trip = st.cap_trip_pending
if st.outcome.extracted_output is None and not st.outcome.goal_check_ended:
st.outcome.extracted_output = st.final_turn_staged_output
ledger_fields: LedgerTerminalFields | None = None
if st.progress is not None and st.progress.ever_armed:
# The ledger is precision-biased (see _ProgressLedger), so read its fire precision as
# trustworthy but its recall as a FLOOR: few fires is not few stuck runs.
ledger_fields = LedgerTerminalFields(
peak_actions_since_progress=st.progress.peak_actions_since_progress,
actions_since_progress=st.progress.actions_since_progress,
form_armed=st.progress.form_armed,
would_fire=st.progress.shadow_reported,
)
st.outcome.telemetry = TerminalTelemetry(
# The sticky flag, not `form_armed`: the latter is the CURRENT look and is cleared by
# progress, so a run that saw a form early and lost it by the end reads False on it. This is
# the partition the two collapsed records used to encode by which one of them fired.
form_ever_armed=st.progress is not None and st.progress.ever_armed,
survival=st.canonical.survival_fields(),
ledger=ledger_fields,
# Present only where the detector actually judged the page at least once, so the field means
# "this was the worst streak" and never "nothing ever looked".
peak_page_state_stall_rounds=st.peak_page_state_stall_rounds if st.page_state_ever_judged else None,
# Present for any run that RE-READ a probe, which is the population the counter is defined
# on. It ships unconditionally there because it is the calibration input for a pending
# threshold: the shadow line fires only above the current cutoff, so nothing below it is
# observable from logs today.
peak_probe_revisits=st.perception.peak_probe_revisits if st.perception.revisit_chances else None,
semantic_commit=semantic_commit_stats,
# Present only for a run that took a perception read, the population the counters describe.
perception_reads=st.reads.log_fields() if st.reads.entries else None,
)
st.outcome.turns = st.turns
st.outcome.no_tool_call_turns = st.no_tool_call_turns
st.outcome.tool_choice_in_effect = "tool_choice" in active_call_kwargs
st.outcome.tool_calls = st.total_tool_calls
st.outcome.tool_seconds = st.tool_seconds
st.outcome.action_steps = st.action_steps
st.outcome.billable_actions = st.billable_actions
st.outcome.messages = st.messages
return st.outcome