mirror of
https://github.com/Skyvern-AI/skyvern.git
synced 2026-10-03 04:06:21 +00:00
5094 lines
284 KiB
Python
5094 lines
284 KiB
Python
"""Faithful Task V3 agent tool-loop.
|
|
|
|
A single persistent LLM conversation drives browser tools via native tool-calling:
|
|
the model emits ``tool_calls``, we execute them, thread the results back as ``tool``
|
|
messages, and repeat until the model calls a terminal tool (``finish``) or a budget
|
|
cap is hit. Perception is a tool the model chooses to call — nothing about the page
|
|
is injected automatically — which is what distinguishes this from the step engine's
|
|
scrape-every-step loop.
|
|
|
|
The loop itself is transport-agnostic: it depends only on an ``LLMCaller``-shaped
|
|
object and a list of ``ToolSpec``. Browser wiring lives in a separate module so this
|
|
core can be unit-tested with scripted fakes.
|
|
"""
|
|
|
|
from __future__ import annotations
|
|
|
|
import asyncio
|
|
import hashlib
|
|
import json
|
|
import re
|
|
import secrets
|
|
import time
|
|
from collections import deque
|
|
from contextvars import ContextVar
|
|
from dataclasses import dataclass, field
|
|
from enum import Enum
|
|
from typing import Any, Awaitable, Callable, Collection, Iterator, Literal, NamedTuple, Sequence, TypeVar
|
|
from urllib.parse import urlsplit
|
|
|
|
import structlog
|
|
|
|
from skyvern.exceptions import SkyvernContextWindowExceededError
|
|
from skyvern.forge.sdk.core import skyvern_context
|
|
from skyvern.forge.sdk.core.skyvern_context import SkyvernContext
|
|
from skyvern.forge.sdk.workflow.context_manager import RANDOM_SECRET_ID_PREFIX
|
|
from skyvern.forge.taskv3.goal_check import (
|
|
GoalCheckAction,
|
|
GoalVerdict,
|
|
NonCompletedStatus,
|
|
ToolTrail,
|
|
TrailEntry,
|
|
UnlistedReask,
|
|
)
|
|
from skyvern.forge.taskv3.handoff_redaction import MAX_HANDOFF_URL_CHARS, sanitize_published_url
|
|
from skyvern.forge.taskv3.target_label import describe_target
|
|
from skyvern.webeye.navigation import clear_task_nav_error_code, redact_url_secrets
|
|
|
|
LOG = structlog.get_logger()
|
|
|
|
ToolStatus = Literal["ok", "error"]
|
|
FinishStatus = Literal["completed", "failed", "terminated"]
|
|
UnlistedReaskCheck = Callable[[NonCompletedStatus, str], Awaitable[UnlistedReask]]
|
|
|
|
# Why a failing tool call failed, as one closed vocabulary. Spelled as a `Literal` rather than `str`
|
|
# because the whole value of the facet is that it is closed: it is written at ~33 literal sites in
|
|
# tools.py, and a typo or a near-synonym added later would split a cohort silently. Enforcement is
|
|
# invocation-shaped: `mypy.ini` sets `follow_imports = skip`, so the annotation binds in tools.py only
|
|
# while THIS module is in the same mypy run. CI's `pre-commit run mypy --all-files` includes it and
|
|
# rejects an undeclared value; checking tools.py alone does not. The AST source census in
|
|
# tests/unit/test_taskv3_loop.py backstops the runs that miss it, and it can only read a value spelled
|
|
# as a literal at the write site.
|
|
ToolErrorClass = Literal[
|
|
# The address did not resolve to what it named.
|
|
"stale_selector",
|
|
"invalid_selector",
|
|
"invalid_mark",
|
|
"ambiguous_selector",
|
|
"ambiguous_frame",
|
|
"stale_ref",
|
|
"ref_not_in_latest",
|
|
"stale_mark",
|
|
"mark_not_in_latest",
|
|
# The target resolved, but the page will not let the act happen.
|
|
"disabled",
|
|
"not_editable",
|
|
# `type`: the page replaced the typed text with a non-empty value of its own; left in place.
|
|
"value_changed_by_page",
|
|
# `type`: the field does not hold the typed text afterwards -- an append that is partial or unchanged, or a
|
|
# one-character-per-box code field whose boxes did not all keep their character.
|
|
"text_not_held",
|
|
"covered",
|
|
"inert",
|
|
"unreachable",
|
|
# `file_upload`: the target is not a file input, holds no single one, and opened no file picker;
|
|
# or it is one whose click would submit its form, so it was not clicked.
|
|
"no_file_input",
|
|
"submits_form",
|
|
# `file_upload`: the input could not be read back and no upload request was seen.
|
|
"attach_unconfirmed",
|
|
# The field resolved and the page cooperated, but the requested VALUE named no single option.
|
|
# Each of these names only what was READ: an absence claim holds solely over a list read in full,
|
|
# which is why a declared-but-truncated list gets `rows_unread` rather than `no_matching_row`.
|
|
"ambiguous_rows",
|
|
"identical_rows",
|
|
"no_matching_row",
|
|
"rows_unread",
|
|
# The field opened, but no option list rendered to read at all.
|
|
"no_option_list",
|
|
# A read asked to resume past the end of what it was reading. Not an address failure and not a
|
|
# page refusal: the call was well-formed and the page cooperated, the offset simply named nothing.
|
|
"offset_past_end",
|
|
# The offset itself was unusable — negative, or not a whole number of characters.
|
|
"invalid_offset",
|
|
# `navigate`: nothing committed — a net error, or the commit budget expired with no response.
|
|
"navigation_failed",
|
|
# The value was a credential's one-time-code field, and no code could be produced for this call.
|
|
"no_one_time_code_source",
|
|
# A one-time-code placeholder was combined with other text, or named no credential field.
|
|
"unusable_one_time_code_value",
|
|
# The handler raised instead of returning; classified by `_raised_error_class`.
|
|
"driver_timeout",
|
|
"timeout_other",
|
|
"handler_raised",
|
|
# An erroring call whose construction site named no class. Deliberately a value rather than an
|
|
# absence, so "errored, unnamed" is countable and cannot be confused with "did not error".
|
|
"other",
|
|
]
|
|
|
|
# The `ok` counterpart to `ToolErrorClass`, for a tool whose success status spans outcomes that are
|
|
# not the same event. `tool_status` is the only total outcome field on the record, so a tool that
|
|
# returns `ok` for "did the thing", "there was nothing to do", and "declined to try" is indexed
|
|
# identically on all three and a blind detector reads as a working one. Naming the branch is
|
|
# telemetry, never a behaviour change: like `error_class` this is never serialized into the tool
|
|
# message, so the model cannot see it (the loop sends `content` only).
|
|
ToolOkClass = Literal[
|
|
# `solve_captcha`. Its three `ok` branches are genuinely different events and the `ok` for
|
|
# "absent" is correct -- no challenge present is not an error and must not force a retry.
|
|
"solved",
|
|
"absent",
|
|
"attempts_exhausted",
|
|
# `solve_captcha` with `image_selector`: characters were read off the image. Not "solved" -- only the
|
|
# page can say whether they were right, after the model types them.
|
|
"image_text_read",
|
|
# `file_upload`. `attached_no_activity` is the file confirmed on the input with no upload request
|
|
# seen -- a form that sends the file with the submit lands here, and so does an unwired handler.
|
|
"upload_seen",
|
|
"consumed_shown",
|
|
"attached_no_activity",
|
|
# `navigate`, in descending order of how far the landed document got: `loaded` saw the load event
|
|
# fire, `document_ready` returned on domcontentloaded with load still outstanding, and
|
|
# `committed_not_loaded` got a document that never became ready inside the readiness budget.
|
|
"loaded",
|
|
"document_ready",
|
|
"committed_not_loaded",
|
|
]
|
|
|
|
# Which `covered` message the model actually got. They are one `tool_error_class`, so without this
|
|
# the split is only recoverable by pulling step archives and classifying the prose.
|
|
CoveredBranch = Literal[
|
|
# The layer was named and its controls enumerated into the message.
|
|
"named",
|
|
# The layer holds a challenge frame, so the message names it and omits the dismissal sentence.
|
|
"challenge",
|
|
# The walk qualified nothing, the hit was the field's OWN container, AND an ancestor was found
|
|
# clipping the point the click would land on. The message names that container and asks for it to
|
|
# be opened or scrolled, because there is no third party here to dismiss. Same real-world
|
|
# condition as `error_class="unreachable"` -- union them for a reachability census.
|
|
"clipped",
|
|
# No reading at all: the probe reported occlusion without naming anything and without locating a
|
|
# clip. Four producers, and for some of them "something is on top of it" is exactly right: an
|
|
# ancestor drawing its own ::before veil (a busy card), a scrim drawn as `body::before` (which
|
|
# hit-tests AS body and so never reaches the naming walk at all), a pinned cover the field
|
|
# collapsed under, and a skinned control whose own label covers it.
|
|
"unnamed",
|
|
# The layer intercepts the pointer but paints nothing, so it is absent from the screenshot.
|
|
"invisible",
|
|
# The layer is the field's own container and a non-interactive element in it takes the pointer;
|
|
# the message names that stamped element to click, so no controls are listed.
|
|
"own_container",
|
|
]
|
|
|
|
# WHICH ELEMENT the probe named as the layer, which the branch cannot recover. `named` with zero
|
|
# controls spans two different events: a real overlay whose controls the enumeration dropped, and a
|
|
# walk that qualified nothing and named the raw hit element -- an option row or a value cell, which
|
|
# has no actionable child and nothing to dismiss. Only the probe knows which, so it is on the record.
|
|
CoveredLayerKind = Literal[
|
|
# The walk found an element that qualifies as a layer: pinned, a layer role, aria-modal, <dialog>,
|
|
# or view-sized.
|
|
"qualified",
|
|
# Nothing qualified, so the probe named the hit element itself.
|
|
"hit_fallback",
|
|
# `clipped` only: the probe named the ancestor it found clipping the click point. Not a layer --
|
|
# nothing is over the control -- but it is the one address the message can offer, so this is the
|
|
# split between a `clipped` the model can act on and one that can only say "that container".
|
|
"clipper",
|
|
# The probe named no element at all.
|
|
"unnamed",
|
|
]
|
|
|
|
|
|
@dataclass
|
|
class ToolResult:
|
|
status: ToolStatus
|
|
content: str
|
|
data: dict[str, Any] | None = None
|
|
# Transient images the loop must show the model on the NEXT call only (the on-demand `look`
|
|
# tool's annotated screenshot). Threaded into one .call()'s ephemeral screenshots= arg and never
|
|
# appended to the transcript, so it costs one image on one turn and is gone the turn after.
|
|
screenshots: list[bytes] | None = None
|
|
# Telemetry only, and deliberately NOT in `data`: callers and tests pin `data` by equality, so a
|
|
# measurement riding in it would change an observable contract. Never shown to the model.
|
|
error_class: ToolErrorClass | None = None
|
|
# Same contract as `error_class`, on the other side of the status. Telemetry only, never shown
|
|
# to the model. Each is read only under its own status, so a result carrying the class for the
|
|
# OTHER one drops it silently; the classmethods below cannot express that pairing (neither
|
|
# accepts the other's kwarg) and the raw constructor is the only route that could.
|
|
ok_class: ToolOkClass | None = None
|
|
# The requested action was refused and never dispatched (a ToolRefusal). A field, not a raise, so
|
|
# every wrapper around the handler still post-processes the call.
|
|
refused: bool = False
|
|
# A refused call that still dispatched input to the page first, to release a previous field's
|
|
# suggestion list. It is charged and the page may have changed, though the requested action never ran.
|
|
touched_page: bool = False
|
|
|
|
@classmethod
|
|
def ok(
|
|
cls,
|
|
content: str,
|
|
data: dict[str, Any] | None = None,
|
|
screenshots: list[bytes] | None = None,
|
|
*,
|
|
ok_class: ToolOkClass | None = None,
|
|
) -> ToolResult:
|
|
return cls("ok", content, data, screenshots, ok_class=ok_class)
|
|
|
|
@classmethod
|
|
def error(
|
|
cls, content: str, data: dict[str, Any] | None = None, *, error_class: ToolErrorClass | None = None
|
|
) -> ToolResult:
|
|
return cls("error", content, data, error_class=error_class)
|
|
|
|
|
|
ToolHandler = Callable[[dict[str, Any]], Awaitable[ToolResult]]
|
|
|
|
|
|
class ToolRefusal(Exception):
|
|
"""Refuse the call before it acts on the page (no input dispatched; a marker attribute written to
|
|
address an element is not an action). The loop sends the message alone as a refused error, which
|
|
does not spend an action step within the grace, so it must carry nothing the model must not see."""
|
|
|
|
def __init__(self, message: str, *, error_class: ToolErrorClass, data: dict[str, Any] | None = None) -> None:
|
|
super().__init__(message)
|
|
self.error_class: ToolErrorClass = error_class
|
|
self.data = data
|
|
|
|
@classmethod
|
|
def of(cls, result: ToolResult) -> ToolRefusal:
|
|
return cls(result.content, error_class=result.error_class or "other", data=result.data)
|
|
|
|
def as_result(self) -> ToolResult:
|
|
return ToolResult("error", str(self), self.data, error_class=self.error_class, refused=True)
|
|
|
|
|
|
# The tool-result `data` keys the target-name/target-kind capture ride on (written by the browser
|
|
# tools, read here). Internal to the loop: only `content` is ever shown to the model, so they cost no
|
|
# tokens. The label composition itself -- the floor vocabulary, the shape filter, the secret matcher --
|
|
# lives in target_label.py; this module only carries the two raw values from probe to `RoundAction`.
|
|
TARGET_LABEL_DATA_KEY = "target_label"
|
|
TARGET_KIND_DATA_KEY = "target_kind"
|
|
# The tool-result `data` key a recorded action's outcome rides on: the machine facts about what the
|
|
# call achieved (`requested_url`, `url`, `http_status`, `page_transitioned`, `navigation_dead_end`),
|
|
# carried verbatim to `RoundAction.outcome` for the caller to persist on the action row. Internal to
|
|
# the loop like the two keys above -- never shown to the model, which reads the tool's own content.
|
|
ACTION_OUTCOME_DATA_KEY = "action_outcome"
|
|
# An HTTP status at or above this reached no usable page, so the action row reads as failed even
|
|
# though the tool honestly returned ok (the model still gets the status and decides what to do).
|
|
ACTION_OUTCOME_FAILED_HTTP_STATUS = 400
|
|
# How long a call spent turning an address into a target, before the act. A context variable rather
|
|
# than a field on the result, because the cohort this exists to price is the one where the handler
|
|
# RAISES -- a driver timeout on a resolved target -- and a result the handler never returned cannot
|
|
# carry anything. Set before the handler runs, so the value survives whichever way the call ends.
|
|
# Accumulated, because the mark wrapper resolves an address and the ref wrapper then runs inside it.
|
|
#
|
|
# WHAT ONE ROW MEANS, written here because a chart of this by `selector_kind` reads the field and
|
|
# cannot see the branch that produced it:
|
|
# absent no address was supplied. Its only meaning -- never "measured, and it was zero".
|
|
# ref / mark the server-side table lookup and the frame routing it implies: the persistent-ref
|
|
# model's own addressing cost, which is what this field exists to price.
|
|
# css the frame ROUTING only: which realm a typed selector resolves in.
|
|
# Working-page acquisition is excluded on every branch: every design has to get the page, so it is
|
|
# not a cost of the addressing model.
|
|
_RESOLVE_SECONDS: ContextVar[float | None] = ContextVar("taskv3_resolve_seconds", default=None)
|
|
# Which case behind the click reach-probe's boolean applied, for the one click call this context covers.
|
|
# A context variable rather than a result field because `click` returns from many places, several of
|
|
# them after the probe has already answered -- the same reason `_RESOLVE_SECONDS` lives here.
|
|
_HIT_CLASS: ContextVar[dict[str, Any] | None] = ContextVar("taskv3_hit_class", default=None)
|
|
# Which `covered` message the refusal rendered, for the one tool call this context covers. A context
|
|
# variable for the same reason `_HIT_CLASS` is one: the covered message is built in a shared helper
|
|
# five call sites reach, several of them after the probe has already answered.
|
|
_COVERED_LAYER: ContextVar[dict[str, Any] | None] = ContextVar("taskv3_covered_layer", default=None)
|
|
# The text-delta read's cost and yield for the one tool call this context covers.
|
|
_TEXT_DELTA: ContextVar[tuple[float, int | None, int, bool, str | None, int] | None] = ContextVar(
|
|
"taskv3_text_delta", default=None
|
|
)
|
|
# The ordinal of the tool call being dispatched, so a tool can say how many calls a report spans.
|
|
_TOOL_CALL_SEQ: ContextVar[int | None] = ContextVar("taskv3_tool_call_seq", default=None)
|
|
|
|
|
|
def record_resolve_seconds(elapsed: float) -> None:
|
|
"""Add `elapsed` to this tool call's address-resolution time. Telemetry only."""
|
|
_RESOLVE_SECONDS.set((_RESOLVE_SECONDS.get() or 0.0) + elapsed)
|
|
|
|
|
|
def record_hit_class(
|
|
hit_class: str,
|
|
*,
|
|
needed: bool,
|
|
probe_seconds: float | None = None,
|
|
isolated: bool | None = None,
|
|
raised: bool = False,
|
|
) -> None:
|
|
"""Record what this click's reach probe returned, the decision it produced, and what it cost.
|
|
|
|
`needed` is on the record because the class alone cannot recover it: a shadow-rooted target
|
|
answers `unknown` with needed=TRUE, so `unknown` would otherwise pool the rows that bought the
|
|
second probe with the rows that answered nothing. `disabled` is answered before any hit test runs;
|
|
every other `hit_class` value names what the hit test RETURNED, never why — nothing in that probe
|
|
tells a real occluder from a hit that fell through, since a modal scrim drawn as `body::before`
|
|
hit-tests as body and still blocks the click.
|
|
`isolated` says which realm answered -- realms, not costs, since its `False` side spans both the
|
|
cheapest fallback and the most expensive retry storm. `raised` marks a probe that threw, whose
|
|
duration is a blow-up rather than a cost. The RECORDED copies are inert: the click path branches on
|
|
its own `needed`, never on anything read back from here.
|
|
"""
|
|
_HIT_CLASS.set(
|
|
{
|
|
"hit_class": hit_class,
|
|
"needed": needed,
|
|
"probe_seconds": probe_seconds,
|
|
"isolated": isolated,
|
|
"raised": raised,
|
|
}
|
|
)
|
|
|
|
|
|
def record_text_delta(
|
|
seconds: float,
|
|
lines: int | None,
|
|
chars: int = 0,
|
|
*,
|
|
over_bound: bool = False,
|
|
skipped: str | None = None,
|
|
pending: int = 0,
|
|
) -> None:
|
|
"""Record the text-delta read. Telemetry only. `lines` is None when the read failed, timed out or
|
|
was past the char bound (`over_bound`), or when the tool errored after a read before it ran; else
|
|
how many lines the result reported before the display cap. `chars` is the section's length.
|
|
`skipped` names why a call did not read at all: "over_bound" (this document was already read past
|
|
the bound) or "unreadable" (its reads failed twice in a row).
|
|
`pending` is how many lines the page showed before the call ran (read before a repeated call): part
|
|
of `lines` when the call's own read succeeded, and still reported when it did not (`lines` None)."""
|
|
_TEXT_DELTA.set((seconds, lines, chars, over_bound, skipped, pending))
|
|
|
|
|
|
def current_tool_call_seq() -> int | None:
|
|
"""The loop's ordinal for the tool call being dispatched; None outside the loop."""
|
|
return _TOOL_CALL_SEQ.get()
|
|
|
|
|
|
def record_covered_layer(branch: CoveredBranch, *, controls: int, layer_kind: CoveredLayerKind) -> None:
|
|
"""Record which `covered` message this call rendered. Telemetry only, never a behaviour change.
|
|
|
|
`controls` is how many controls the MESSAGE named, not how many the probe found: a control with
|
|
neither a selector nor a label is dropped from the message, and the existing eight-slot
|
|
truncation caps what the model is handed. So this is the count the model acted on. WHAT A
|
|
DENOMINATOR MEANS HERE: zero is not one event. Under `unnamed` it means there was no layer to
|
|
enumerate; under `clipped` it is zero always, because a container is not a layer and its
|
|
controls are never the thing to act on; under `named` it means the layer had no control the
|
|
enumeration could name. Cut on `tool_error_class:covered`, then group by `covered_branch`, and
|
|
read the count within a branch.
|
|
"""
|
|
_COVERED_LAYER.set({"branch": branch, "controls": int(controls), "layer_kind": layer_kind})
|
|
|
|
|
|
# What observe() prints and the model hands back, BYTE-IDENTICAL in both directions: the digest
|
|
# prints `ref=12` and that exact string is the selector argument, so "copy it as printed" has one
|
|
# reading. Deliberately not an attribute-selector shape and deliberately not bracketed -- the ref
|
|
# namespace is a server-side table, and accepting a `[ref=12]` form would let a page that authors a
|
|
# `ref` attribute collide with it, which is the wrong-element class this addressing exists to close.
|
|
# It lives here rather than in tools.py because both modules classify against it and a second copy
|
|
# would drift.
|
|
REF_SELECTOR_RE = re.compile(r"^\s*ref=(\d+)\s*$")
|
|
|
|
|
|
DriverTimeoutPredicate = Callable[[BaseException], bool]
|
|
|
|
|
|
def _no_driver_here(exc: BaseException) -> bool:
|
|
return False
|
|
|
|
|
|
# Whether an exception is the browser driver's OWN timeout. Injected rather than imported, because
|
|
# the driver package ships only in the `local`/`server` extras: importing it at this module's scope
|
|
# would make the loop unimportable in a base install and would break the scripted-fake unit testing
|
|
# the module docstring promises. `tools.py` -- the only module that can build a handler capable of
|
|
# raising a driver error -- installs the real predicate as it imports, so the driver cohort cannot
|
|
# read empty while a browser tool exists to fill it.
|
|
_is_driver_timeout: DriverTimeoutPredicate = _no_driver_here
|
|
|
|
|
|
def set_driver_timeout_predicate(predicate: DriverTimeoutPredicate) -> None:
|
|
global _is_driver_timeout
|
|
_is_driver_timeout = predicate
|
|
|
|
|
|
def _raised_error_class(exc: BaseException) -> ToolErrorClass:
|
|
"""Classify an exception that escaped a tool handler, for the failure-cost read.
|
|
|
|
A bare `TimeoutError` is NOT evidence of a driver timeout: `file_upload` awaits a source fetch
|
|
that raises one, and on 3.11 `asyncio.TimeoutError` IS `builtins.TimeoutError`. Folding those
|
|
into the driver cohort would corrupt the very number this field exists to produce.
|
|
|
|
The driver class is decided by the installed predicate -- `is_driver_timeout_error`, which
|
|
recognises BOTH Playwright-family packages -- and never by the exception's own module name: the
|
|
browser image rewrites this repository's driver imports to the fork, so a module test would
|
|
recognise driver timeouts only where the fork is absent -- i.e. nowhere that matters -- and
|
|
report the cohort as empty in production. Nothing here reads the exception's MESSAGE: a message
|
|
can carry page text and this field is indexed.
|
|
"""
|
|
if _is_driver_timeout(exc):
|
|
return "driver_timeout"
|
|
if isinstance(exc, TimeoutError):
|
|
return "timeout_other"
|
|
return "handler_raised"
|
|
|
|
|
|
def mark_is_filler(mark: Any) -> bool:
|
|
# look() numbers marks from 1, so 0 addresses nothing: it is an upstream filling the optional slot,
|
|
# not the model addressing by mark. `to_openai_tool` sets the `strict` key that provokes that, so
|
|
# this is the tolerance for an upstream that fills it anyway.
|
|
return mark == 0 and not isinstance(mark, str)
|
|
|
|
|
|
def _selector_kind(args: dict[str, Any]) -> str:
|
|
"""How the model addressed its target on this call, read off the ARGS AS SENT.
|
|
|
|
Must be taken before dispatch: the ref and act-by-mark wrappers rewrite `args["selector"]` in
|
|
place, so the same read afterwards reports the resolved address rather than the one the model
|
|
chose. It is the cut for `resolve_seconds`, but the two cohorts are NOT symmetric and the
|
|
contract above `_RESOLVE_SECONDS` says how: a `css` row bounds frame routing only.
|
|
"""
|
|
if args.get("mark") is not None and not mark_is_filler(args.get("mark")):
|
|
return "mark"
|
|
selector = args.get("selector")
|
|
# `not selector`, matching what the wrappers themselves treat as absent. Note this does NOT make
|
|
# the record self-consistent: `selector_present` is read after dispatch, and act-by-mark leaves
|
|
# the selector it resolved in `args`, so a mark call logs kind=mark WITH selector_present=True.
|
|
# The two fields describe different moments on purpose -- read `selector_kind` for what the model
|
|
# sent.
|
|
if not isinstance(selector, str) or not selector:
|
|
return "none"
|
|
return "ref" if REF_SELECTOR_RE.match(selector) else "css"
|
|
|
|
|
|
class RoundAction(NamedTuple):
|
|
"""One dispatched page action, as the caller persists it."""
|
|
|
|
tool: str
|
|
args: dict[str, Any]
|
|
succeeded: bool
|
|
# The target's page-visible name, read off the element before the action ran. None when the tool
|
|
# names no element, when the target has no readable name, or when the probe could not run.
|
|
target_name: str | None = None
|
|
# The target's role/type, from a fixed vocabulary (see target_label.py) -- computed the same time
|
|
# as target_name and independent of whether a name was found.
|
|
target_kind: str | None = None
|
|
# Whether this action consumed a budget unit. Carried rather than re-derived from the tool name
|
|
# downstream: the loop already read it off the ToolSpec, and a second name list is a second place
|
|
# for a new tool to be missing from.
|
|
billable: bool = False
|
|
# What the action achieved, as the tool itself reported it (see ACTION_OUTCOME_DATA_KEY): machine
|
|
# facts only, for the caller to persist alongside the verb. None when the tool reported none --
|
|
# a call that errored before it reached a page has only its `error` below to offer.
|
|
outcome: dict[str, Any] | None = None
|
|
# The tool's own error text when the call failed, which is all a call that never reached a page
|
|
# can say about itself -- a refusal the engine issued on purpose reads as an unexplained failed
|
|
# row without it. Model-facing prose, so a caller that persists it must redact and cap it.
|
|
error: str | None = None
|
|
|
|
|
|
# A probe consulted after a billable/download-signaling tool result; a truthy return ends the run as
|
|
# completed with that reason, without the model ever calling finish. A blocker consulted from
|
|
# finish(completed) itself; a truthy return rejects that verdict with the message as the reason. Both
|
|
# receive the basenames tools staged into the downloads dir this run, to exclude from detection.
|
|
CompletionProbe = Callable[[frozenset[str]], Awaitable[str | None]]
|
|
CompletionBlocker = Callable[[frozenset[str]], Awaitable[str | None]]
|
|
# Consulted from finish for EVERY verdict, unlike CompletionBlocker: it takes the finish status and
|
|
# gates on state the caller already tracks (e.g. a verification-code budget), not on staged
|
|
# downloads. One callback, because a completed claim made on a blank verification step and a
|
|
# non-complete verdict given up with polling budget still unspent are the same concern seen from two
|
|
# sides; splitting them into two hooks would scatter it.
|
|
VerificationBlocker = Callable[[str], Awaitable[str | None]]
|
|
# The status a re-ask conversion asks the verification gate about: a completion the model never claimed.
|
|
CONVERSION_VERIFICATION_STATUS = "converted"
|
|
|
|
|
|
@dataclass
|
|
class ToolSpec:
|
|
name: str
|
|
description: str
|
|
parameters: dict[str, Any]
|
|
handler: ToolHandler
|
|
terminal: bool = False
|
|
billable: bool = False # a page-mutating browser action that meters like a step-engine action
|
|
recordable: bool = False # persisted as an action row (with screenshot) but not billed/budgeted
|
|
compactable: bool = False # a large perception result safe to elide from the transcript once superseded
|
|
# Its result hands out handles (observe refs, look marks) that the tool's NEXT call disposes and renumbers,
|
|
# so an older result is not merely older but wrong to act on. Compaction keeps exactly one of these,
|
|
# whatever the page showed; a result without handles stays usable and is superseded only by the same content.
|
|
issues_handles: bool = False
|
|
# Drives the live page without being billed or recorded. Read ONLY as engagement ("did this run
|
|
# touch the page at all"), never as budget or artifact parity -- those are `billable` and
|
|
# `recordable`, and widening either to cover this would change action rows and the step budget.
|
|
# Exists because `open_verification_link` calls page.goto() while carrying neither flag, so an
|
|
# engagement predicate built on those two alone calls a run that navigated "never tried".
|
|
engages_page: bool = False
|
|
# Whether a call's target carries the toggle state observe prints (checked/pressed). Consulted only
|
|
# to exempt a click from the post-failure submit skip; set by the browser tools on click alone.
|
|
toggle_probe: Callable[[dict[str, Any]], Awaitable[bool]] | None = None
|
|
|
|
@property
|
|
def touches_page(self) -> bool:
|
|
return self.billable or self.recordable or self.engages_page
|
|
|
|
def to_openai_tool(self) -> dict[str, Any]:
|
|
return {
|
|
"type": "function",
|
|
"function": {
|
|
"name": self.name,
|
|
"description": self.description,
|
|
"parameters": self.parameters,
|
|
# Stated rather than left unset because OpenRouter reads an unset `strict` as license
|
|
# to fill every declared property with a type default (`mark: 0`, `selector: ""`),
|
|
# which the act wrappers then refuse. `True` is not the alternative: these schemas
|
|
# are not strict-shaped and it 400s. See GOTCHAS.md "Loop" for the measured matrix.
|
|
"strict": False,
|
|
},
|
|
}
|
|
|
|
|
|
@dataclass
|
|
class LoopOutcome:
|
|
status: Literal["completed", "failed", "terminated", "budget_exhausted", "loop_error", "canceled"]
|
|
reason: str
|
|
# Which loop guard authored this verdict (`PERCEPTION_STALL_GUARD` and its siblings), or None for a
|
|
# terminal the model or a budget produced. This is the machine-readable class that used to be a
|
|
# prefix on `reason`; `reason` itself is customer-facing text (see `_guard_verdict`).
|
|
guard: str | None = None
|
|
extracted_output: Any = None
|
|
# The raw budget-cap string (e.g. "max_turns (40) reached") that granted this run its one final
|
|
# observed turn, carried on whatever outcome the final turn produces — a finish verdict or, if
|
|
# the model didn't finish, the honest budget_exhausted exit. None for a run that never tripped a cap.
|
|
cap_trip: str | None = None
|
|
turns: int = 0
|
|
tool_calls: int = 0
|
|
action_steps: int = 0
|
|
# Wall-clock spent inside tool handlers, summed over the run. Serial by construction, so it is
|
|
# directly comparable against the run's total duration.
|
|
tool_seconds: float = 0.0
|
|
# Turns where the model answered with prose instead of a tool call, costing a full round trip
|
|
# plus the NO_TOOL_CALL_NUDGE recovery turn.
|
|
no_tool_call_turns: int = 0
|
|
# Whether tool_choice was still being sent when the run ended. Distinguishes a run that was
|
|
# asked to force tool calls from one where the request was degraded away mid-run.
|
|
tool_choice_in_effect: bool = False
|
|
billable_actions: list[str] = field(default_factory=list)
|
|
# Perception snapshots are compacted in place during the run, so superseded observe/get_html
|
|
# content is already elided here — treat as lossy if ever persisted for audit.
|
|
messages: list[dict[str, Any]] = field(default_factory=list)
|
|
# The progress signals for the caller's one terminal record, set on every path out of the loop.
|
|
# The None default only covers an outcome built by someone else; the caller still writes its
|
|
# record and simply carries no progress fields on it.
|
|
telemetry: TerminalTelemetry | None = None
|
|
goal_check: dict[str, Any] | None = None
|
|
# The finish-gate goal check ended this run; its verdict discards the model's output, so the
|
|
# final-turn stamp must not refill it.
|
|
goal_check_ended: bool = False
|
|
# Set when the unlisted-outcome re-ask turned the model's failed/terminated finish into completed; a
|
|
# post-loop veto of that completion restores this status and reason instead of failing the task.
|
|
converted_from: NonCompletedStatus | None = None
|
|
converted_from_reason: str = ""
|
|
unlisted_reask: dict[str, Any] | None = None
|
|
|
|
|
|
def _get(obj: Any, key: str, default: Any = None) -> Any:
|
|
# raw_response=True returns a model_dump() dict, but test fakes and some
|
|
# providers hand back objects — accept either shape.
|
|
if isinstance(obj, dict):
|
|
return obj.get(key, default)
|
|
return getattr(obj, key, default)
|
|
|
|
|
|
def _extract_message(response: Any) -> Any:
|
|
choices = _get(response, "choices") or []
|
|
if not choices:
|
|
return None
|
|
return _get(choices[0], "message")
|
|
|
|
|
|
def _extract_text(response: Any) -> str | None:
|
|
message = _extract_message(response)
|
|
if message is None:
|
|
return None
|
|
return _get(message, "content")
|
|
|
|
|
|
def _extract_reasoning_summary(response: Any) -> str | None:
|
|
"""The provider's reasoning summary (message.reasoning_content), when litellm's chat->responses
|
|
bridge returned one -- surfaces the provider's reasoning summary where available; a tool call
|
|
with empty message.content on a continuation turn often has none either."""
|
|
message = _extract_message(response)
|
|
if message is None:
|
|
return None
|
|
return _get(message, "reasoning_content")
|
|
|
|
|
|
def _extract_tool_calls(response: Any) -> list[tuple[str, str, dict[str, Any]]]:
|
|
message = _extract_message(response)
|
|
if message is None:
|
|
return []
|
|
raw_tool_calls = _get(message, "tool_calls") or []
|
|
tool_calls: list[tuple[str, str, dict[str, Any]]] = []
|
|
for raw in raw_tool_calls:
|
|
function = _get(raw, "function") or {}
|
|
name = _get(function, "name")
|
|
if not name:
|
|
continue
|
|
tool_call_id = _get(raw, "id") or f"call_{len(tool_calls)}"
|
|
arguments = _get(function, "arguments")
|
|
if isinstance(arguments, str):
|
|
try:
|
|
parsed_args = json.loads(arguments) if arguments else {}
|
|
except json.JSONDecodeError:
|
|
parsed_args = {}
|
|
elif isinstance(arguments, dict):
|
|
parsed_args = arguments
|
|
else:
|
|
parsed_args = {}
|
|
tool_calls.append((tool_call_id, name, parsed_args))
|
|
return tool_calls
|
|
|
|
|
|
NO_TOOL_CALL_NUDGE = (
|
|
"You did not call a tool. Call a browser tool to make progress, or call "
|
|
"finish(status, reason, extracted_output) if the goal is complete. Emit a tool call now."
|
|
)
|
|
|
|
# Perception-stall policy: N consecutive identical (marker-canonicalized) snapshots from the same (compactable) tool
|
|
# AND from the same probe (that tool with the same arguments) mean the page has stopped changing in
|
|
# response to actions, and a page gated by something the run cannot perceive or operate otherwise
|
|
# burns the whole budget on identical re-observes. The per-probe term is what keeps two different
|
|
# probes returning the same string from reading as one frozen page. Only compactable tools count:
|
|
# action tools legitimately return the same string every call ("waited"), so they can never witness
|
|
# "the page is unchanged".
|
|
PERCEPTION_STALL_NUDGE_AFTER = 6
|
|
PERCEPTION_STALL_TERMINATE_AFTER = 15
|
|
|
|
# SKY-17222. Read bodies (tools without handles) the model has already seen are retained by SIZE, not
|
|
# count: once they sum past HIGH, the oldest are elided down to LOW in one rewrite (an epoch). HIGH is
|
|
# the previous ceiling (two get_html windows, `tools.HTML_MAX_CHARS` each), so the worst case does not
|
|
# move. Between epochs the transcript only grows at the end, which is what a provider prefix cache
|
|
# needs: it hits only when the whole previous request is a byte prefix of the next one.
|
|
PERCEPTION_RETAIN_CHARS_LOW = 20_000
|
|
PERCEPTION_RETAIN_CHARS_HIGH = 2 * PERCEPTION_RETAIN_CHARS_LOW
|
|
# However large they are, the newest reads the model has seen that an epoch keeps: two paged windows.
|
|
PERCEPTION_RETAIN_MIN_READS = 2
|
|
# A same-bytes marker replaces a body only when it saves most of it; on a short body it costs more.
|
|
PERCEPTION_MARKER_MIN_RATIO = 4
|
|
|
|
# The stall verdict's machine class. Stable and facetable — telemetry keys on it to measure how often
|
|
# the policy fires — and it travels beside the verdict (`LoopOutcome.guard`, the guard's log line, the
|
|
# engine's terminal record), never inside its text: a verdict's `reason` is persisted verbatim as the
|
|
# customer-facing `task.failure_reason`, so a dashboard facet in it is a facet the customer reads
|
|
# (SKY-16271). Change it only with the dashboards that read it.
|
|
PERCEPTION_STALL_GUARD = "perception_stall"
|
|
|
|
# Action-loop policy: N repeated executions of the same billable action (same tool + same args)
|
|
# with no new evidence the page changed mean the run is re-trying against an unchanged outcome —
|
|
# the live shape is re-submitting into the same rejection banner, which the stall policy cannot
|
|
# see because interleaved actions and varied probes keep the perception stream changing while the
|
|
# SITUATION stays the same. Evidence of change is a REPEATED probe (same tool + same args)
|
|
# returning different content, or a download landing; a first-time probe has no baseline and is
|
|
# evidence of nothing, so varied-selector probing cannot launder repetition into "progress".
|
|
ACTION_LOOP_NUDGE_AFTER = 3
|
|
# 8, not 6: RAW repeats of one action key peak at 6 across 50 completed prod runs while stuck runs
|
|
# reach 36, so 6 sat on the completed population's edge. DO NOT LOWER THIS. The effective
|
|
# post-clearing counter — what this constant is actually compared against — was then measured by
|
|
# replaying the clearing rule over per-call telemetry, and NO SEPARATION WAS OBSERVED: two completed
|
|
# runs peaked at 2 and 4, four stuck runs at 1, 3, 3 and below (upper bounds, computed identically
|
|
# on both sides), and the highest observed value fell in a COMPLETED run. n=6 is far too small to
|
|
# establish inversion as a property of either population — that spread is also consistent with
|
|
# noise — but it is no evidence FOR a lower threshold either, and lowering a safety threshold
|
|
# requires positive evidence. 8 is above everything observed, which is why the change is safe; it
|
|
# is also why nothing in the sample fires. REVISIT if the effective distribution ever becomes
|
|
# measurable at population scale; do not lower it before then. Repeat count may not be a stuck-ness
|
|
# signal at all: SKY-15602 tracks the real problem, which is that the loop has no definition of
|
|
# goal progress, only of page change.
|
|
ACTION_LOOP_TERMINATE_AFTER = 8
|
|
|
|
# Facetable sibling of PERCEPTION_STALL_GUARD; same dashboard contract.
|
|
ACTION_LOOP_GUARD = "action_loop"
|
|
|
|
# Progress-gated action-step budget extension (SKY-15264): a run that hits its action-step cap while
|
|
# the page is still demonstrably changing (a repeated probe returned fresh content, a navigation or
|
|
# download landed) earns an extension of half the original cap — the observed failure population is
|
|
# genuinely long multi-page forms dying mid-progress, while a stalled run must be refused exactly as
|
|
# before. Evidence must be at most this many action rounds old: staler change evidence says nothing
|
|
# about the run's current state.
|
|
ACTION_BUDGET_EXTENSION_EVIDENCE_WINDOW = 8
|
|
# A long form does not stop being long because one extension already fired: 41.4% of long-block
|
|
# step-cap deaths die at the ALREADY-EXTENDED cap (SKY-15666). So the grant repeats under the same
|
|
# evidence predicate, bounded two ways. Each grant is half the ORIGINAL cap, never half the current
|
|
# one: a run that has already spent 1.5x its budget must not draw a proportionally larger handout,
|
|
# and constant-size grants keep every extension priced the same and re-earned on fresh evidence.
|
|
# Total growth stops at this multiple of the original cap. The wall clock — the one cap that is not
|
|
# a function of the action-step budget — is the outer runaway stop and is never re-derived.
|
|
ACTION_BUDGET_EXTENSION_MAX_FACTOR = 3
|
|
# Refusals left uncharged in a row before a refusal is charged like any failed call. The repeat guard cannot
|
|
# bound them (a stale address resets it), so this does.
|
|
UNCHARGED_REFUSAL_GRACE = ACTION_LOOP_NUDGE_AFTER
|
|
# Facetable event names — both the grant and the refusal are queryable so the gate's decision
|
|
# precision is measurable on the canary; change only with the dashboards that read them.
|
|
ACTION_BUDGET_EXTENDED_EVENT = "taskv3 loop action budget extended"
|
|
ACTION_BUDGET_EXTENSION_REFUSED_EVENT = "taskv3 loop action budget extension refused"
|
|
BATCH_SKIP_TOGGLE_EXEMPT_EVENT = "taskv3 loop batch skip exempted toggle click"
|
|
CREDENTIAL_RESUBMIT_REFUSED_EVENT = "taskv3 loop credential resubmit refused"
|
|
EXTRACTION_ENTRY_REFUSED_EVENT = "taskv3 loop extraction entry refused"
|
|
# Every tool that authors input on the page. Defined here rather than in tools.py because tools.py imports
|
|
# this module; the extraction-block refusal and tools.py's frame-work ledger both read this one set.
|
|
FILL_TOOLS = frozenset({"type", "select_option", "select_combobox", "file_upload"})
|
|
# The arguments that carry what a fill tool typed or chose, per the tool schemas in tools.py. Selectors and
|
|
# refs are excluded: they name a field, and a field's label on the page is not a value the block entered.
|
|
ENTERED_VALUE_ARGS: dict[str, tuple[str, ...]] = {
|
|
"type": ("text",),
|
|
"select_option": ("value", "label", "values", "labels"),
|
|
"select_combobox": ("value", "search"),
|
|
}
|
|
# A wrap-up turn granted by a guard that a later budget extension raised past its trip; facetable
|
|
# so a released latch is distinguishable from one that never fired.
|
|
FINAL_TURN_RELEASED_EVENT = "taskv3 loop final turn grant released by budget extension"
|
|
|
|
# Final-turn grant (budget-exhaustion final turn): a budget cap trip buys one more unconstrained
|
|
# model turn instead of ending the run immediately, so a model mid-extraction can still call finish
|
|
# with what it has. NOT once per run since SKY-15666: an extension that relieves the latching guard
|
|
# releases the latch and re-arms it for a later trip, so this fires up to one more time than the run
|
|
# had grants. Facetable — change only with the dashboards that read it.
|
|
FINAL_TURN_GRANTED_EVENT = "taskv3 loop final turn granted"
|
|
|
|
# Page-state stall policy (SKY-15265): rounds of billable batches that left the page fingerprint
|
|
# byte-identical, with no page-change flag, mean the run is cycling on a frozen document in shapes
|
|
# the per-tool guards cannot see (varied probes never streak; scroll/wait carry no digest at all).
|
|
# Completions of the worst-affected canary block finish in <=9 rounds; the nudge re-plans the model
|
|
# once before the verdict.
|
|
PAGE_STATE_STALL_NUDGE_AFTER = 8
|
|
PAGE_STATE_STALL_TERMINATE_AFTER = 12
|
|
# The verdict is SHADOW-ONLY for now: the fingerprint is blind to work inside iframes (main-frame
|
|
# innerHTML only), so a live termination could kill healthy embedded-widget runs. The nudge ships
|
|
# live (benign direction); the shadow event measures the would-terminate precision on the canary,
|
|
# and promotion to a live verdict is a separate release decision on that data.
|
|
PAGE_STATE_STALL_SHADOW_EVENT = "taskv3 loop page state stall would terminate"
|
|
# Facetable sibling of PERCEPTION_STALL_GUARD; reserved for the future live verdict, which owes a
|
|
# customer-facing sentence of its own the day it stops being shadow-only.
|
|
PAGE_STATE_STALL_GUARD = "page_state_stall"
|
|
|
|
# Hard "the resource does not exist / is gone" HTTP statuses. A navigation landing on one of these is
|
|
# a genuine non-capability dead-end (a dead or removed posting), which v1 routes to `terminated`. Both
|
|
# the in-loop `navigate` tool and the pre-loop initial-URL navigation classify against this set. NARROW
|
|
# on purpose: auth (401/403), rate-limit (429) and transient server errors (5xx) are recoverable or
|
|
# capability failures, not dead-ends, and are left to the model / stay `failed`.
|
|
NAVIGATION_DEAD_END_STATUSES = frozenset({404, 410})
|
|
|
|
|
|
class _NavDeadEnd(NamedTuple):
|
|
"""A navigation that landed on a hard dead-end status, and the URL it landed on when the tool
|
|
reported one."""
|
|
|
|
status: int
|
|
url: str | None
|
|
|
|
|
|
# Defined here (not tools.py) so the batch-dispatch poisoning check below can compare against it
|
|
# without an import cycle -- tools.py already imports ToolResult/ToolSpec from this module.
|
|
PAGE_UNAVAILABLE_ERROR = "browser page unavailable"
|
|
|
|
# A navigation landed on a hard dead-end (HTTP 404/410): the target posting does not exist or was
|
|
# removed, so the goal cannot be completed there. Ends the run as `terminated`, matching v1's terminate
|
|
# verdict for the same condition. Covers both the in-loop `navigate` tool and the pre-loop initial-URL
|
|
# navigation. Facetable sibling of the classes above.
|
|
NAV_DEAD_END_GUARD = "navigation_dead_end"
|
|
# A page-level handler kept asking for the page to be reloaded past the per-run cap: the page cannot
|
|
# be stabilized, and acting on it would mean acting on a page declared stale.
|
|
PAGE_REFRESH_EXHAUSTED_GUARD = "page_refresh_exhausted"
|
|
|
|
# A URL long enough to be a payload rather than a place is DROPPED from a verdict, never cut: the
|
|
# customer-facing reason is redacted by whole-value match, so a truncated URL would no longer match a
|
|
# registered secret it was cut from (the rule `clean_target_label` applies to an element name). The cap
|
|
# is the handoff one rather than a second number: both answer "how much URL may leave the run".
|
|
VERDICT_URL_MAX_CHARS = MAX_HANDOFF_URL_CHARS
|
|
|
|
# Emitted, never acted on, when the oscillation rule WOULD have terminated. The step engine's
|
|
# tripwires (skyvern/forge/sdk/fail_fast/shadow.py) earn the right to act by publishing this event
|
|
# first and deriving a decision precision from it; a rule that ADDS terminations gets the same
|
|
# treatment rather than being trusted because it looks right.
|
|
PERCEPTION_STALL_SHADOW_EVENT = "taskv3 loop perception stall would fire"
|
|
|
|
# Emitted once per run, where the argument-blind per-tool counter first reached the terminator and
|
|
# the per-probe term did not. It marks a verdict withheld, not a run spared: most such runs end on
|
|
# another guard, so read it joined to the final status.
|
|
PERCEPTION_STALL_SUPPRESSED_EVENT = "taskv3 loop perception stall suppressed_main_fire"
|
|
|
|
# Guard-attribution hashes: sha256 over a per-run secret salt, truncated to 64 bits — enough to join a
|
|
# firing to the value the guard compared within one run (cardinality ≤ max_tool_calls), while the salt
|
|
# keeps a low-entropy input (a short typed text, a selector) non-enumerable, which truncation alone does not.
|
|
TELEMETRY_HASH_HEX_LEN = 16
|
|
|
|
|
|
def telemetry_hash(salt: str, *parts: str) -> str:
|
|
return hashlib.sha256("\x1f".join((salt, *parts)).encode()).hexdigest()[:TELEMETRY_HASH_HEX_LEN]
|
|
|
|
|
|
# The value shape observe()'s enrichment mints ('t' + monotonic counter, optional '-<n>'
|
|
# disambiguator — tools._OBSERVE_JS): identity handles, not page semantics, and a node-replacing
|
|
# framework re-mints them on every read, so hashed raw they hide a frozen page from the stall
|
|
# guard. Page-authored data-tv3 values (any other shape) are page content and stay significant,
|
|
# as is a page-authored data-tv3-ref: observe addresses by a server-held ref, never by an
|
|
# attribute, so nothing in the markup with that name is a handle this engine minted.
|
|
_TV3_MARKER_VALUE_RE = re.compile(r'data-tv3="t\d+(?:-\d+)?"')
|
|
_TV3_PICK_VALUE_RE = re.compile(r'data-tv3-pick="[^"]*"')
|
|
|
|
# get_html truncates to a fixed budget before the loop ever sees the content, so a marker the cut
|
|
# leaves open at the tail has no closing quote for the pattern above and its churning digits would
|
|
# be the one leak that survives canonicalization. The lookahead assumes the truncation notice itself
|
|
# carries no quote character, and this sub must run AFTER closed markers are rewritten to the
|
|
# quote-bearing placeholder — either broken silently brings the leak back.
|
|
_TV3_MARKER_CUT_RE = re.compile(r'(?:data-tv3="t\d*(?:-\d*)?|data-tv3-pick="[a-f0-9]*)(?=[^"]*\Z)')
|
|
|
|
# A read can now start at an offset, so a marker can be cut open at the HEAD of a window too. Same
|
|
# leak, same canonicalization, opposite end — but the boundary can land at ANY of the sixteen
|
|
# characters of `data-tv3="tN"`, not only inside the digits, so the pattern is every suffix of the
|
|
# attribute's fixed prefix (longest first) plus the empty one for a cut inside the value. Anchored at
|
|
# the very start of the content, so the only thing it can take is the fragment a window made.
|
|
|
|
|
|
# observe prints its own address as `ref=N` at the head of each element line. The number is
|
|
# engine-minted identity, not page semantics: a framework that remounts a control between readings
|
|
# gives the replacement a new one, so hashed raw they hide a semantically frozen page from the stall
|
|
# guard -- the same reason the marker values above are canonicalized. Anchored at line start, so it
|
|
# can only take the address, never a value further along the line; the caller scopes it to
|
|
# observe's own payload so page-authored bytes from get_html are never subject to it.
|
|
_TV3_REF_ADDRESS_RE = re.compile(r"^ref=\d+", re.MULTILINE)
|
|
|
|
|
|
_PERCEPTION_URL_LINE_RE = re.compile(r"^url=\S+", flags=re.MULTILINE)
|
|
|
|
|
|
def _fold_reported_spans(content: str, spans: Sequence[Sequence[int]] | None, placeholder: str) -> str:
|
|
# Spans the TOOL reported, as offsets into `content` as returned. Anything out of range or overlapping
|
|
# is not ours to trust and is kept.
|
|
pieces: list[str] = []
|
|
cursor = 0
|
|
for span in sorted(list(s) for s in spans or () if len(s) == 2 and all(type(i) is int for i in s)):
|
|
start, end = span
|
|
if cursor <= start < end <= len(content):
|
|
pieces.append(content[cursor:start])
|
|
pieces.append(placeholder)
|
|
cursor = end
|
|
pieces.append(content[cursor:])
|
|
return "".join(pieces)
|
|
|
|
|
|
def _canonical_perception_content(
|
|
content: str,
|
|
*,
|
|
is_observe: bool = False,
|
|
head_fragment_len: int = 0,
|
|
notice_at: int | None = None,
|
|
clip_spans: Sequence[Sequence[int]] | None = None,
|
|
delta_at: int | None = None,
|
|
delta_end: int | None = None,
|
|
) -> str:
|
|
# The newly-shown-text section at [delta_at, delta_end) reports what changed since an earlier call,
|
|
# not the page's state, so it is cut first; every other reported span lies before it.
|
|
if type(delta_at) is int and 0 <= delta_at <= len(content):
|
|
end = delta_end if type(delta_end) is int and delta_at <= delta_end <= len(content) else len(content)
|
|
content = content[:delta_at] + content[end:]
|
|
# observe's clip counts ("…[+N chars]") annotate what the display cut; a page whose clipped tail
|
|
# changes has not changed what the model can see, so the count is folded at the offsets observe
|
|
# reported. That a field is clipped at all stays: one that now fits whole has changed. First, because
|
|
# every other pass here shifts offsets -- which holds only while no tool reports clip spans together
|
|
# with notice_at or head_fragment_len.
|
|
content = _fold_reported_spans(content, clip_spans, " …[+*]")
|
|
# The ref pass is scoped to observe's own payload, not to every compactable result: get_html
|
|
# returns page-authored bytes, and a page can write a line that opens `ref=<digits>` there. The
|
|
# marker passes below are attribute-shaped and page-authored values in that shape stay significant.
|
|
addressed = _TV3_REF_ADDRESS_RE.sub("ref=*", content) if is_observe else content
|
|
# Before the marker passes: the notice is what the tail lookahead below scans through, and a
|
|
# window's head fragment is what the value pattern cannot close.
|
|
# Only for a read the TOOL said it cut, and only the LAST match — which is then provably the
|
|
# notice it appended, since our notice comes after all of the window's page content. Folded
|
|
# unconditionally this reaches page-authored text that merely LOOKS like a notice, in `observe`
|
|
# and `look` as well, and a page whose notice-shaped numbers change then digests identically on
|
|
# every read: the stall guard sees frozen and ends a run that was still moving.
|
|
# Both folds are applied at boundaries the TOOL reported, never by recognizing a shape. Every
|
|
# prefix of the marker attribute is legal page text, and both the page (a forged unterminated
|
|
# prefix) and the server (a download filename) can author text wearing the cut notice's shape —
|
|
# so no pattern and no match-selection rule can tell ours from theirs. Folding something that is
|
|
# not ours makes content that genuinely differs read as frozen, which the perception-stall guard
|
|
# TERMINATES on: it ends a run that was still making progress.
|
|
#
|
|
# The notice first, because its index is into the string as the tool returned it and the head
|
|
# fold would shift everything after itself.
|
|
noticed = addressed
|
|
if notice_at is not None and 0 <= notice_at < len(addressed):
|
|
closing = addressed.find("]", notice_at)
|
|
if closing != -1:
|
|
noticed = addressed[:notice_at] + "…[*]" + addressed[closing + 1 :]
|
|
head_folded = '*"' + noticed[head_fragment_len:] if head_fragment_len else noticed
|
|
closed = _TV3_MARKER_VALUE_RE.sub(lambda m: m.group(0).partition("=")[0] + '="*"', head_folded)
|
|
closed = _TV3_PICK_VALUE_RE.sub('data-tv3-pick="*"', closed)
|
|
return _TV3_MARKER_CUT_RE.sub(
|
|
lambda m: (
|
|
m.group(0)
|
|
if notice_at is None and m.group(0).startswith("data-tv3-pick=")
|
|
else m.group(0).partition("=")[0] + '="*'
|
|
),
|
|
closed,
|
|
)
|
|
|
|
|
|
def _content_only_perception(
|
|
content: str,
|
|
*,
|
|
is_observe: bool = False,
|
|
head_fragment_len: int = 0,
|
|
notice_at: int | None = None,
|
|
clip_spans: Sequence[Sequence[int]] | None = None,
|
|
delta_at: int | None = None,
|
|
delta_end: int | None = None,
|
|
) -> str:
|
|
# The URL is a hint, not content: history.pushState moves it without changing the document. The
|
|
# full canonicalization (URL included) keeps clearing the repeat guards — a wizard whose pages
|
|
# differ only by URL must survive — but budget-extension evidence hashes THIS, so a URL flip
|
|
# alone can never earn budget.
|
|
return _PERCEPTION_URL_LINE_RE.sub(
|
|
"url=*",
|
|
_canonical_perception_content(
|
|
content,
|
|
is_observe=is_observe,
|
|
head_fragment_len=head_fragment_len,
|
|
notice_at=notice_at,
|
|
clip_spans=clip_spans,
|
|
delta_at=delta_at,
|
|
delta_end=delta_end,
|
|
),
|
|
)
|
|
|
|
|
|
# How many recent states a probe remembers. This length IS the longest oscillation period that can
|
|
# be recognised, so it is a detection limit and not a memory tuning knob.
|
|
PERCEPTION_RING = 8
|
|
|
|
# Run-scoped revisit memory (SKY-14998). The ring above is a DETECTION LIMIT: a run that cycles with
|
|
# a period longer than it returns to states the ring has already evicted, so every ring-bounded guard
|
|
# is structurally blind to the loop rather than judging it wrongly. Measured production case: a
|
|
# macro-cycle of period ~44 billable touches, against a ring of 8 and a canonical window of 10.
|
|
# This remembers perception states for the WHOLE run instead, which is what makes a long cycle
|
|
# visible at all. It is a memory, not a verdict: LOG-ONLY, nothing reads it for control flow.
|
|
PERCEPTION_REVISIT_EVENT = "taskv3 perception state revisited"
|
|
# Two returns to a state is a panel toggling; the third is the earliest a cycle is worth reporting.
|
|
PERCEPTION_REVISIT_LOG_AFTER = 3
|
|
# Bounded storage: one 64-char digest per DISTINCT perception state, so the worst case is ~32KB of digests per
|
|
# run. At the cap the memory stops admitting NEW states and keeps counting the ones it already
|
|
# holds — it degrades to a smaller memory rather than growing without bound, and every emission says
|
|
# whether it was capped so a truncated run is never read as a complete one.
|
|
PERCEPTION_REVISIT_CAP = 512
|
|
|
|
|
|
@dataclass
|
|
class _RevisitMemory:
|
|
"""How many times each perception state has been seen this run, unbounded in time and bounded in
|
|
size. Distinct from _ProbeStreak's ring, which only remembers the last PERCEPTION_RING states and
|
|
therefore cannot see a cycle longer than that."""
|
|
|
|
cap: int = PERCEPTION_REVISIT_CAP
|
|
# digest -> (times seen, novel-state count at its last sighting)
|
|
_seen: dict[str, tuple[int, int]] = field(default_factory=dict)
|
|
capped: bool = False
|
|
distinct_states: int = 0
|
|
peak_revisits: int = 0
|
|
# Revisits that had fresh ground covered since the last sighting, so they were not reported.
|
|
# Kept as a count so the healthy population stays measurable in aggregate without a line each.
|
|
progressed_revisits: int = 0
|
|
_novel_states: int = 0
|
|
|
|
def record(self, digest: str) -> tuple[int, int]:
|
|
"""Returns (times this state has now been seen, distinct NEW states seen since it was last
|
|
seen). Zero new states between two sightings is a replay; a drill-down that opens a fresh
|
|
page between returns to its list is not, and that difference is what makes the signal
|
|
filterable rather than firing on healthy and stuck runs alike.
|
|
|
|
Novelty is measured against the WHOLE run, not the probe's ring: the ring is bounded at
|
|
PERCEPTION_RING, so a cycle longer than it reads as fresh content there — the exact
|
|
blindness this memory exists to lift, and reusing that derivation would reintroduce it.
|
|
"""
|
|
entry = self._seen.get(digest)
|
|
if entry is None:
|
|
if len(self._seen) >= self.cap:
|
|
# Refused, not evicted: dropping a known state to admit a new one would reset a
|
|
# streak that is the whole point of the memory.
|
|
self.capped = True
|
|
return 0, 0
|
|
self._novel_states += 1
|
|
self._seen[digest] = (1, self._novel_states)
|
|
self.distinct_states = len(self._seen)
|
|
return 1, 0
|
|
seen, novel_at_last_sighting = entry
|
|
seen += 1
|
|
new_states_since = self._novel_states - novel_at_last_sighting
|
|
self._seen[digest] = (seen, self._novel_states)
|
|
self.peak_revisits = max(self.peak_revisits, seen)
|
|
if new_states_since:
|
|
self.progressed_revisits += 1
|
|
return seen, new_states_since
|
|
|
|
|
|
@dataclass
|
|
class _ProbeStreak:
|
|
history: deque[str]
|
|
# Consecutive snapshots identical to the previous one; a match means two reads, so it opens at 2.
|
|
identical: int = 0
|
|
# Consecutive snapshots matching ANY state still in the ring (a superset of ``identical``).
|
|
revisits: int = 0
|
|
|
|
|
|
@dataclass
|
|
class _Snapshot:
|
|
"""Streak readings after one snapshot; the verdict reads ``live``, the warning ``tool_identical``."""
|
|
|
|
# Per-tool consecutive-identical count, argument-blind: the counter this guard shipped with.
|
|
tool_identical: int
|
|
probe_identical: int
|
|
probe_revisits: int
|
|
# A repeated probe returned different content: in-loop evidence that the page changed.
|
|
progressed: bool
|
|
|
|
@property
|
|
def live(self) -> int:
|
|
return min(self.tool_identical, self.probe_identical)
|
|
|
|
|
|
class _PerceptionLedger:
|
|
"""No-progress streaks of perception snapshots, kept per tool AND per probe (tool, canonical args).
|
|
|
|
The loop acts only where BOTH agree: the tool has returned the same string N times running and
|
|
the same probe has too. The per-probe term is what stops distinct probes that happen to return
|
|
the same string (a form full of not-yet-chosen dropdowns) from reading as one frozen page; the
|
|
per-tool term is what keeps firing a subset of what the argument-blind counter fired, so a
|
|
frozen page interleaved with a live sibling probe — which content alone cannot tell from a live
|
|
page whose static region is re-read — is never terminated where it was not before.
|
|
|
|
Streaks are scoped to the probe that produced them. Progress in one probe says nothing about
|
|
another: a sibling that ticks every read (a clock, a log tail) must not keep a frozen page
|
|
alive to the budget cap, which is exactly what clearing on any probe's progress would do.
|
|
|
|
Counts per RESULT, not per round: the thresholds are denominated in snapshots, and a turn that
|
|
batches five identical probes has taken five of them.
|
|
|
|
``revisits`` additionally treats a return to any state still in the ring as no progress, which
|
|
is what catches a control toggling open and shut. That is NEW firing, so it is only reported.
|
|
"""
|
|
|
|
def __init__(self) -> None:
|
|
self._probes: dict[tuple[str, str], _ProbeStreak] = {}
|
|
self._tools: dict[str, tuple[str, int]] = {}
|
|
self.content_only: dict[tuple[str, str], deque[str]] = {}
|
|
self.shadow_reported = False
|
|
self.suppressed_reported = False
|
|
# Run-level, deliberately NOT cleared by reset(): the worst revisit streak the run ever
|
|
# reached, and how many reads could have produced one. A first look at a probe key cannot
|
|
# revisit anything, so it is not a chance — counting it would make a run of one-shot probes
|
|
# read as "looked and found no looping".
|
|
self.peak_probe_revisits = 0
|
|
self.revisit_chances = 0
|
|
|
|
def reset(self) -> None:
|
|
"""Forget every streak: the document they described is gone (a reload)."""
|
|
self._probes.clear()
|
|
self._tools.clear()
|
|
self.content_only.clear()
|
|
|
|
def first_time(self, key: tuple[str, str]) -> bool:
|
|
return key not in self._probes
|
|
|
|
def record(self, key: tuple[str, str], content_digest: str) -> _Snapshot:
|
|
tool_name = key[0]
|
|
prev = self._tools.get(tool_name)
|
|
tool_identical = prev[1] + 1 if prev is not None and prev[0] == content_digest else 1
|
|
self._tools[tool_name] = (content_digest, tool_identical)
|
|
|
|
probe = self._probes.get(key)
|
|
if probe is None:
|
|
self._probes[key] = _ProbeStreak(deque([content_digest], maxlen=PERCEPTION_RING))
|
|
return _Snapshot(tool_identical, 0, 0, False)
|
|
self.revisit_chances += 1
|
|
progressed = content_digest != probe.history[-1]
|
|
if progressed:
|
|
probe.identical = 0
|
|
else:
|
|
probe.identical = probe.identical + 1 if probe.identical else 2
|
|
if content_digest in probe.history:
|
|
probe.revisits = probe.revisits + 1 if probe.revisits else 2
|
|
else:
|
|
probe.revisits = 0
|
|
probe.history.append(content_digest)
|
|
self.peak_probe_revisits = max(self.peak_probe_revisits, probe.revisits)
|
|
return _Snapshot(tool_identical, probe.identical, probe.revisits, progressed)
|
|
|
|
def next_snapshot_can_trip(self, threshold: int) -> bool:
|
|
"""Whether ONE more read of some probe, returning what it last returned, reaches ``threshold``
|
|
live. This is the trip's exact precondition, not an estimate of it: the next read continues
|
|
the tool counter only if that content is also the tool's last content, so a probe that went
|
|
dormant while its tool moved on to other content cannot be the one that trips."""
|
|
for (tool_name, _), probe in self._probes.items():
|
|
tool_last_content, tool_identical = self._tools[tool_name]
|
|
if probe.history[-1] != tool_last_content:
|
|
continue
|
|
next_probe_identical = probe.identical + 1 if probe.identical else 2
|
|
if min(tool_identical + 1, next_probe_identical) >= threshold:
|
|
return True
|
|
return False
|
|
|
|
|
|
# Net-progress ledger (SKY-15020 Lever C). The repetition guards above ask whether the page or
|
|
# action REPEATED; this asks whether the run made NET PROGRESS. Real stuck-ness is often VARIED
|
|
# actions with zero net progress (SKY-14998: 21 input-timeouts on different selectors — varied
|
|
# actions, varied perception — so no repetition guard trips and the run oscillates to the cap). The
|
|
# ledger is SHADOW-ONLY and ADDITIVE: it emits a "would fail-fast" event but terminates nothing and
|
|
# leaves the three guards untouched. Adding live terminations is a release-posture change gated on
|
|
# this event's own decision precision plus operator sign-off.
|
|
PROGRESS_LEDGER_WINDOW = 8
|
|
# Facetable event names; the offline precision/survival metrics key on these — change only with the
|
|
# dashboards that read them.
|
|
# Canonical progress signal, Phase A (SKY-15379): ONE target-churn + compound-progress tracker,
|
|
# LOG-ONLY. The loop answers "is this run progressing?" five separate ways today; this tracker is
|
|
# the candidate replacement signal, emitting the parity/precision data any consolidation needs
|
|
# while changing no behavior. Its target key is the canonicalized SELECTOR, not tool+args — the
|
|
# incumbent action-loop key provably missed prod repeat-loops whose args varied on a fixed target.
|
|
CANONICAL_LOOP_EVENT = "taskv3 canonical progress loop detected"
|
|
CANONICAL_EXTEND_DELTA_EVENT = "taskv3 canonical progress extension delta"
|
|
# Log-only thresholds; the Phase-B intervention thresholds are tuned FROM this data, so the fire
|
|
# points are logged at every rung of the distribution rather than one cliff.
|
|
CANONICAL_LOOP_WINDOW = 10
|
|
CANONICAL_LOOP_FIRE_COUNTS = (3, 4, 6, 8)
|
|
# The rung that counts as "looping" for the extension-delta read — named so reordering the logging
|
|
# rungs above can never silently move the threshold Phase B tunes against.
|
|
CANONICAL_LOOP_TOUCHES = 4
|
|
|
|
PROGRESS_LEDGER_SHADOW_EVENT = "taskv3 loop progress ledger would fail-fast"
|
|
|
|
|
|
@dataclass
|
|
class _ProgressLedger:
|
|
"""Billable actions since the last net-progress signal, plus its per-run peak.
|
|
|
|
The already-computed distance-to-done metric is the observe summary's ``invalid_fields`` count.
|
|
Net progress is any of: a navigation or download landing (hard progress), the count reaching a
|
|
NEW LOW (a real form advance), or the count RISING since the last look (a fresh page's required
|
|
fields, or a submit that surfaced new errors — either way the context changed, so the prior
|
|
no-progress streak is stale). Novelty is deliberately NOT progress: varied thrash produces novel
|
|
perception, which is why the streak is counted against the invalid-field trend, not the page
|
|
digest.
|
|
|
|
Two invariants keep the over-termination direction safe:
|
|
- The verdict is taken on an OBSERVE that CONFIRMS no progress, never on the raw action count: a
|
|
run that batches several fixes before re-observing (markers stay valid until the page
|
|
re-renders) has not yet shown a stalled look, so the streak withholds rather than fires.
|
|
- ``form_armed`` reflects the CURRENT look, not a sticky earlier one, so a form-less page (a
|
|
confirmation/extraction page, or a solved form) can never be judged stuck here.
|
|
|
|
The design is intentionally biased toward PRECISION over recall — a shadow verdict headed for a
|
|
future live terminator must not fire on a progressing run. A real page-transition signal (the click
|
|
tool's ``page_transitioned``, from its already-computed url_before/url_after) closes ONE precision
|
|
gap the rise heuristic left: a click that moves the URL re-baselines cleanly as hard progress, so a
|
|
fresh page whose ``invalid_fields`` count coincidentally equals the prior page's no longer reads as a
|
|
stalled look. Only the POSITIVE direction is used — a URL change proves a transition — because URL
|
|
equality does NOT prove same-page (a URL-stable SPA multi-step form advances without moving the URL),
|
|
so the ledger never suppresses a re-baseline on an unchanged URL. That leaves known false-NEGATIVES
|
|
(the SAFE direction) the shadow numbers under-count: a run whose ``invalid_fields`` OSCILLATES or
|
|
CREEPS still re-baselines on every up-swing (the rise heuristic cannot separate same-page thrash from
|
|
a real advance without a stronger same-page oracle — the Lever C recall follow-up), and a form
|
|
silently rejected while showing zero invalid fields never arms.
|
|
|
|
Reads only what the loop already computed — no extra LLM turn, no re-observe.
|
|
"""
|
|
|
|
window: int = PROGRESS_LEDGER_WINDOW
|
|
actions_since_progress: int = 0
|
|
peak_actions_since_progress: int = 0
|
|
invalid_baseline: int | None = None # floor for the current page/context; rebased on rise or new low
|
|
last_invalid: int | None = None
|
|
form_armed: bool = False
|
|
ever_armed: bool = False # a form was seen at least once — the runs the survival record is emitted for
|
|
shadow_reported: bool = False
|
|
|
|
def observe(self, invalid_fields: int) -> bool:
|
|
"""Record an observation; return True the first time one CONFIRMS window-length no-progress."""
|
|
prev = self.last_invalid
|
|
self.last_invalid = invalid_fields
|
|
# Reflects THIS look only: a page with no invalid fields must not be judged stuck here.
|
|
self.form_armed = invalid_fields > 0
|
|
self.ever_armed = self.ever_armed or self.form_armed
|
|
if self.invalid_baseline is None:
|
|
self.invalid_baseline = invalid_fields
|
|
return False
|
|
if invalid_fields < self.invalid_baseline:
|
|
# A new low: real net progress on the form. Reset the streak and re-baseline.
|
|
self.actions_since_progress = 0
|
|
self.invalid_baseline = invalid_fields
|
|
return False
|
|
if prev is not None and invalid_fields > prev:
|
|
# The count rose since the last look: a new page's fresh required fields, or a submit that
|
|
# surfaced new errors. Re-baseline to the new floor instead of measuring it against a
|
|
# stale, lower one. A real click-driven transition instead arrives as hard_progress()
|
|
# upstream (baseline cleared); this rise path is the fallback for same-page count changes
|
|
# and for transitions no tool witnessed.
|
|
self.actions_since_progress = 0
|
|
self.invalid_baseline = invalid_fields
|
|
return False
|
|
# Flat, with no new low and no rise: this look confirms no net progress since the last one.
|
|
if self.actions_since_progress >= self.window and self.form_armed and not self.shadow_reported:
|
|
self.shadow_reported = True
|
|
return True
|
|
return False
|
|
|
|
def hard_progress(self) -> None:
|
|
# Navigation or a download landing: unambiguous progress. The old page's distance metric is
|
|
# stale, so drop the baseline; the next observe re-arms from the new page.
|
|
self.actions_since_progress = 0
|
|
self.form_armed = False
|
|
self.invalid_baseline = None
|
|
|
|
def on_billable(self) -> None:
|
|
"""Count one billable action toward the no-progress streak (the verdict is taken on observe)."""
|
|
self.actions_since_progress += 1
|
|
self.peak_actions_since_progress = max(self.peak_actions_since_progress, self.actions_since_progress)
|
|
|
|
|
|
# Failure-evidence gate: a finish(failed) issued shortly after a submit-class action or a
|
|
# solve_captcha attempt is held for ONE evidence turn, because submissions and captcha protocols
|
|
# complete asynchronously — the sampled false-negative verdicts fired 2-7s after the model's last
|
|
# look while the page went on to show the submission confirmation. Trigger tools are the ones whose
|
|
# page effects can land after their tool result; the window is in loop turns so intervening
|
|
# perception does NOT disarm it (the state can flip after the last observe while a protocol is in
|
|
# flight). The true verdict-to-flip latency is unmeasured in the sampled replays: the quiescence
|
|
# wait exits on the first stable fingerprint pair (so honest gated failures pay ~one sample), the
|
|
# cap only bounds a still-mutating page, and the effective evidence window is dominated by the
|
|
# deferral round-trip itself (one LLM turn + the observe).
|
|
# Completed-side settle deferrals. 0 disables that gate while leaving the failure-evidence gate
|
|
# (which shares the fingerprint sampler) intact.
|
|
DEFAULT_MAX_SETTLE_DEFERRALS = 2
|
|
FAILURE_EVIDENCE_WINDOW_TURNS = 5
|
|
FAILURE_EVIDENCE_SETTLE_MAX_SECONDS = 8.0
|
|
# A deferral needs room for its corrected cycle; without it the gate would convert an honest
|
|
# failure verdict into budget_exhausted (a budget cap landing mid-deferral). The worst-case cycle
|
|
# is wait + observe + re-finish — the deferral message invites an optional brief wait — so both
|
|
# the turn and tool-call reservations are 3, the latter read from a per-call refreshed counter.
|
|
# The deadline headroom must additionally fund the settle cap plus the cycle's LLM round trips.
|
|
FAILURE_EVIDENCE_MIN_DEADLINE_HEADROOM_SECONDS = 60.0
|
|
FAILURE_EVIDENCE_MIN_TOOL_CALLS = 3
|
|
FAILURE_EVIDENCE_MIN_TURNS = 3
|
|
# A verification give-up deferral asks for a BLOCKING poll slice (auth_tools caps one at 120s) on top
|
|
# of that cycle, so the 60s above would let the gate convert an honest failure into the
|
|
# budget-exhausted end it exists to prevent. Not imported from auth_tools: that module imports this one.
|
|
VERIFICATION_GIVEUP_MIN_DEADLINE_HEADROOM_SECONDS = 180.0
|
|
|
|
|
|
def _is_enter_submit(tool_name: str, args: dict[str, Any]) -> bool:
|
|
"""An Enter keypress, or a type or select_combobox that pressed Enter: the submit shapes with no selector to key on."""
|
|
if tool_name == "press_key":
|
|
raw = str(args.get("key", ""))
|
|
# Space on a focused button activates it exactly like a click; a literal " " would strip to "".
|
|
if raw == " ":
|
|
return True
|
|
# Playwright accepts Modifier+Key chords, and Control/Meta+Enter is a real submit.
|
|
key = raw.strip().lower().rsplit("+", 1)[-1]
|
|
return key in ("enter", "return", "numpadenter", "space")
|
|
if tool_name in ("type", "select_combobox"):
|
|
return bool(args.get("press_enter"))
|
|
return False
|
|
|
|
|
|
def _call_selector(args: dict[str, Any]) -> str | None:
|
|
selector = args.get("selector")
|
|
if isinstance(selector, str) and selector:
|
|
return selector
|
|
mark = args.get("mark")
|
|
return f"mark={mark}" if isinstance(mark, (int, str)) else None
|
|
|
|
|
|
_PAGE_PROBE_TIMEOUT_SECONDS = 10.0
|
|
|
|
_T = TypeVar("_T")
|
|
|
|
|
|
async def _bounded_probe(probe: Awaitable[_T], timeout: float | None = None) -> _T:
|
|
"""Every awaited page.evaluate probe goes through here: the loop's deadline and cancellation checks
|
|
run BETWEEN awaits, so a hung renderer can only be interrupted by a bound on the await itself.
|
|
A timeout raises (asyncio.TimeoutError) and each call site's raise handling decides what a
|
|
missing reading means there. ``timeout`` defaults to the module-level cap read at CALL time (not
|
|
baked in as a parameter default) so tests can still monkeypatch _PAGE_PROBE_TIMEOUT_SECONDS."""
|
|
return await asyncio.wait_for(probe, timeout=_PAGE_PROBE_TIMEOUT_SECONDS if timeout is None else timeout)
|
|
|
|
|
|
async def _sample_probe(probe: Callable[[], Awaitable[str | None]], deadline_at: float | None = None) -> str | None:
|
|
"""A raising or hung probe is as uninformative as a None one: all mean "no reading," not "unchanged."
|
|
Generic over what it samples -- the page_probe (document identity) and page_fingerprint (rendered
|
|
content) callables share this exact shape. ``deadline_at`` caps the wait to whatever's left of the
|
|
loop's own deadline (never longer than the default), so a hung probe cannot outlive it; a deadline
|
|
already passed skips the probe call entirely and reads as a missing sample, same as a timeout."""
|
|
timeout = _PAGE_PROBE_TIMEOUT_SECONDS
|
|
if deadline_at is not None:
|
|
remaining = deadline_at - time.monotonic()
|
|
if remaining <= 0:
|
|
return None
|
|
timeout = min(timeout, remaining)
|
|
try:
|
|
return await _bounded_probe(probe(), timeout=timeout)
|
|
except Exception:
|
|
return None
|
|
|
|
|
|
# The code tool drives the page through Playwright directly, so the loop sees one tool call where an
|
|
# arbitrary number of clicks and submits may have happened. Named here, beside the detector, because
|
|
# the detector is what has to know: it is the one fact about that tool the submit guards need, and a
|
|
# guard that learns about a caller from a list somebody remembered to update is the shape of the bug
|
|
# this closes.
|
|
CODE_TOOL_NAME = "execute_python"
|
|
|
|
|
|
def _may_submit(tool_name: str, args: dict[str, Any]) -> bool:
|
|
"""A click, an Enter press, or a type or select_combobox that pressed Enter: the loop cannot tell a submit from any of them.
|
|
|
|
The code tool counts too: its body is opaque to the loop, so it is treated as possibly having
|
|
submitted rather than as certainly not having.
|
|
"""
|
|
return tool_name in ("click", CODE_TOOL_NAME) or _is_enter_submit(tool_name, args)
|
|
|
|
|
|
# Secret values reach the model only as `placeholder_...` tokens, so the guard below keys on the
|
|
# token the model passed back and never on what it resolves to — it cannot handle a plaintext secret.
|
|
# The prefix covers every registered secret, not only a login: AWS and Azure parameters and card
|
|
# fields are minted the same way, which is why the refusal names a rejected value rather than a
|
|
# failed sign-in. This matches greedily where the registry matches longest-known-key, so it shares
|
|
# find_embedded_placeholder_tokens' boundary limitation and adds one: two tokens concatenated with no
|
|
# separator merge into one, and a token the model reproduces with a different trailing fragment reads
|
|
# as a different key and goes uncounted. Both fail toward under-counting. It cannot call the registry,
|
|
# because loop.py must not import `app`; thread a token callable from agent.py if either is observed.
|
|
_CREDENTIAL_PLACEHOLDER_RE = re.compile(re.escape(RANDOM_SECRET_ID_PREFIX) + r"[A-Za-z0-9_]+")
|
|
# file_upload is excluded: its placeholder names a file, not a credential, and re-attaching a file
|
|
# after a submit is ordinary recovery rather than a second authentication attempt. The code tool is
|
|
# included for the reason _may_submit includes it — its body is opaque, so it is treated as possibly
|
|
# having entered the credential rather than as certainly not.
|
|
CREDENTIAL_ENTRY_TOOLS = (FILL_TOOLS | frozenset({CODE_TOOL_NAME})) - frozenset({"file_upload"})
|
|
# How many times one credential may be submitted within a task, and the whole knob. Replaying the
|
|
# rule over every v3 block in a 22h window: budget 1 refuses 38 blocks of which 23 COMPLETED, so it
|
|
# would break more runs than it protects; budget 3 reaches the bottom of the range the observed
|
|
# lockouts came from (3-5 submissions). 2 is the largest budget that stays under the observed harm
|
|
# threshold. It is chosen against today's submit precision — _may_submit counts any click, so a
|
|
# click that submits nothing still spends budget — and must be re-derived if that precision improves.
|
|
CREDENTIAL_SUBMIT_BUDGET = 2
|
|
# A login identifier spends no lockout allowance, but identifier-first flows can send a code or hit lookup
|
|
# throttles on every submit, so it gets a higher budget rather than none — 5 is an unmeasured judgment
|
|
# with no replay behind it, unlike CREDENTIAL_SUBMIT_BUDGET.
|
|
LOGIN_IDENTIFIER_SUBMIT_BUDGET = 5
|
|
|
|
|
|
def _credential_placeholders(args: dict[str, Any]) -> set[str]:
|
|
"""Every credential placeholder token this call's arguments carry, across all of them.
|
|
|
|
Scanning the values rather than naming `text`/`value`/`chosen` keeps the guard from being a
|
|
list of argument conventions that a new fill tool can be added without. Nested because the
|
|
arguments are the model's JSON: `select_option` carries `values`/`labels` as arrays, so a
|
|
top-level-strings-only scan would let a credential through in a list.
|
|
"""
|
|
|
|
def walk(value: Any) -> Iterator[str]:
|
|
if isinstance(value, str):
|
|
yield from _CREDENTIAL_PLACEHOLDER_RE.findall(value)
|
|
elif isinstance(value, dict):
|
|
for nested in value.values():
|
|
yield from walk(nested)
|
|
elif isinstance(value, (list, tuple)):
|
|
for nested in value:
|
|
yield from walk(nested)
|
|
|
|
return set(walk(args))
|
|
|
|
|
|
def _trail_url_before(result: ToolResult, ctx: SkyvernContext | None) -> str | None:
|
|
url_before = (result.data or {}).get("url_before")
|
|
if not isinstance(url_before, str) or not url_before:
|
|
return None
|
|
return ctx.hide_from_model(url_before) if ctx is not None else url_before
|
|
|
|
|
|
def entered_values(tool_name: str, args: dict[str, Any]) -> tuple[str, ...]:
|
|
values: list[str] = []
|
|
for key in ENTERED_VALUE_ARGS.get(tool_name, ()):
|
|
value = args.get(key)
|
|
if isinstance(value, str):
|
|
values.append(value)
|
|
elif isinstance(value, list):
|
|
values.extend(item for item in value if isinstance(item, str))
|
|
return tuple(values)
|
|
|
|
|
|
# Bounds the one probe a would-be-skipped click pays; a probe that overruns keeps the skip.
|
|
TOGGLE_PROBE_TIMEOUT_SECONDS = 2.0
|
|
|
|
|
|
async def _is_toggle_click(tool_name: str, spec: ToolSpec | None, args: dict[str, Any]) -> bool:
|
|
"""A click whose target exposes a toggle state chooses an answer rather than submitting. Fails closed."""
|
|
if tool_name != "click" or spec is None or spec.toggle_probe is None:
|
|
return False
|
|
try:
|
|
is_toggle = await asyncio.wait_for(spec.toggle_probe(args), timeout=TOGGLE_PROBE_TIMEOUT_SECONDS)
|
|
except Exception:
|
|
LOG.info("taskv3 toggle probe failed; keeping the batch skip", exc_info=True)
|
|
return False
|
|
if is_toggle is True:
|
|
LOG.info(BATCH_SKIP_TOGGLE_EXEMPT_EVENT, tool=tool_name)
|
|
return True
|
|
return False
|
|
|
|
|
|
def _is_finish(tool_name: str) -> bool:
|
|
return tool_name == "finish"
|
|
|
|
|
|
def _outcome_reports_failure(outcome: dict[str, Any] | None) -> bool:
|
|
"""Whether a tool that returned ok nonetheless reported reaching no usable page. Read off the
|
|
machine facts the tool exposed (an HTTP status), so the tool never has to adjudicate its own
|
|
success -- and only the persisted row moves: the model still reads the tool's own ok result."""
|
|
if not outcome:
|
|
return False
|
|
status = outcome.get("http_status")
|
|
return isinstance(status, int) and status >= ACTION_OUTCOME_FAILED_HTTP_STATUS
|
|
|
|
|
|
def _arms_failure_evidence(tool_name: str, args: dict[str, Any], ok: bool) -> bool:
|
|
"""solve_captcha arms on ANY dispatch — its "not solved" error is exactly the verdict the async
|
|
protocol can contradict. Other actions arm only when they reached the page AND in their
|
|
submit-shaped form."""
|
|
if tool_name == "solve_captcha":
|
|
return True
|
|
if not ok:
|
|
return False
|
|
return _may_submit(tool_name, args)
|
|
|
|
|
|
@dataclass
|
|
class ActivityRecency:
|
|
"""Written by the tool loop each turn/action, read by the finish tool's holds and the engine's summary line.
|
|
|
|
Most fields are recency/headroom signals sampled per turn. `action_attempts` and `perceptions`
|
|
are the exception: run-lifetime cumulative counters, never reset.
|
|
"""
|
|
|
|
turn: int = 0
|
|
turns_remaining: int | None = None
|
|
tool_calls_remaining: int | None = None
|
|
tokens_remaining: int | None = None
|
|
last_turn_tokens: int = 0
|
|
last_trigger_turn: int | None = None
|
|
# True while one more read of some probe, returning what it last returned, would trip the stall
|
|
# terminator: a deferral-forced observe must never be the snapshot that trips it. KNOWN LIMIT: a
|
|
# run that reaches the edge and then stops reading that tool altogether leaves this true for the
|
|
# rest of the run, disabling the deferral (same as the argument-blind counter it shipped with).
|
|
perception_stall_imminent: bool = False
|
|
# True from the moment the loop's granted final turn starts running. A hold asks for a retry
|
|
# turn that no longer exists — honoring it would silently convert the model's verdict into
|
|
# budget_exhausted, the exact conversion the hold headroom gates exist to prevent.
|
|
final_turn_active: bool = False
|
|
# Attempts, not successes: every `touches_page` call (billable, recordable or `engages_page`) whatever
|
|
# its result, plus each pre-dispatch refusal (mark-stale, extraction entry, credential resubmit).
|
|
action_attempts: int = 0
|
|
# Successful perceptions (the compactable tools: observe / get_html / look).
|
|
perceptions: int = 0
|
|
# State at the FIRST failed/terminated finish past the failure-evidence gate, emitted once per run
|
|
# by the engine's summary line. Not the run's first verdict: a zero-action `completed` refused by a
|
|
# gate above reaches this point later as failed/terminated.
|
|
attempts_at_hold_gate: int | None = None
|
|
perceptions_at_hold_gate: int | None = None
|
|
status_at_hold_gate: str | None = None
|
|
# Raised only by the goal-check enforce hold and consumed later in the SAME batch, whose queued calls
|
|
# predate the hold. Named `_batch_skip` because `test_canonical_ring_state_is_touched_only_through_the_tracker`
|
|
# greps this module for the canonical ring's field-name fragments.
|
|
held_verdict_batch_skip: bool = False
|
|
|
|
def armed(self, window: int = FAILURE_EVIDENCE_WINDOW_TURNS) -> bool:
|
|
return self.last_trigger_turn is not None and (self.turn - self.last_trigger_turn) <= window
|
|
|
|
|
|
def _names_submit_control(tool_name: str, args: dict[str, Any], ok: bool) -> str | None:
|
|
"""The selector of a control a successful action acted on directly, or None.
|
|
|
|
Only `click`. An Enter press and a type-that-pressed-Enter submit through a control they do not
|
|
name, and a captcha dispatch names none at all, so for all three "is that control still in
|
|
flight" has no subject — and the selector they do carry is a text field, whose value is the
|
|
model's own typed text."""
|
|
if not ok or tool_name != "click":
|
|
return None
|
|
selector = args.get("selector")
|
|
return selector if isinstance(selector, str) and selector else None
|
|
|
|
|
|
def _has_hold_headroom(
|
|
activity: ActivityRecency | None,
|
|
deadline_at: float | None,
|
|
min_deadline_headroom_seconds: float = FAILURE_EVIDENCE_MIN_DEADLINE_HEADROOM_SECONDS,
|
|
) -> bool:
|
|
"""Whether a deferral has the budget to buy the re-verification turn it asks for.
|
|
|
|
Without it the run ends budget_exhausted, which is unmapped and lands on failed -- turning an
|
|
honest hold into the false failure it exists to avoid. Every axis the failure-evidence gate
|
|
reserves, for the same reason: a token exhaustion and a stall streak one short of its terminator
|
|
each convert the deferral into a verdict the gate did not choose."""
|
|
if activity is not None:
|
|
if activity.final_turn_active:
|
|
return False
|
|
if activity.turns_remaining is not None and activity.turns_remaining < FAILURE_EVIDENCE_MIN_TURNS:
|
|
return False
|
|
if (
|
|
activity.tool_calls_remaining is not None
|
|
and activity.tool_calls_remaining < FAILURE_EVIDENCE_MIN_TOOL_CALLS
|
|
):
|
|
return False
|
|
if activity.tokens_remaining is not None and activity.tokens_remaining < FAILURE_EVIDENCE_MIN_TURNS * max(
|
|
activity.last_turn_tokens, 1
|
|
):
|
|
return False
|
|
if activity.perception_stall_imminent:
|
|
return False
|
|
if deadline_at is not None and deadline_at - time.monotonic() < min_deadline_headroom_seconds:
|
|
return False
|
|
return True
|
|
|
|
|
|
def _cap_trip_relieved(
|
|
cap_trip: str | None,
|
|
max_tokens: int | None,
|
|
total_tokens: int,
|
|
final_turn_token_reserve: int,
|
|
max_turns: int,
|
|
turns: int,
|
|
max_tool_calls: int,
|
|
total_tool_calls: int,
|
|
) -> bool:
|
|
"""Whether the guard that produced `cap_trip` has since been raised past its trip.
|
|
|
|
Only the three guards a budget extension re-derives can be relieved. Mirrors the top-of-turn
|
|
checks exactly, reserve included, so a released latch cannot re-trip on the very next turn.
|
|
|
|
The two exclusions differ in kind. The wall clock is never re-derived, and its branch is
|
|
unreachable in practice besides: the gate refuses every grant once the clock is blown
|
|
("insufficient_deadline_headroom"), so a deadline latch and a grant cannot coexist in one run.
|
|
The action-step trip is excluded on purpose and IS reachable — a refusal can latch the wrap-up
|
|
turn and the next turn's refreshed evidence can then earn a grant, which the spent-grant exit
|
|
discards. That wastes an earned extension and is worth revisiting, but releasing on it is a
|
|
behaviour change beyond raising the guards and is deliberately not made here."""
|
|
if cap_trip is None:
|
|
return False
|
|
if cap_trip.startswith("max_tokens"):
|
|
return max_tokens is not None and total_tokens < max(0, max_tokens - final_turn_token_reserve)
|
|
if cap_trip.startswith("max_turns"):
|
|
return turns < max_turns
|
|
if cap_trip.startswith("max_tool_calls"):
|
|
return total_tool_calls < max_tool_calls
|
|
return False
|
|
|
|
|
|
def _budget_extension_gate(
|
|
action_steps: int,
|
|
last_change_evidence_step: int | None,
|
|
action_warned: set[tuple[str, str]],
|
|
progress_stalled: bool,
|
|
activity: ActivityRecency | None,
|
|
deadline_at: float | None,
|
|
extension: int,
|
|
seconds_per_step: float | None = None,
|
|
headroom_gain: tuple[int, int, int] = (0, 0, 0),
|
|
) -> tuple[bool, str]:
|
|
"""Whether an exhausted action-step budget may be extended, and the deciding reason.
|
|
|
|
Progress is the property, not its absence-of-stall proxy: the gate requires POSITIVE recent
|
|
evidence the page changed (a repeated probe returning fresh content, a navigation or download —
|
|
the same events that clear the action-retry ledger), then vetoes on any live stall signal, and
|
|
finally requires the turn/token/deadline headroom to actually fund the extension — granting
|
|
budget the runaway guards would immediately revoke converts an honest exhaustion into a worse
|
|
one. `headroom_gain` is the extra (turns, tool calls, tokens) that granting would itself create
|
|
by re-deriving the runaway backstops from the extended cap, and is added to what the run has
|
|
left: the question is whether the extension is fundable AFTER the grant, so judging it on
|
|
pre-grant headroom would refuse extensions the grant pays for.
|
|
|
|
Which of the three checks stay live depends on that gain. Under the standard sizing policy the
|
|
turn and tool-call gains exceed anything one extension can consume, so those two become
|
|
satisfied by construction — correctly, since a guard re-derived from the new cap provably
|
|
cannot revoke it — and they still bind for a caller passing no `backstops_for_cap` (the
|
|
default) or its own guards. TOKENS are the live discriminator: the gain per step is that
|
|
guard's own per-step allowance, so a run burning more than that per turn — the re-read-the-page
|
|
spiral the backstop exists for — still fails, and so does one whose gain is zero because the
|
|
guard has clamped at its ceiling.
|
|
|
|
The checks are a funding FLOOR sized off the run's own observed burn (turns per step so far,
|
|
last turn's tokens), not a guarantee the extension completes — the facetable grant/refusal
|
|
events measure that on the canary."""
|
|
extra_turns, extra_tool_calls, extra_tokens = headroom_gain
|
|
if (
|
|
last_change_evidence_step is None
|
|
or action_steps - last_change_evidence_step > ACTION_BUDGET_EXTENSION_EVIDENCE_WINDOW
|
|
):
|
|
return False, "no_recent_page_change_evidence"
|
|
if action_warned:
|
|
return False, "warned_action_retry_streak"
|
|
if progress_stalled:
|
|
return False, "no_net_progress_window"
|
|
if activity is not None and activity.turns_remaining is not None:
|
|
turns_remaining = activity.turns_remaining + extra_turns
|
|
# Fractional comparison (cross-multiplied): floor division would read 1.9 observed
|
|
# turns-per-step as 1 and fund an extension the remaining turns cannot run.
|
|
if turns_remaining < extension or turns_remaining * max(action_steps, 1) < extension * activity.turn:
|
|
return False, "insufficient_turn_headroom"
|
|
if activity is not None and activity.tool_calls_remaining is not None:
|
|
# +1 reserves the terminal finish call: funding only the actions trades one budget
|
|
# exhaustion for another at the very last call.
|
|
if activity.tool_calls_remaining + extra_tool_calls < extension + 1:
|
|
return False, "insufficient_tool_call_headroom"
|
|
if (
|
|
activity is not None
|
|
and activity.tokens_remaining is not None
|
|
and activity.tokens_remaining + extra_tokens < extension * max(activity.last_turn_tokens, 1)
|
|
):
|
|
return False, "insufficient_token_headroom"
|
|
if activity is not None and activity.perception_stall_imminent:
|
|
return False, "perception_stall_imminent"
|
|
if deadline_at is not None:
|
|
# Fund the extension in wall-clock at the run's own observed pace, never below the flat
|
|
# minimum the deferral gates use.
|
|
required = FAILURE_EVIDENCE_MIN_DEADLINE_HEADROOM_SECONDS
|
|
if seconds_per_step is not None:
|
|
required = max(required, extension * seconds_per_step)
|
|
if deadline_at - time.monotonic() < required:
|
|
return False, "insufficient_deadline_headroom"
|
|
return True, "recent_page_change_evidence"
|
|
|
|
|
|
@dataclass
|
|
class SubmitWatch:
|
|
"""The control a click last acted on, and whether a completed verdict has already been held once
|
|
against it.
|
|
|
|
Deliberately NOT part of `ActivityRecency`. That record arms failure evidence, where a broad
|
|
trigger (any click, any captcha dispatch) and a decaying turn window are both correct. Neither is
|
|
correct here: a captcha dispatch carries no selector and would erase the control, and a turn
|
|
window expires while the run is doing the waiting this gate asked for. There is no window — the
|
|
probe is the arbiter, because a control that resolves and still reads as in flight IS the
|
|
question, where elapsed turns are only a proxy for it."""
|
|
|
|
selector: str | None = None
|
|
deferred: bool = False
|
|
|
|
def record(self, selector: str) -> None:
|
|
self.selector = selector
|
|
self.deferred = False
|
|
|
|
def clear(self) -> None:
|
|
self.selector = None
|
|
self.deferred = False
|
|
|
|
|
|
@dataclass
|
|
class SemanticCommitStats:
|
|
"""Tier-1 semantic commit reads and how many accepted. One instance per run, shared by every
|
|
tool call, so a per-run accept rate is a division and not a join."""
|
|
|
|
opportunities: int = 0
|
|
accepts: int = 0
|
|
|
|
def log_fields(self) -> dict[str, int]:
|
|
return {"semantic_commit_opportunities": self.opportunities, "semantic_commit_accepts": self.accepts}
|
|
|
|
|
|
@dataclass
|
|
class LedgerTerminalFields:
|
|
"""The ledger's end-of-run numbers, present only for a run that ever armed a form."""
|
|
|
|
peak_actions_since_progress: int
|
|
actions_since_progress: int
|
|
# The CURRENT look, cleared by progress — not the partition. See `TerminalTelemetry.form_ever_armed`.
|
|
form_armed: bool
|
|
would_fire: bool
|
|
|
|
def log_fields(self) -> dict[str, int | bool]:
|
|
return {
|
|
"peak_actions_since_progress": self.peak_actions_since_progress,
|
|
"actions_since_progress": self.actions_since_progress,
|
|
"form_armed": self.form_armed,
|
|
"would_fire": self.would_fire,
|
|
}
|
|
|
|
|
|
@dataclass
|
|
class TerminalTelemetry:
|
|
"""The run's progress signals, assembled where their state lives and emitted by the caller on the
|
|
one terminal record it already writes per run. Every optional member is OMITTED rather than zeroed
|
|
when its tier could not have fired, because a 0 would read as "fired nothing" instead of "never
|
|
ran" and would drag an offline rate's denominator."""
|
|
|
|
# Sticky, and the partition. `form_armed` on the same record is the current look and answers a
|
|
# different question, so filter populations on this one.
|
|
form_ever_armed: bool
|
|
survival: dict[str, int]
|
|
ledger: LedgerTerminalFields | None = None
|
|
peak_page_state_stall_rounds: int | None = None
|
|
peak_probe_revisits: int | None = None
|
|
semantic_commit: SemanticCommitStats | None = None
|
|
perception_reads: dict[str, int] | None = None
|
|
|
|
def log_fields(self) -> dict[str, Any]:
|
|
fields: dict[str, Any] = {"form_ever_armed": self.form_ever_armed, **self.survival}
|
|
if self.ledger is not None:
|
|
fields.update(self.ledger.log_fields())
|
|
if self.peak_page_state_stall_rounds is not None:
|
|
fields["peak_page_state_stall_rounds"] = self.peak_page_state_stall_rounds
|
|
if self.peak_probe_revisits is not None:
|
|
fields["peak_probe_revisits"] = self.peak_probe_revisits
|
|
if self.semantic_commit is not None:
|
|
fields.update(self.semantic_commit.log_fields())
|
|
if self.perception_reads is not None:
|
|
fields.update(self.perception_reads)
|
|
return fields
|
|
|
|
|
|
def _unblocker_options(available_tools: set[str]) -> list[str]:
|
|
options = []
|
|
if "solve_captcha" in available_tools:
|
|
options.append("if the page may be waiting on a verification widget, call solve_captcha")
|
|
if "get_html" in available_tools:
|
|
options.append("take ONE targeted get_html look at the region that should be changing")
|
|
if "look" in available_tools:
|
|
options.append("if you can't tell what's on the page or why an action isn't taking, call look to see it")
|
|
options.append("if the goal is already met, call finish(status=completed)")
|
|
options.append("if genuinely blocked, call finish(status=terminated) naming the blocker as the reason")
|
|
return options
|
|
|
|
|
|
def _page_state_nudge_text(rounds: int) -> str:
|
|
return (
|
|
f"Your last {rounds} action rounds left the page's rendered content completely unchanged — "
|
|
"whatever you are trying is not affecting this page. Stop repeating the current approach: "
|
|
"re-plan from a fresh observe, try a genuinely different control or path, or finish honestly "
|
|
"with a non-completed status, naming what is blocking you."
|
|
)
|
|
|
|
|
|
def _stall_nudge_text(stalled: list[tuple[str, int]], available_tools: set[str]) -> str:
|
|
"""One warning naming every stalled perception tool and the unblockers this run actually has —
|
|
a model that cannot see the gate won't reach for solve_captcha unless the symptom names it."""
|
|
symptoms = "; ".join(f"{name} has returned identical output {count} times in a row" for name, count in stalled)
|
|
return (
|
|
f"The page is not changing: {symptoms}, despite your actions (transient element-marker ids are "
|
|
"ignored when comparing). Do not keep re-observing, "
|
|
"waiting, or repeating the same action. Your options: " + "; ".join(_unblocker_options(available_tools)) + "."
|
|
)
|
|
|
|
|
|
def _action_target(args: dict[str, Any]) -> str:
|
|
return str(args.get("selector") or args.get("url") or args.get("key") or "the same target")
|
|
|
|
|
|
def _action_nudge_text(
|
|
repeats: list[tuple[str, dict[str, Any], int]], available_tools: set[str], *, page_moved: bool | None
|
|
) -> str:
|
|
"""The transcript cannot show the model its own repetition (superseded snapshots are compacted
|
|
away), so the warning carries that memory: which action, how many times, and whether the page
|
|
fingerprint moved since the streak began. page_moved is None when no pair of samples can answer
|
|
that — a missing reading is not evidence of a still page, so neither claim is made."""
|
|
symptoms = "; ".join(
|
|
f"you have called {name} on {_action_target(args)} {count} times" for name, args, count in repeats
|
|
)
|
|
read_the_page = "Read the current page before deciding: what you need may already be shown there."
|
|
if page_moved:
|
|
state = (
|
|
f"You are repeating the same action: {symptoms}. The page has changed since before the streak "
|
|
f"began, and the repeat attempts land on that same state, so repeating it will not change it "
|
|
f"further. {read_the_page}"
|
|
)
|
|
elif page_moved is None:
|
|
state = f"You are repeating the same action: {symptoms}. {read_the_page}"
|
|
else:
|
|
state = (
|
|
f"You are repeating the same action without effect: {symptoms}, and the page state you "
|
|
"last observed is unchanged since before the first attempt."
|
|
)
|
|
return (
|
|
f"{state} A message inviting you to retry (e.g. 'please submit again') is not an instruction to "
|
|
"loop — at most one retry, then report the outcome honestly. Your options: "
|
|
+ "; ".join(_unblocker_options(available_tools))
|
|
+ "."
|
|
)
|
|
|
|
|
|
def _reload_failed_nudge_text() -> str:
|
|
return (
|
|
"A page-level handler asked for the page to be reloaded, but the reload failed, so the page may be "
|
|
"stale or unresponsive. Re-observe before acting, and do not re-submit a form unless the fresh page "
|
|
"shows it was not already submitted."
|
|
)
|
|
|
|
|
|
def _refresh_nudge_text() -> str:
|
|
"""A page-level handler reloaded the page outside the model's own tool calls, so nothing else
|
|
in the transcript tells it the last observation is now stale."""
|
|
return (
|
|
"The page was refreshed by a page-level handler after the last tool call. Its state may have "
|
|
"changed: re-observe before acting, and do not re-submit a form until the fresh page shows it "
|
|
"was not already submitted."
|
|
)
|
|
|
|
|
|
def _budget_extended_observation(cap: str, recency: ActivityRecency | None) -> str:
|
|
"""The retraction of a `_budget_exhausted_observation` whose cap has since been raised.
|
|
|
|
APPENDED, never popped: `LoopState.reads` is keyed by absolute message index, so deleting the
|
|
stale message would silently re-anchor compaction onto the wrong ones. Without this the model
|
|
keeps reading "this is the final turn" for the rest of the run and wraps up early — which spends
|
|
the extension the release exists to preserve, through the prompt instead of through a counter."""
|
|
payload = {
|
|
"budget_exhausted": False,
|
|
"cap": cap,
|
|
"headroom": {
|
|
"tool_calls": recency.tool_calls_remaining if recency is not None else None,
|
|
"tokens": recency.tokens_remaining if recency is not None else None,
|
|
},
|
|
}
|
|
json_line = json.dumps(payload, separators=(",", ":"))
|
|
return (
|
|
f"{json_line}\nThat budget was extended: the earlier final-turn notice no longer applies and "
|
|
"the run continues. Keep working toward the goal."
|
|
)
|
|
|
|
|
|
def _budget_exhausted_observation(cap: str, recency: ActivityRecency | None) -> str:
|
|
"""A typed, model-facing fact (not an instruction) that this is the run's last granted turn.
|
|
Deliberately silent on what to do next -- finishing with a partial result, retrying, or reporting
|
|
failure are all legitimate depending on what the model has."""
|
|
payload = {
|
|
"budget_exhausted": True,
|
|
"cap": cap,
|
|
"headroom": {
|
|
"tool_calls": recency.tool_calls_remaining if recency is not None else None,
|
|
"tokens": recency.tokens_remaining if recency is not None else None,
|
|
},
|
|
}
|
|
json_line = json.dumps(payload, separators=(",", ":"))
|
|
return f"{json_line}\nThe run's budget cap has tripped; this is the final turn before the run ends."
|
|
|
|
|
|
def _guard_verdict(guard: str, reason: str) -> LoopOutcome:
|
|
"""The one way a loop guard ends a run: the customer-facing sentence in `reason`, the machine class
|
|
in `guard`.
|
|
|
|
They are separate because `reason` leaves the system verbatim as the task's `failure_reason` — a
|
|
field on the customer webhook and the run view — so the facet a dashboard groups on, the counter
|
|
that tripped, the tool that reported it and the selector it compared all belong on the guard's own
|
|
log line instead (SKY-16271). `_budget_exhausted_reason` below holds the same standard for the
|
|
budget exits.
|
|
"""
|
|
return LoopOutcome("terminated", reason, guard=guard)
|
|
|
|
|
|
def _verdict_url(url: Any, caller_known_urls: Collection[str]) -> str | None:
|
|
"""A URL fit to name in a customer-facing verdict, or None when nothing can be named.
|
|
|
|
`sanitize_published_url` owns the invariant this surface rests on -- a verdict publishes only what
|
|
the caller already gave us, plus the host -- so the scheme and host are always named and the path
|
|
only when the landed URL is one of `caller_known_urls`, elided to `/…` otherwise.
|
|
"""
|
|
if not isinstance(url, str) or not url.startswith(("http://", "https://")) or any(ch.isspace() for ch in url):
|
|
return None
|
|
published = sanitize_published_url(url, caller_known_urls)
|
|
# Only a caller URL can be this long once the path is elided, and the caller can read it in their
|
|
# own task config; cutting it would name a page that does not exist, so it is dropped whole.
|
|
if published is not None and len(published) > VERDICT_URL_MAX_CHARS:
|
|
return None
|
|
return published
|
|
|
|
|
|
def _dead_end_reason(status: int, url: Any, *, page_noun: str, caller_known_urls: Collection[str]) -> str:
|
|
"""The HTTP status stays in the sentence: `classify_from_failure_reason` derives NAVIGATION_FAILURE
|
|
from it, and it is the one fact separating a page that is gone from a page that is merely wrong."""
|
|
where = _verdict_url(url, caller_known_urls)
|
|
named = f"{page_noun} ({where})" if where else page_noun
|
|
return (
|
|
f"{named} returned HTTP {status}: it no longer exists or has been removed, so the task could "
|
|
"not be completed there."
|
|
)
|
|
|
|
|
|
def _verdict_target(
|
|
tool: str,
|
|
name: Any,
|
|
kind: Any,
|
|
label_secret_values: Callable[[], Collection[str]] | None,
|
|
ctx: SkyvernContext | None,
|
|
) -> str | None:
|
|
"""The control a verdict names, or None when the acting call reported neither a name nor a kind.
|
|
|
|
A page-supplied name is only printable against the run's full drop-check secret set -- its
|
|
registered parameters included, unfloored, since a name is dropped whole rather than scrubbed --
|
|
and only the caller can see that registry, so it arrives as a resolver read at verdict time. Same
|
|
set the persisted timeline label is checked against: a credential-shaped label must not be dropped
|
|
from the row and printed here. With no resolver, no context, or a resolver that fails, the verdict
|
|
falls back to the kind floor rather than publishing page text it could not check -- a missing
|
|
context is a missing HALF of that set (the codes minted this turn), not a reason to skip it.
|
|
"""
|
|
if not name and not kind:
|
|
return None
|
|
if label_secret_values is None or ctx is None:
|
|
return describe_target(tool, None, kind, ())
|
|
try:
|
|
secret_values = set(label_secret_values())
|
|
except Exception:
|
|
LOG.warning("taskv3 loop could not resolve a verdict's drop-check secrets", tool=tool)
|
|
return describe_target(tool, None, kind, ())
|
|
# Unfloored, and read live: a code minted this turn is not in the resolver's floored view.
|
|
secret_values |= set(ctx.runtime_secret_values)
|
|
return describe_target(tool, name, kind, secret_values)
|
|
|
|
|
|
def _action_loop_reason(target: str | None) -> str:
|
|
"""`target` is the control the run kept acting on, or None when the repeated call reported no name
|
|
(a failed call reports none) — then the verdict names no place rather than guessing one."""
|
|
where = f" on {target}" if target else ""
|
|
return (
|
|
f"The run repeated the same action{where} and the page did not change in response, so the task "
|
|
"could not make progress — commonly the site rejecting the action and showing the same page again."
|
|
)
|
|
|
|
|
|
def _budget_exhausted_reason(cap_trip: str) -> str:
|
|
"""Human sentence for a no-finish budget exit. Never includes the raw cap literal (that lives
|
|
only in `cap_trip` and the logs) so a status/reason readout doesn't leak an internal counter."""
|
|
if "deadline" in cap_trip:
|
|
axis = "time"
|
|
elif "max_tokens" in cap_trip:
|
|
axis = "token"
|
|
elif "max_turns" in cap_trip:
|
|
axis = "turn"
|
|
elif "max_tool_calls" in cap_trip:
|
|
axis = "tool-call"
|
|
elif "maximum steps" in cap_trip:
|
|
axis = "step"
|
|
else:
|
|
axis = "run"
|
|
return f"The run reached its {axis} budget before the model finished; the recorded output may be partial."
|
|
|
|
|
|
_TOOL_CALL_RECORD_FIELDS = frozenset(
|
|
{
|
|
"tool",
|
|
"tool_status",
|
|
"duration_seconds",
|
|
"result_chars",
|
|
"selector_present",
|
|
"selector_kind",
|
|
"tool_error_class",
|
|
"tool_ok_class",
|
|
"resolve_seconds",
|
|
"billable",
|
|
"turn",
|
|
"batch_size",
|
|
"batch_index",
|
|
"action_key_hash",
|
|
"snapshot_digest",
|
|
"probe_first_time",
|
|
"hit_class",
|
|
"hit_needed",
|
|
"hit_probe_seconds",
|
|
"hit_probe_isolated",
|
|
"hit_probe_raised",
|
|
"covered_branch",
|
|
"covered_controls",
|
|
"covered_layer_kind",
|
|
"requested_url",
|
|
"landed_url",
|
|
"nav_error_code",
|
|
"same_page",
|
|
"readiness_read_failed",
|
|
"menu_note",
|
|
"menu_rows",
|
|
"withhold_reason",
|
|
"text_delta_seconds",
|
|
"text_delta_lines",
|
|
"text_delta_chars",
|
|
"text_delta_over_bound",
|
|
"text_delta_skipped",
|
|
"text_delta_pending_lines",
|
|
}
|
|
)
|
|
|
|
# The counts observe's `summary` carries onto its record (see `_observe_summary_fields`).
|
|
OBSERVE_SUMMARY_FIELDS = frozenset(
|
|
{
|
|
"text_dropped",
|
|
"hidden_listed",
|
|
"hidden_dropped",
|
|
"hidden_dropped_off_canvas",
|
|
"hidden_dropped_visibility",
|
|
"hidden_dropped_zero_rect",
|
|
"hidden_dropped_off_viewport",
|
|
"off_viewport_unreachable_unnamed",
|
|
"off_viewport_unnamed_host_exempt",
|
|
"phantom_dropped",
|
|
"iframes_in_component_roots",
|
|
"undiscovered_roots",
|
|
"omitted_unnameable",
|
|
"invalid_fields",
|
|
"markers_minted",
|
|
"markers_reused",
|
|
"group_texts_found",
|
|
"a11y_removed_listed",
|
|
"duplicate_digest_lines",
|
|
"frames_same_origin",
|
|
"frames_cross_origin",
|
|
"frames_same_origin_interactive",
|
|
"frames_peeked",
|
|
"frames_peek_failed",
|
|
"frame_scan_failed",
|
|
"frame_unreadable_regions",
|
|
"elements_listed",
|
|
"pointer_roots_listed",
|
|
"pointer_capped",
|
|
"pointer_truncated",
|
|
"pointer_dropped",
|
|
"pointer_scan_stopped",
|
|
"pointer_scan_failed",
|
|
"elements_truncated",
|
|
"elements_truncated_in_components",
|
|
"elements_dropped",
|
|
"elements_truncated_by_page_cap",
|
|
}
|
|
)
|
|
|
|
# Every field a "taskv3 tool call finished" record can carry. A grader filtering these records checks its keys
|
|
# against this set, so a misspelled or renamed field fails loudly instead of matching nothing.
|
|
TOOL_CALL_RECORD_FIELD_NAMES = _TOOL_CALL_RECORD_FIELDS | OBSERVE_SUMMARY_FIELDS | {"charged"}
|
|
|
|
# A host is bounded in the DNS but not in a string `urlsplit` was handed, and a record field is
|
|
# indexed: cap the scrubbed value rather than trust what it was reduced from.
|
|
LOGGED_URL_MAX_CHARS = 500
|
|
|
|
# A hostname is letters (any script, so an IDN still logs its host), digits, dots and hyphens, or an
|
|
# IPv6 literal, which `hostname` hands back with the brackets stripped. Anything else in what
|
|
# `urlsplit` called the host means it found no host at all, whatever it returned: `urlsplit` does not
|
|
# treat a backslash as a path separator, so `https://host\signin\TOKEN` parses the whole run as the
|
|
# authority while the browser normalizes it to `/` and navigates to a path — the token would survive
|
|
# the scrub as part of the "host" and land in an indexed field.
|
|
_LOGGABLE_HOST = re.compile(r"[\w.\-]+|[0-9A-Fa-f.:]+")
|
|
|
|
|
|
class _ProgressEvidence(str, Enum):
|
|
"""Why the canonical ring was cleared. Every clear names the positive evidence class that
|
|
justified it — an "assume changed" fallback (a missing sample read charitably) may never clear."""
|
|
|
|
REFRESH_RELOAD = "refresh_reload"
|
|
FRESH_DOWNLOAD_OR_NAVIGATION = "fresh_download_or_navigation"
|
|
PAGE_TRANSITIONED = "page_transitioned"
|
|
INVALID_FIELDS_BASELINE_MOVE = "invalid_fields_baseline_move"
|
|
PERCEPTION_DIGEST = "perception_digest"
|
|
CROSS_BATCH_MOVEMENT = "cross_batch_movement"
|
|
PROBE_MISMATCH = "probe_mismatch"
|
|
PAGE_STATE_VERDICT = "page_state_verdict"
|
|
TERMINAL_BATCH_FINGERPRINT = "terminal_batch_fingerprint"
|
|
|
|
|
|
@dataclass
|
|
class _CanonicalProgressTracker:
|
|
"""Touches on canonicalized targets since the last compound-progress event (SKY-15379, Phase A).
|
|
|
|
A touch is one billable tool dispatch, keyed by its selector (tool name when selector-less).
|
|
Progress — any of the ledger's hard-progress events, an invalid-fields new-low, or a confirmed
|
|
page change — clears the ring, so a repeat streak can only accumulate across a genuinely
|
|
unchanged situation; a progressing run is untrippable by construction. Log-only: the counts
|
|
feed telemetry, nothing reads them for control flow.
|
|
"""
|
|
|
|
window: int = CANONICAL_LOOP_WINDOW
|
|
_touches: list[tuple[str, bool]] = field(default_factory=list)
|
|
# Bumped on every clear: a pending loop event minted under an older generation was completed
|
|
# by (or followed by) absorbed progress and must not be emitted.
|
|
gen: int = 0
|
|
# Rungs already fired this generation. Once the window saturates, same-target counts can PARK
|
|
# on a fire rung while churn evicts other keys — a rung is one threshold crossing, not one
|
|
# event per touch spent at it.
|
|
_fired: set[tuple[str, int]] = field(default_factory=set)
|
|
# Run-level high-water marks and per-evidence clear tallies, deliberately NOT cleared by
|
|
# progress(): the survival record needs the worst churn the run ever reached and every clear it
|
|
# ever made, the same way the ledger keeps peak_actions_since_progress across its own clears.
|
|
# Clearing these with the ring would report the last streak, not the peak.
|
|
peak_same_touches: int = 0
|
|
peak_same_errors: int = 0
|
|
_evidence_counts: dict[str, int] = field(default_factory=dict)
|
|
|
|
def record_touch(self, target_key: str, is_error: bool) -> tuple[int, int]:
|
|
"""Record one billable touch; returns (same-target touches, same-target errors) in-window."""
|
|
self._touches.append((target_key, is_error))
|
|
if len(self._touches) > self.window:
|
|
self._touches.pop(0)
|
|
same = [err for key, err in self._touches if key == target_key]
|
|
# A rung re-arms once the in-window count dips below it: a fresh streak after full eviction
|
|
# is a genuine re-crossing, distinct from a saturated window PARKED on a fire count.
|
|
self._fired = {entry for entry in self._fired if entry[0] != target_key or entry[1] <= len(same)}
|
|
same_errors = sum(1 for err in same if err)
|
|
self.peak_same_touches = max(self.peak_same_touches, len(same))
|
|
self.peak_same_errors = max(self.peak_same_errors, same_errors)
|
|
return len(same), same_errors
|
|
|
|
def progress(self, evidence: _ProgressEvidence) -> None:
|
|
"""The one clear choke point: a caller must name its evidence class, so no site can wipe
|
|
the ring on a local, untyped notion of progress."""
|
|
self._touches.clear()
|
|
self._fired.clear()
|
|
self.gen += 1
|
|
self._evidence_counts[evidence.value] = self._evidence_counts.get(evidence.value, 0) + 1
|
|
LOG.debug("taskv3 canonical progress clear", evidence=evidence.value)
|
|
|
|
def claim_rung(self, target_key: str, count: int) -> bool:
|
|
"""A rung is one threshold crossing per generation: the first claim wins, and a repeat
|
|
claim while a saturated window PARKS on a fire count is refused."""
|
|
if (target_key, count) in self._fired:
|
|
return False
|
|
self._fired.add((target_key, count))
|
|
return True
|
|
|
|
def invalidate_marks(self) -> None:
|
|
"""A look renumbered the manifest, so every mark=N key now names an arbitrary element: drop
|
|
them rather than let four different controls alias into one false streak. A surviving mark
|
|
streak is therefore always within a single manifest generation. Selector keys stay — their
|
|
identity outlives the renumbering."""
|
|
self._touches = [touch for touch in self._touches if not touch[0].startswith("mark=")]
|
|
self._fired = {entry for entry in self._fired if not entry[0].startswith("mark=")}
|
|
|
|
def survival_fields(self) -> dict[str, int]:
|
|
"""The per-run survival numbers, read through the class for the same reason progress() is
|
|
the one write choke point: the ring's state has no references outside this body. Every
|
|
evidence class carries an explicit 0, so a class that never fired stays distinguishable
|
|
from one this record forgot to emit."""
|
|
return {
|
|
"peak_same_touches": self.peak_same_touches,
|
|
"peak_same_errors": self.peak_same_errors,
|
|
"looping_targets": self.looping_targets(),
|
|
**{f"clear_{member.value}": self._evidence_counts.get(member.value, 0) for member in _ProgressEvidence},
|
|
}
|
|
|
|
def looping_targets(self) -> int:
|
|
counts: dict[str, list[bool]] = {}
|
|
for key, err in self._touches:
|
|
counts.setdefault(key, []).append(err)
|
|
return sum(
|
|
1
|
|
for errs in counts.values()
|
|
if len(errs) >= CANONICAL_LOOP_TOUCHES and sum(errs) >= CANONICAL_LOOP_TOUCHES - 1
|
|
)
|
|
|
|
|
|
def _logged_url(url: str) -> str:
|
|
"""A URL reduced to what an indexed field may carry: scheme and host, never path, query, userinfo
|
|
or fragment.
|
|
|
|
Unconditional, the model's own argument included. A signed or sign-in URL is a bearer secret
|
|
wherever it came from, and the model types back the one a page just showed it, so "the model typed
|
|
this" is no evidence the URL is safe to keep whole.
|
|
|
|
The exception is a payload ref, which is logged as the token it is: a name for the target that is
|
|
not the address. Membership in the run's minted refs, never shape — the same rule the model-facing
|
|
boundary masks by.
|
|
|
|
Never raises. The argument is model-typed, and this is evaluated to build the kwargs of the loop's
|
|
`taskv3 tool handler raised` line — inside that `except` block — so a raise here escapes the block,
|
|
the per-call try and the batch loop, aborting the run instead of producing a tool error.
|
|
"""
|
|
ctx = skyvern_context.current()
|
|
if ctx is not None and url in ctx.opaque_url_refs:
|
|
return url[:LOGGED_URL_MAX_CHARS]
|
|
try:
|
|
host = urlsplit(url).hostname
|
|
if host is None or not _LOGGABLE_HOST.fullmatch(host):
|
|
return "<redacted>"
|
|
return redact_url_secrets(url)[:LOGGED_URL_MAX_CHARS]
|
|
except ValueError:
|
|
# urlsplit parses the port and the IPv6 brackets lazily, on attribute access: a non-numeric or
|
|
# out-of-range port and an unclosed bracket each raise here, not at the split.
|
|
return "<redacted>"
|
|
|
|
|
|
def _navigate_record_fields(tool_name: str, args: dict[str, Any], result: ToolResult | None) -> dict[str, str | bool]:
|
|
"""What a `navigate` record is about: the URL asked for, where it landed, the driver's code, and
|
|
whether the browser was already on the page it was sent to.
|
|
|
|
Empty for every other tool, so no facet is added fleet-wide. Both URLs are attribution only —
|
|
scheme and host — which is what a fleet read of navigate outcomes needs and all a log line may
|
|
hold. The requested one is the ARGUMENT, never the resolved one: an opaque payload ref or a
|
|
credential placeholder must stay unresolved here.
|
|
"""
|
|
if tool_name != "navigate":
|
|
return {}
|
|
fields: dict[str, str | bool] = {}
|
|
requested = args.get("url")
|
|
if isinstance(requested, str) and requested:
|
|
fields["requested_url"] = _logged_url(requested)
|
|
data = (result.data if result is not None else None) or {}
|
|
landed = data.get("landed_url")
|
|
if isinstance(landed, str) and landed:
|
|
# Scrubbed again rather than trusted: the tool already reduced it, and the rule belongs to the
|
|
# field, not to whichever caller filled it.
|
|
fields["landed_url"] = _logged_url(landed)
|
|
# A driver's own net:: code, which is a closed vocabulary carrying no address.
|
|
nav_error_code = data.get("nav_error_code")
|
|
if isinstance(nav_error_code, str) and nav_error_code:
|
|
fields["nav_error_code"] = nav_error_code[:LOGGED_URL_MAX_CHARS]
|
|
# Whether the requested URL was the one the browser was already showing. Nothing reads it but a
|
|
# post-deploy rate: how often the model sends the browser back to the page it is already on.
|
|
same_page = data.get("same_page")
|
|
if isinstance(same_page, bool):
|
|
fields["same_page"] = same_page
|
|
# Which fact the `committed_not_loaded` class is about on this row: a document that never became
|
|
# ready, or a readyState read that failed on a wedged renderer. Present only on that class, so a
|
|
# rate taken over it can exclude probe failures instead of silently mixing them in.
|
|
readiness_read_failed = data.get("readiness_read_failed")
|
|
if isinstance(readiness_read_failed, bool):
|
|
fields["readiness_read_failed"] = readiness_read_failed
|
|
return fields
|
|
|
|
|
|
def _menu_note_record_fields(tool_name: str, result: ToolResult | None) -> dict[str, str | int]:
|
|
"""Whether a click's menu-open note listed its rows or withheld them, and why, re-checked against the closed
|
|
vocabulary. Empty for every other tool and for clicks with no note."""
|
|
if tool_name != "click" or result is None:
|
|
return {}
|
|
data = result.data or {}
|
|
fields: dict[str, str | int] = {}
|
|
note = data.get("menu_note")
|
|
if note in ("listed", "withheld"):
|
|
fields["menu_note"] = note
|
|
rows = data.get("menu_rows")
|
|
if isinstance(rows, int) and not isinstance(rows, bool):
|
|
fields["menu_rows"] = rows
|
|
reason = data.get("withhold_reason")
|
|
if note == "withheld" and reason in ("declared_row_over_caps", "bare_text_beside", "single_row_pieces"):
|
|
fields["withhold_reason"] = reason
|
|
return fields
|
|
|
|
|
|
def _observe_summary_fields(result: ToolResult) -> dict[str, int]:
|
|
"""Counts only: the summary is built by the tool, but an indexed field is re-checked here."""
|
|
summary = (result.data or {}).get("summary")
|
|
if not isinstance(summary, dict):
|
|
return {}
|
|
return {
|
|
key: value
|
|
for key, value in summary.items()
|
|
if key not in _TOOL_CALL_RECORD_FIELDS and isinstance(value, int) and not isinstance(value, bool)
|
|
}
|
|
|
|
|
|
def _append_skipped_tool_results(
|
|
messages: list[dict[str, Any]], remaining: list[tuple[str, str, dict[str, Any]]], reason: str
|
|
) -> None:
|
|
"""Answer tool_calls we stopped before executing, so every id in the assistant turn has a
|
|
matching tool result. An unanswered tool_call is an invalid transcript for the next call."""
|
|
for tool_call_id, tool_name, _args in remaining:
|
|
messages.append(
|
|
{"role": "tool", "tool_call_id": tool_call_id, "name": tool_name, "content": f"skipped: {reason}"}
|
|
)
|
|
|
|
|
|
# The `failed` / `terminated` line, stated once and read by every model-facing surface that mentions
|
|
# the finish statuses. It is not a new policy: v1's planner and completion-check prompts never offer
|
|
# the model `failed` at all -- `TERMINATE` is its only negative verdict, and every v1 `failed` is
|
|
# written by the harness itself (an exception, a retry/step ceiling). `TaskStatus.can_update_to`
|
|
# encodes the same split: `terminated` is reachable only from `running`, while `failed` can be set on
|
|
# a run that never started. So `failed` is the engine's own breakdown and `terminated` is the verdict
|
|
# a run reaches about the task -- which is what v3 never told the model (SKY-16648).
|
|
# The closing override is v1's own contract: its validation prompt TERMINATEs when a terminate criterion
|
|
# holds, so a criterion-bearing goal outranks the negative-answer-is-completed default.
|
|
# Wording is load-bearing: two trims that kept the meaning moved 24/40 negative lookups to terminated.
|
|
# Kept under ~1000 characters: no other tool this engine ships has a longer description, and the
|
|
# OpenAI function schema has historically capped a description at 1024.
|
|
FINISH_STATUS_TAXONOMY = (
|
|
"Call this only when the goal is met, or when you have established that it cannot be met. The "
|
|
"three statuses are not interchangeable:\n"
|
|
"- completed: the goal was achieved. A goal that asked you to find out or report something is "
|
|
"achieved once you have a definite answer, including a negative one -- report it in "
|
|
"extracted_output.\n"
|
|
"- terminated: the TASK cannot be achieved, and you can say why from what you saw -- a record "
|
|
"you needed does not exist, a lookup returned no match, a rule excludes it, a wall you cannot "
|
|
"pass, a site that is down.\n"
|
|
"- failed: YOU could not carry the task out, so you cannot say what the outcome is -- your "
|
|
"tools kept erroring, you lost track of the page, or you never got far enough to judge.\n"
|
|
"The test is whose side the reason is on, not whether you have one: the site or the task means "
|
|
"terminated, you or your tools means failed. The goal's own completion or termination criteria "
|
|
"override these defaults."
|
|
)
|
|
|
|
|
|
def _non_completed(status: object) -> NonCompletedStatus | None:
|
|
if status == "failed":
|
|
return "failed"
|
|
if status == "terminated":
|
|
return "terminated"
|
|
return None
|
|
|
|
|
|
def make_finish_tool(
|
|
page_fingerprint: Callable[[], Awaitable[str | None]] | None = None,
|
|
max_settle_deferrals: int = DEFAULT_MAX_SETTLE_DEFERRALS,
|
|
should_cancel: Callable[[], Awaitable[bool]] | None = None,
|
|
deadline_at: float | None = None,
|
|
settle_wait_seconds: float = 0.7,
|
|
activity: ActivityRecency | None = None,
|
|
max_failure_deferrals: int = 1,
|
|
failure_settle_max_seconds: float = FAILURE_EVIDENCE_SETTLE_MAX_SECONDS,
|
|
pending_marker: Callable[[str], Awaitable[str | None]] | None = None,
|
|
submit_watch: SubmitWatch | None = None,
|
|
completion_blocker: CompletionBlocker | None = None,
|
|
staged_downloads: set[str] | None = None,
|
|
verification_blocker: VerificationBlocker | None = None,
|
|
goal_check: Callable[[], Awaitable[GoalVerdict]] | None = None,
|
|
goal_check_enforce: bool = False,
|
|
unlisted_reask: UnlistedReaskCheck | None = None,
|
|
document_identity: Callable[[], Awaitable[str | None]] | None = None,
|
|
) -> ToolSpec:
|
|
"""`page_fingerprint` samples an opaque fingerprint of the page's rendered content (None when no
|
|
page is available). A finish(completed) is deferred (bounded by `max_settle_deferrals`, then
|
|
accepted) unless two samples `settle_wait_seconds` apart match, so the model re-verifies against
|
|
the settled state instead of a mid-render shell — delayed loads otherwise produce stochastic
|
|
false completions. A sampling error is unknown, not settled: it defers. The wait between samples
|
|
is capped at `deadline_at` (time.monotonic clock) and abandoned once `should_cancel` reports
|
|
True, so probing cannot outlive the loop's own bounds.
|
|
|
|
The symmetric failure side: when `activity` reports recent submit-class/captcha activity, a
|
|
finish(failed) is held for ONE evidence turn (`max_failure_deferrals`, per run like the
|
|
completed-side cap, not per verdict attempt) — a quiescence wait
|
|
bounded by `failure_settle_max_seconds`, then a deferral asking the model to re-observe —
|
|
because async submissions and captcha protocols otherwise produce false-negative verdicts.
|
|
|
|
`verification_blocker` is the one gate consulted for EVERY verdict: it refuses a completed claim
|
|
once the verification source terminally failed, and holds a failed OR terminated one while the
|
|
run is still awaiting a code it has unspent polling budget for -- a give-up at one 120s slice of
|
|
a 15-minute budget throws away minutes of waiting the run already owns. The hold is bounded by
|
|
the callee (the budget shrinks under every productive hold) and refused here without the deadline
|
|
headroom to fund the blocking poll slice it asks for. Apart from that gate, terminated is
|
|
ungated on both sides.
|
|
|
|
`pending_marker` reports the text the page still shows the control in `submit_watch` as in
|
|
flight with, or None. A settled page is not a submitted one -- a submit frozen mid-flight is
|
|
maximally stable, so the settle probe is satisfied by exactly the state it should refuse. A
|
|
completed verdict is held ONCE against that marker, with the re-observe it asks for budgeted; if
|
|
the model insists a second time its verdict stands, so a run that declares completion on a
|
|
still-pending control remains possible after this gate. Deliberately a POSITIVE observation -- a
|
|
probe that fails reports nothing, and nothing is not evidence of pending, so it accepts rather
|
|
than holding a run on probe flakiness.
|
|
|
|
`unlisted_reask` asks once per run whether a failed or terminated verdict that is about to stand only
|
|
missed a screen the site skipped. A grounded yes becomes completed unless a completed-side gate vetoes it;
|
|
it never gives a turn back, and a completed verdict cannot reach it. Its settle gate samples as many times as
|
|
the completed path would defer. When `document_identity` is wired, it is read before the judge and again after
|
|
the settle window, and a conversion whose identity changed or could not be read is refused, settled or not."""
|
|
deferrals = 0
|
|
failure_deferrals = 0
|
|
goal_check_held = False
|
|
reask_asked = False
|
|
|
|
async def _bounded_fingerprint() -> str | None:
|
|
"""page_fingerprint(), timed out against whatever is left of the run's deadline instead of
|
|
the flat default cap -- so a HANGING sampler can no longer run the full default timeout once
|
|
the deadline is nearly gone. Mirrors _sample_probe's own convention once the deadline has
|
|
fully elapsed: the call is skipped entirely (never even invoked, so a hanging implementation
|
|
is never awaited) and reads as a missing sample, same as a genuinely absent page. That is
|
|
deliberately NOT the same as an exception -- a real sampling error must still propagate and
|
|
defer (see _settled/_quiesced's docstrings), and a fast probe that would have answered
|
|
instantly must not be denied the chance just because the deadline's clock already read zero;
|
|
only a positive-but-reduced timeout, not an outright skip, can preserve that."""
|
|
assert page_fingerprint is not None # gated by each call site's own None check
|
|
timeout = _PAGE_PROBE_TIMEOUT_SECONDS
|
|
if deadline_at is not None:
|
|
remaining = deadline_at - time.monotonic()
|
|
if remaining <= 0:
|
|
return None
|
|
timeout = min(timeout, remaining)
|
|
return await _bounded_probe(page_fingerprint(), timeout=timeout)
|
|
|
|
async def _quiesced() -> bool:
|
|
"""Bounded wait for the page to stop mutating before the failure verdict's evidence turn.
|
|
Returns False when there is no page to observe (a deferral would burn a turn for nothing)."""
|
|
prev = await _bounded_fingerprint()
|
|
if prev is None:
|
|
return False
|
|
cap_at = time.monotonic() + failure_settle_max_seconds
|
|
while True:
|
|
wait = min(settle_wait_seconds, cap_at - time.monotonic())
|
|
if deadline_at is not None:
|
|
wait = min(wait, deadline_at - time.monotonic())
|
|
if wait <= 0:
|
|
return True
|
|
await asyncio.sleep(wait)
|
|
if should_cancel is not None and await should_cancel():
|
|
return True # defer: the loop's cancellation check ends the run before another turn
|
|
current = await _bounded_fingerprint()
|
|
if current is None:
|
|
return False
|
|
if current == prev:
|
|
return True
|
|
prev = current
|
|
|
|
async def _settled() -> bool:
|
|
first = await _bounded_fingerprint()
|
|
if first is None:
|
|
return True # no page to sample (non-recovering peek): accept the verdict as-is
|
|
wait = settle_wait_seconds
|
|
if deadline_at is not None:
|
|
wait = min(wait, deadline_at - time.monotonic())
|
|
if wait > 0:
|
|
await asyncio.sleep(wait)
|
|
if should_cancel is not None and await should_cancel():
|
|
return False # defer: the loop's cancellation check ends the run before another turn
|
|
return first == await _bounded_fingerprint()
|
|
|
|
async def _conversion_veto(result: UnlistedReask, identity_before: str | None) -> str | None:
|
|
"""The completed-side gates, as vetoes only: the model never claimed completion, so none may hold."""
|
|
try:
|
|
if should_cancel is not None and await should_cancel():
|
|
return "canceled"
|
|
except Exception:
|
|
return "canceled"
|
|
if pending_marker is not None and submit_watch is not None and submit_watch.selector:
|
|
timeout = _PAGE_PROBE_TIMEOUT_SECONDS
|
|
if deadline_at is not None:
|
|
timeout = min(timeout, deadline_at - time.monotonic())
|
|
if timeout <= 0:
|
|
return "deadline"
|
|
try:
|
|
if await _bounded_probe(pending_marker(submit_watch.selector), timeout=timeout):
|
|
return "pending_marker"
|
|
except Exception:
|
|
return "pending_marker"
|
|
if verification_blocker is not None:
|
|
try:
|
|
if await verification_blocker(CONVERSION_VERIFICATION_STATUS):
|
|
return "verification_blocker"
|
|
except Exception:
|
|
return "verification_blocker"
|
|
if page_fingerprint is not None:
|
|
if document_identity is not None and identity_before is None:
|
|
return "identity_unreadable"
|
|
settled = False
|
|
for _ in range(max_settle_deferrals + 1):
|
|
result.settle_rounds += 1
|
|
try:
|
|
settled = await _settled()
|
|
except Exception:
|
|
pass # as on the completed path: an unreadable sample is an unsettled one
|
|
result.settled = settled
|
|
if settled:
|
|
break
|
|
try:
|
|
if should_cancel is not None and await should_cancel():
|
|
return "canceled"
|
|
except Exception:
|
|
return "canceled"
|
|
if deadline_at is not None and deadline_at - time.monotonic() <= 0:
|
|
return "deadline"
|
|
if deadline_at is not None and deadline_at - time.monotonic() <= 0:
|
|
return "deadline"
|
|
if document_identity is not None:
|
|
identity_after = await _sample_probe(document_identity, deadline_at=deadline_at)
|
|
if identity_after is None:
|
|
return "identity_unreadable"
|
|
if identity_before != identity_after:
|
|
return "navigating"
|
|
elif not settled:
|
|
return "unsettled"
|
|
if goal_check is not None:
|
|
try:
|
|
verdict = await goal_check()
|
|
except Exception:
|
|
LOG.warning("taskv3 goal check of a reask conversion failed", exc_info=True)
|
|
return "goal_check" if goal_check_enforce else None
|
|
result.goal_check_verdict = verdict.verdict
|
|
# One check, no hold: the model never claimed completion, so there is no turn to give back.
|
|
if goal_check_enforce and verdict.is_contradiction:
|
|
return "goal_check"
|
|
return None
|
|
|
|
async def _reask(original: NonCompletedStatus, args: dict[str, Any]) -> ToolResult | None:
|
|
assert unlisted_reask is not None
|
|
reason = args.get("reason") or ""
|
|
# Before the judge: a navigation that commits while it runs must not become the baseline.
|
|
identity_before = (
|
|
await _sample_probe(document_identity, deadline_at=deadline_at)
|
|
if document_identity is not None and page_fingerprint is not None
|
|
else None
|
|
)
|
|
try:
|
|
result: UnlistedReask | None = await unlisted_reask(original, reason)
|
|
except Exception:
|
|
LOG.warning("taskv3 unlisted reask failed; keeping the verdict", exc_info=True)
|
|
result = None
|
|
if result is not None and result.converts:
|
|
result.veto = await _conversion_veto(result, identity_before)
|
|
result.converted = result.veto is None
|
|
LOG.info(
|
|
"taskv3 finish unlisted reask",
|
|
original_status=original,
|
|
verdict=result.verdict if result is not None else None,
|
|
terminate_criterion_holds=result.terminate_criterion_holds if result is not None else None,
|
|
converts=result.converts if result is not None else False,
|
|
converted=result.converted if result is not None else False,
|
|
veto=result.veto if result is not None else None,
|
|
settled=result.settled if result is not None else None,
|
|
settle_rounds=result.settle_rounds if result is not None else 0,
|
|
goal_check_verdict=result.goal_check_verdict if result is not None else None,
|
|
llm_key=result.llm_key if result is not None else None,
|
|
reask_llm_key=result.reask_llm_key if result is not None else None,
|
|
skipped_reason=result.skipped_reason if result is not None else "reask_error",
|
|
# Never the quote itself: it is page text, possibly customer data.
|
|
quote_chars=result.quote_chars if result is not None else 0,
|
|
latency_s=result.latency_s if result is not None else None,
|
|
turn=activity.turn if activity is not None else None,
|
|
)
|
|
if result is None or not result.converted:
|
|
return None
|
|
return ToolResult.ok(
|
|
content="Task attempt ended. No further actions are permitted.",
|
|
data={
|
|
"status": "completed",
|
|
"reason": result.reason,
|
|
"extracted_output": args.get("extracted_output"),
|
|
"converted_from": original,
|
|
"converted_from_reason": reason,
|
|
},
|
|
)
|
|
|
|
async def handler(args: dict[str, Any]) -> ToolResult:
|
|
nonlocal deferrals, failure_deferrals, goal_check_held, reask_asked
|
|
status = args.get("status")
|
|
if status not in ("completed", "failed", "terminated"):
|
|
return ToolResult.error(
|
|
f"invalid finish status: {status!r}; call finish again with status=completed|failed|terminated"
|
|
)
|
|
if (
|
|
status == "completed"
|
|
and pending_marker is not None
|
|
and submit_watch is not None
|
|
and submit_watch.selector
|
|
and not submit_watch.deferred
|
|
# Holding costs a turn, the tool calls of the re-observe it asks for, and deadline
|
|
# seconds. Without the headroom to spend them the run ends budget_exhausted, which is
|
|
# unmapped and lands on failed -- turning an honest hold into a false failure.
|
|
and _has_hold_headroom(activity, deadline_at)
|
|
):
|
|
try:
|
|
marker = await _bounded_probe(pending_marker(submit_watch.selector))
|
|
except Exception:
|
|
# Warning, not debug: the only way here is a broken probe, and a silently disabled
|
|
# gate reads exactly like a page that was never pending.
|
|
LOG.warning("taskv3 pending-marker probe failed; not treating it as pending", exc_info=True)
|
|
marker = None
|
|
if marker:
|
|
submit_watch.deferred = True
|
|
LOG.info("taskv3 completed verdict held: submission still in flight", marker=marker)
|
|
return ToolResult.error(
|
|
f"the page still shows a submission in flight: {marker}. That is not a "
|
|
"settled failure OR a confirmation -- wait for it to resolve and re-observe. "
|
|
"Finish with status=completed only once the page shows the submission "
|
|
"landed; if it never resolves, say so with status=terminated."
|
|
)
|
|
if status == "completed" and completion_blocker is not None:
|
|
try:
|
|
blocker_message = await completion_blocker(frozenset(staged_downloads or ()))
|
|
except Exception:
|
|
# Fail closed: a download-gated task must not complete on a storage hiccup with no file.
|
|
LOG.warning("taskv3 completion_blocker failed; failing closed", exc_info=True)
|
|
return ToolResult.error(
|
|
"Could not verify that a file download has finished; retry finish(status=completed) "
|
|
"once the download has landed, or finish with the non-completed status the "
|
|
"finish tool's own rule gives this outcome."
|
|
)
|
|
if blocker_message:
|
|
return ToolResult.error(blocker_message)
|
|
if verification_blocker is not None and (
|
|
status == "completed"
|
|
# A non-complete verdict is only ever HELD, never refused, so unlike the completed side it
|
|
# must fund the retry it asks for -- and that retry is a blocking poll slice, not just a
|
|
# re-observe. Absent the accounting to check that (no `activity`), the hold is not
|
|
# justifiable and the verdict stands.
|
|
or (
|
|
activity is not None
|
|
and _has_hold_headroom(
|
|
activity,
|
|
deadline_at,
|
|
min_deadline_headroom_seconds=VERIFICATION_GIVEUP_MIN_DEADLINE_HEADROOM_SECONDS,
|
|
)
|
|
)
|
|
):
|
|
try:
|
|
verification_message = await verification_blocker(status)
|
|
except Exception:
|
|
if status == "completed":
|
|
# Fail closed: an exception here must not let a blank verification step read as done.
|
|
LOG.warning("taskv3 verification_blocker failed; failing closed", exc_info=True)
|
|
return ToolResult.error(
|
|
"Could not verify that the verification-code step completed cleanly; retry "
|
|
"finish(status=completed) once verified, or finish with the non-completed "
|
|
"status the finish tool's own rule gives this outcome."
|
|
)
|
|
# Fail open on the give-up side: the opposite verdict. A broken gate must not trap a
|
|
# run that wants to end.
|
|
LOG.warning("taskv3 verification_blocker failed; honoring the verdict", exc_info=True)
|
|
verification_message = None
|
|
if verification_message:
|
|
return ToolResult.error(verification_message)
|
|
if (
|
|
status == "completed"
|
|
and page_fingerprint is not None
|
|
and deferrals < max_settle_deferrals
|
|
# On the granted final turn the re-verification turn a deferral asks for no longer
|
|
# exists: holding would convert the verdict into budget_exhausted instead.
|
|
and not (activity is not None and activity.final_turn_active)
|
|
):
|
|
try:
|
|
settled = await _settled()
|
|
except Exception:
|
|
# Fail closed: an exception while probing is evidence of nothing, so the verdict is
|
|
# deferred for re-verification rather than validated. The deferral cap still bounds it.
|
|
settled = False
|
|
if not settled:
|
|
deferrals += 1
|
|
return ToolResult.error(
|
|
"the page was still rendering, or could not be verified as settled, when you "
|
|
"called finish. Wait for it to settle, re-observe, confirm the goal's effect is "
|
|
"present in the loaded content (not a loading indicator or empty container), "
|
|
"then finish again."
|
|
)
|
|
if status == "completed" and goal_check is not None:
|
|
try:
|
|
verdict: GoalVerdict | None = await goal_check()
|
|
except Exception:
|
|
LOG.warning("taskv3 finish goal check failed; accepting the verdict", exc_info=True)
|
|
verdict = None
|
|
if verdict is not None:
|
|
# Every first contradiction is held, `impossible` included: a transient error page or
|
|
# a page still loading reads as the site refusing, and one re-check absorbs it. Only a
|
|
# verdict AFTER a hold can end the run, and the second verdict decides how.
|
|
action: GoalCheckAction
|
|
if verdict.verdict == "achieved":
|
|
action = "accept"
|
|
elif not goal_check_held:
|
|
action = "hold"
|
|
verdict.no_headroom = not _has_hold_headroom(activity, deadline_at)
|
|
elif verdict.verdict == "impossible":
|
|
action = "terminate"
|
|
else:
|
|
action = "fail"
|
|
verdict.action = action
|
|
second: GoalVerdict | None = None
|
|
second_skipped_reason: str | None = None
|
|
if not goal_check_enforce and action == "hold" and not verdict.no_headroom:
|
|
# Shadow accepts here and the run ends, so the second check enforce would make after
|
|
# the held turn is asked now instead, once. An upper bound on enforce's false
|
|
# failures: enforce's agent gets a turn to scroll or re-observe first.
|
|
try:
|
|
cancelled = should_cancel is not None and await should_cancel()
|
|
except Exception:
|
|
cancelled = True
|
|
if cancelled:
|
|
second_skipped_reason = "cancelled"
|
|
else:
|
|
try:
|
|
second = await goal_check()
|
|
except Exception:
|
|
LOG.warning("taskv3 finish goal check recheck failed", exc_info=True)
|
|
if second is not None:
|
|
second.recheck = True
|
|
second_skipped_reason = second.skipped_reason
|
|
second.action = (
|
|
("terminate" if second.verdict == "impossible" else "fail")
|
|
if second.is_contradiction
|
|
else "accept"
|
|
)
|
|
# A skipped or failed recheck is no answer; an ungrounded contradiction is one
|
|
# (enforce would accept it).
|
|
if second.skipped_reason in (None, "ungrounded_quote"):
|
|
verdict.would_fail = second.is_contradiction
|
|
LOG.info(
|
|
"taskv3 finish goal check",
|
|
mode="enforce" if goal_check_enforce else "shadow",
|
|
verdict=verdict.verdict,
|
|
would_be_action=action,
|
|
no_headroom=verdict.no_headroom,
|
|
# Never the quote itself: it is page text, possibly customer data, and this index is
|
|
# searchable. The judge's full response is stored as the step's LLM response artifact.
|
|
quote_chars=len(verdict.quote),
|
|
quote_source=verdict.quote_source,
|
|
skipped_reason=verdict.skipped_reason,
|
|
latency_s=verdict.latency_s,
|
|
second_verdict=second.verdict if second is not None else None,
|
|
second_skipped_reason=second_skipped_reason,
|
|
second_latency_s=second.latency_s if second is not None else None,
|
|
would_fail=verdict.would_fail,
|
|
turn=activity.turn if activity is not None else None,
|
|
)
|
|
try:
|
|
canceled_during_check = should_cancel is not None and await should_cancel()
|
|
except Exception:
|
|
canceled_during_check = False
|
|
if canceled_during_check:
|
|
# Defer, like the settle probe: the loop's cancellation check ends the run before another
|
|
# turn, instead of this finish persisting a completion for a canceled run.
|
|
return ToolResult.error("the run was canceled while the completion was being checked.")
|
|
if goal_check_enforce and action == "hold" and not verdict.no_headroom:
|
|
goal_check_held = True
|
|
if activity is not None:
|
|
activity.held_verdict_batch_skip = True
|
|
found = (
|
|
"the site preventing the goal"
|
|
if verdict.verdict == "impossible"
|
|
else "them contradicting the goal"
|
|
)
|
|
return ToolResult.error(
|
|
"completed verdict held once: an independent check of the page and of your recent "
|
|
f"tool results found {found}: {verdict.bounded_missing.rstrip('.')}. Look at the page "
|
|
"once more and confirm which verdict is right before finishing again."
|
|
)
|
|
if goal_check_enforce and action in ("fail", "terminate"):
|
|
return ToolResult.ok(
|
|
content="Task attempt ended. No further actions are permitted.",
|
|
data={
|
|
"status": "failed" if action == "fail" else "terminated",
|
|
"reason": f"goal check: {verdict.bounded_missing}",
|
|
"extracted_output": None,
|
|
"goal_check_ended": True,
|
|
},
|
|
)
|
|
if (
|
|
status == "failed"
|
|
and activity is not None
|
|
and page_fingerprint is not None
|
|
and failure_deferrals < max_failure_deferrals
|
|
# Same rule as the completed-side holds: no retry turn exists on the granted final turn.
|
|
and not activity.final_turn_active
|
|
and activity.armed()
|
|
# The corrected cycle needs budget for its worst case (wait + observe + re-finish);
|
|
# without headroom on every budget axis a deferral would convert an honest failure
|
|
# into budget_exhausted (or, for a stall-streak one short of the terminator, into a
|
|
# generic stall termination that replaces the model's accurate reason).
|
|
and (activity.turns_remaining is None or activity.turns_remaining >= FAILURE_EVIDENCE_MIN_TURNS)
|
|
and (
|
|
activity.tool_calls_remaining is None
|
|
or activity.tool_calls_remaining >= FAILURE_EVIDENCE_MIN_TOOL_CALLS
|
|
)
|
|
# The token margin is deliberately approximate: sized off the triggering turn, while
|
|
# the deferral turns carry a slightly larger transcript.
|
|
and (
|
|
activity.tokens_remaining is None
|
|
or activity.tokens_remaining >= FAILURE_EVIDENCE_MIN_TURNS * max(activity.last_turn_tokens, 1)
|
|
)
|
|
and not activity.perception_stall_imminent
|
|
and (
|
|
deadline_at is None or deadline_at - time.monotonic() >= FAILURE_EVIDENCE_MIN_DEADLINE_HEADROOM_SECONDS
|
|
)
|
|
):
|
|
should_defer = True
|
|
try:
|
|
# False only when there is no page to observe; cancellation mid-wait still defers
|
|
# (the loop's own cancel check ends the run first).
|
|
should_defer = await _quiesced()
|
|
except Exception:
|
|
pass # unknown page state still defers: the model's re-observe is the evidence step
|
|
if should_defer:
|
|
failure_deferrals += 1
|
|
LOG.info("taskv3 finish failure deferred for evidence", turn=activity.turn)
|
|
return ToolResult.error(
|
|
"failure verdict held for one evidence check: it follows recent page actions or "
|
|
"a captcha attempt whose effects can land after your last look — submissions and "
|
|
"captcha protocols often complete asynchronously, so the page may no longer show "
|
|
"the state this verdict was based on. Re-observe the page once (waiting briefly "
|
|
"first if it may still be processing): only a positive confirmation of the goal "
|
|
"(e.g. a submission confirmation banner) justifies finishing with "
|
|
"status=completed; if it still shows the blocked or failed state, or shows no "
|
|
"positive confirmation at all, finish again -- status=terminated if the page "
|
|
"itself is what stopped you, status=failed if the breakdown was yours -- and "
|
|
"the verdict will stand."
|
|
)
|
|
# Sampled here, after the failure-evidence gate, rather than at the run's first finish: a zero-action
|
|
# `finish(completed)` refused by a gate above and followed by `finish(failed)` records the failed verdict.
|
|
if activity is not None and status in ("failed", "terminated") and activity.status_at_hold_gate is None:
|
|
activity.attempts_at_hold_gate = activity.action_attempts
|
|
activity.perceptions_at_hold_gate = activity.perceptions
|
|
activity.status_at_hold_gate = status
|
|
original = _non_completed(status)
|
|
if original is not None and unlisted_reask is not None and not reask_asked and not goal_check_held:
|
|
reask_asked = True
|
|
converted = await _reask(original, args)
|
|
try:
|
|
canceled_during_reask = should_cancel is not None and await should_cancel()
|
|
except Exception:
|
|
canceled_during_reask = False
|
|
if canceled_during_reask:
|
|
# Defer, like the goal check: the loop's cancellation check persists `canceled` before another turn.
|
|
return ToolResult.error("the run was canceled while the verdict was being checked.")
|
|
if converted is not None:
|
|
return converted
|
|
return ToolResult.ok(
|
|
content="Task attempt ended. No further actions are permitted.",
|
|
data={
|
|
"status": status,
|
|
"reason": args.get("reason") or "",
|
|
"extracted_output": args.get("extracted_output"),
|
|
},
|
|
)
|
|
|
|
return ToolSpec(
|
|
name="finish",
|
|
description=("End the task and report whether the browser goal was completed. " + FINISH_STATUS_TAXONOMY),
|
|
parameters={
|
|
"type": "object",
|
|
"properties": {
|
|
"status": {"type": "string", "enum": ["completed", "failed", "terminated"]},
|
|
"reason": {"type": "string", "maxLength": 2000},
|
|
"extracted_output": {"description": "Structured output requested by the goal, if any."},
|
|
},
|
|
"required": ["status", "reason"],
|
|
},
|
|
handler=handler,
|
|
terminal=True,
|
|
)
|
|
|
|
|
|
# Caps the arguments an elision placeholder echoes: a selector is model-authored and unbounded.
|
|
_READ_LABEL_MAX_CHARS = 120
|
|
|
|
|
|
def _declared_args_key(spec: ToolSpec, args: dict[str, Any]) -> str:
|
|
"""The argument part of a read's key (`_PerceptionEntry.key`): only the arguments the tool DECLARES.
|
|
|
|
An argument the tool does not declare cannot change what it reads, so it cannot make two calls different
|
|
reads (the page it reads can, which is the page part of the key). The specs are not emitted strict, so a
|
|
provider is free to add one — and for an argumentless tool that would split the key and retain two
|
|
snapshots where the tool only ever describes the page as it is NOW.
|
|
|
|
For `look` that is not merely wasted context. Every call disposes the previous handles, clears
|
|
`_look_manifest` and renumbers the marks (`tools.py`), so a retained older legend describes numbers
|
|
that now address different controls, and `click(mark=N)` following it acts on the wrong one. It
|
|
fails open: the stale legend looks perfectly valid. `look` and `observe` declare none and carry no
|
|
page part (`issues_handles`), so all their calls collapse to one key and exactly one survives.
|
|
"""
|
|
declared = (spec.parameters or {}).get("properties") or {}
|
|
return json.dumps({k: v for k, v in args.items() if k in declared}, sort_keys=True, default=str)
|
|
|
|
|
|
def _read_label(tool_name: str, args_key: str) -> str:
|
|
"""How an elided snapshot names the read it dropped, e.g. `get_html(selector=#rows, offset=20000)`.
|
|
|
|
A tool whose reads cannot differ is named bare, exactly as before: observe and look take no
|
|
arguments, so decorating them would add a token to every elision and distinguish nothing. The
|
|
arguments are the model's own, echoed from the assistant message that already carries them, so
|
|
this discloses nothing the transcript did not already hold — but it is capped anyway, because a
|
|
selector has no length the model cannot choose.
|
|
"""
|
|
try:
|
|
args = json.loads(args_key)
|
|
except (TypeError, ValueError):
|
|
return tool_name
|
|
if not isinstance(args, dict) or not args:
|
|
return tool_name
|
|
rendered = ", ".join(f"{k}={args[k]}" for k in sorted(args))
|
|
if len(rendered) > _READ_LABEL_MAX_CHARS:
|
|
rendered = rendered[:_READ_LABEL_MAX_CHARS] + "…"
|
|
return f"{tool_name}({rendered})"
|
|
|
|
|
|
_COMPACTED_PREFIX = "[superseded "
|
|
|
|
|
|
@dataclass(slots=True)
|
|
class _PerceptionEntry:
|
|
tool: str
|
|
args: str # `_declared_args_key`; the only part of the key a placeholder or marker may name
|
|
page: str # the page probe sampled before the call; "" for a tool that issues handles
|
|
# sha256 of the model-facing bytes before `delta`, not the canonical digest: a re-minted marker is new bytes.
|
|
raw: str
|
|
turn: int
|
|
handles: bool
|
|
# The tool-reported tail at `delta_at` (a text delta of what an action changed). Outside `raw`, so it never
|
|
# makes an unchanged read look changed, and carried verbatim through every placeholder and marker.
|
|
delta: str = ""
|
|
shown: bool = True
|
|
|
|
@property
|
|
def key(self) -> tuple[str, str, str]:
|
|
return (self.tool, self.page, self.args)
|
|
|
|
|
|
@dataclass
|
|
class _PerceptionStore:
|
|
"""Every successful perception read, by message index, and the rewrites that bound them.
|
|
|
|
Read bodies are budgeted by size (`PERCEPTION_RETAIN_CHARS_HIGH`/`_LOW`) and rewritten only in
|
|
epochs, so between epochs each request is a byte prefix of the next. Tools that issue handles
|
|
(`observe`, `look`) keep exactly one body and elide the older one at once: the next call renumbers
|
|
refs/marks, so a stale legend addresses the wrong control and fails open.
|
|
|
|
A re-read whose bytes equal a body still shown becomes a marker citing it. Nothing is predicted: the
|
|
read ran, and this compares what it returned. A marker is not an entry, and an epoch moves a cited
|
|
body into the newest citing marker's slot before it elides anything, so a marker never outlives its body.
|
|
"""
|
|
|
|
entries: dict[int, _PerceptionEntry] = field(default_factory=dict)
|
|
# marker message index -> (cited body index, turn the marker's read was taken, the marker's own delta)
|
|
markers: dict[int, tuple[int, int, str]] = field(default_factory=dict)
|
|
unchanged_marks: int = 0
|
|
epoch_rewrites: int = 0
|
|
retained_chars_peak: int = 0
|
|
|
|
def admit(
|
|
self,
|
|
index: int,
|
|
*,
|
|
tool: str,
|
|
args: str,
|
|
page: str,
|
|
handles: bool,
|
|
content: str,
|
|
turn: int,
|
|
delta_at: int | None = None,
|
|
) -> str:
|
|
"""Record the read about to be appended at `index`; returns what the model is shown for it."""
|
|
cut = delta_at if delta_at is not None and 0 <= delta_at <= len(content) else len(content)
|
|
head, delta = content[:cut], content[cut:]
|
|
entry = _PerceptionEntry(
|
|
tool=tool,
|
|
args=args,
|
|
page="" if handles else page,
|
|
raw=hashlib.sha256(head.encode()).hexdigest(),
|
|
turn=turn,
|
|
handles=handles,
|
|
delta=delta,
|
|
)
|
|
if not handles:
|
|
prior_index = max((i for i, e in self.entries.items() if e.key == entry.key), default=None)
|
|
prior = None if prior_index is None else self.entries[prior_index]
|
|
if prior_index is not None and prior is not None and prior.shown:
|
|
label = _read_label(tool, args)
|
|
if prior.raw == entry.raw:
|
|
# The marker locates its body only by label (the turn number is not in the transcript), so a
|
|
# newer body with the same label from another page would be the one the model looks at.
|
|
latest_same_label = max(
|
|
i for i, e in self.entries.items() if e.shown and e.tool == tool and e.args == args
|
|
)
|
|
marker = (
|
|
f"[{label} returned the same {len(head)} chars as your read at turn {prior.turn}, "
|
|
"still shown above]"
|
|
)
|
|
if latest_same_label == prior_index and len(head) >= PERCEPTION_MARKER_MIN_RATIO * len(marker):
|
|
self.markers[index] = (prior_index, turn, delta)
|
|
self.unchanged_marks += 1
|
|
return marker + delta
|
|
else:
|
|
# On the NEW read: labelling the old one would rewrite a message the provider has cached.
|
|
content = (
|
|
f"[{label} returned different content from your read at turn {prior.turn}; that one "
|
|
f"predates your later actions, this is the current one]\n{content}"
|
|
)
|
|
self.entries[index] = entry
|
|
return content
|
|
|
|
def compact(self, messages: list[dict[str, Any]]) -> None:
|
|
"""Run before every LLM call. The unread round (after the last assistant message) is never elided
|
|
and never charged: one batched turn must not evict every earlier read."""
|
|
last_assistant_idx = -1
|
|
for i in range(len(messages) - 1, -1, -1):
|
|
if messages[i].get("role") == "assistant":
|
|
last_assistant_idx = i
|
|
break
|
|
newest: dict[tuple[str, str, str], int] = {}
|
|
for i in sorted(self.entries):
|
|
if self.entries[i].shown:
|
|
newest[self.entries[i].key] = i
|
|
for i, e in list(self.entries.items()):
|
|
if e.shown and e.handles and i < last_assistant_idx and newest[e.key] != i:
|
|
self._elide(messages, i)
|
|
|
|
read = self._read_bodies(last_assistant_idx)
|
|
charged = sum(len(messages[i]["content"]) for i in read)
|
|
# Two full-size reads (window plus its continuation notice) exceed HIGH, so without a floor an epoch
|
|
# would break paging, which the count window kept at two. The floor never triggers an epoch on its own.
|
|
if charged > PERCEPTION_RETAIN_CHARS_HIGH and len(read) > len(self._floor(read, newest)):
|
|
self.epoch_rewrites += 1
|
|
self._move_cited_bodies(messages)
|
|
newest = {}
|
|
for i in sorted(self.entries):
|
|
if self.entries[i].shown:
|
|
newest[self.entries[i].key] = i
|
|
read = self._read_bodies(last_assistant_idx)
|
|
charged = sum(len(messages[i]["content"]) for i in read)
|
|
floor = self._floor(read, newest)
|
|
older = [i for i in read if newest[self.entries[i].key] != i]
|
|
current = [i for i in read if newest[self.entries[i].key] == i]
|
|
for i in older + current:
|
|
if charged <= PERCEPTION_RETAIN_CHARS_LOW:
|
|
break
|
|
if i in floor:
|
|
continue
|
|
charged -= len(messages[i]["content"])
|
|
self._elide(messages, i)
|
|
self.retained_chars_peak = max(self.retained_chars_peak, charged)
|
|
|
|
def _floor(self, read: list[int], newest: dict[tuple[str, str, str], int]) -> set[int]:
|
|
"""The newest reads the model has seen, one per key, that an epoch never elides."""
|
|
return set([i for i in read if newest[self.entries[i].key] == i][-PERCEPTION_RETAIN_MIN_READS:])
|
|
|
|
def _read_bodies(self, last_assistant_idx: int) -> list[int]:
|
|
return sorted(i for i, e in self.entries.items() if e.shown and not e.handles and i < last_assistant_idx)
|
|
|
|
def _placeholder(self, entry: _PerceptionEntry) -> str:
|
|
return f"{_COMPACTED_PREFIX}{_read_label(entry.tool, entry.args)} output elided to bound context]"
|
|
|
|
def _elide(self, messages: list[dict[str, Any]], index: int) -> None:
|
|
entry = self.entries[index]
|
|
messages[index]["content"] = self._placeholder(entry) + entry.delta
|
|
entry.shown = False
|
|
|
|
def _move_cited_bodies(self, messages: list[dict[str, Any]]) -> None:
|
|
citing: dict[int, list[int]] = {}
|
|
for marker_index, (body_index, _, _) in self.markers.items():
|
|
citing.setdefault(body_index, []).append(marker_index)
|
|
for body_index, marker_indices in citing.items():
|
|
target = max(marker_indices)
|
|
body = self.entries[body_index]
|
|
shown = messages[body_index]["content"]
|
|
_, turn, target_delta = self.markers[target]
|
|
messages[target]["content"] = shown[: len(shown) - len(body.delta)] + target_delta
|
|
self._elide(messages, body_index)
|
|
for i in marker_indices:
|
|
if i != target:
|
|
messages[i]["content"] = self._placeholder(body) + self.markers[i][2]
|
|
self.entries[target] = _PerceptionEntry(
|
|
tool=body.tool,
|
|
args=body.args,
|
|
page=body.page,
|
|
raw=body.raw,
|
|
turn=turn,
|
|
handles=False,
|
|
delta=target_delta,
|
|
)
|
|
self.markers.clear()
|
|
|
|
def log_fields(self) -> dict[str, int]:
|
|
return {
|
|
"perception_unchanged_marks": self.unchanged_marks,
|
|
"perception_epoch_rewrites": self.epoch_rewrites,
|
|
"perception_retained_chars_peak": self.retained_chars_peak,
|
|
}
|
|
|
|
|
|
@dataclass(kw_only=True, slots=True)
|
|
class LoopState:
|
|
"""Run-scoped state of run_agent_tool_loop, held as one typed object rather than threaded
|
|
through the closure as nonlocals.
|
|
|
|
Several fields encode an ordering invariant that no type can express, so the invariant is
|
|
documented at the field it constrains.
|
|
"""
|
|
|
|
outcome: LoopOutcome | None = None
|
|
pending_nav_dead_end: _NavDeadEnd | None = None
|
|
stall_nudges_due: list[tuple[str, int]] = field(default_factory=list)
|
|
# Page-state stall detector (SKY-15265): consecutive billable rounds on a byte-identical
|
|
# fingerprint, whether the one re-plan nudge went out, and whether one is due this turn.
|
|
trailing_page_state_stall_rounds: int = 0
|
|
peak_page_state_stall_rounds: int = 0
|
|
# Whether the detector ever reached a verdict. A wired sampler is not a taken sample: it is only
|
|
# read behind a billable call, so a run that lands none never judges the page at all.
|
|
page_state_ever_judged: bool = False
|
|
page_state_nudge_delivered: bool = False
|
|
page_state_nudge_due: bool = False
|
|
# The download-completion probe refused by the verification gate. The model was told the run
|
|
# auto-completes on download, so a silent refusal leaves it acting blindly until a stall guard
|
|
# terminates it; this hands it the reason once so it can finish failed instead.
|
|
verification_refusal_nudge: str | None = None
|
|
verification_refusal_nudged: bool = False
|
|
# The last fingerprint sample from the PREVIOUS batch: a delayed render can land between one
|
|
# batch's after-sample and the next batch's before-sample, so movement is checked across
|
|
# batches, not only within them.
|
|
page_state_prev_fp: str | None = None
|
|
refresh_cycles: int = 0
|
|
refresh_nudge_due: bool = False
|
|
reload_failed_nudge_due: bool = False
|
|
pending_screenshots: list[bytes] = field(default_factory=list)
|
|
# The action round of the latest positive page-change evidence (SKY-15264, SKY-15666); the
|
|
# budget-extension gate reads it, so what counts as evidence is load-bearing, not cosmetic.
|
|
last_change_evidence_step: int | None = None
|
|
# Budget caps. The four are re-derived and applied ATOMICALLY on an extension grant: a partial
|
|
# update converts a step-cap death into a token-cap death, so they move together or not at all.
|
|
max_turns: int
|
|
max_tool_calls: int
|
|
max_action_steps: int | None = None
|
|
max_tokens: int | None = None
|
|
turns: int = 0
|
|
no_tool_call_turns: int = 0
|
|
total_tool_calls: int = 0
|
|
tool_seconds: float = 0.0
|
|
total_tokens: int = 0
|
|
action_steps: int = 0
|
|
# Refused billable calls left uncharged since the last charged, non-refused one.
|
|
uncharged_refusals: int = 0
|
|
budget_extended_notice: str | None = None
|
|
# Progress-gated budget extension (SKY-15264, SKY-15666): how many extensions have been granted,
|
|
# and the cap they are all sized and bounded against — captured before the first grant so growth
|
|
# stays linear. The evidence input is last_change_evidence_step above.
|
|
budget_extensions_granted: int = 0
|
|
original_action_steps: int | None = None
|
|
token_clamp_reported: bool = False
|
|
# Final-turn grant (budget-exhaustion final turn): mirrors budget_extension_granted's shape. A
|
|
# budget-cap trip anywhere in the loop sets this and buys one more unconstrained model turn. One
|
|
# grant per latch, but an extension that relieves the latching guard releases it and re-arms the
|
|
# latch for a later trip; cap_trip_pending remembers which cap granted it, so a finish() outcome
|
|
# produced on that turn can carry the fact forward even though finish is a different code path.
|
|
# final_turn_started distinguishes "granted, turn not run yet" from "granted turn already ran": a
|
|
# mid-batch grant (max_tool_calls, the action-step gate) leaves the SAME counter tripped for the
|
|
# very next top-of-turn check, one code site removed from the grant with no turn boundary between
|
|
# them -- without this flag that check would end the run before the granted turn ever started.
|
|
final_turn_granted: bool = False
|
|
final_turn_started: bool = False
|
|
cap_trip_pending: str | None = None
|
|
# Extraction carried by a finish the granted turn staged but never got honored (skipped behind a
|
|
# failed or refused call, or refused by a fail-closed blocker). The verdict itself stays
|
|
# unhonored — its premise didn't hold — but the data rides the spent-grant exit.
|
|
final_turn_staged_output: Any = None
|
|
# We own the message array and assign it to the caller's message_history before
|
|
# each call, passing prompt=None: LLMCaller.use_message_history never appends the
|
|
# assistant reply or tool results itself, so multi-turn tool use must be threaded here.
|
|
messages: list[dict[str, Any]] = field(default_factory=list)
|
|
# Successful perception results by index into `messages`, recorded as they are appended, so
|
|
# compaction never infers "real snapshot" from content size and a skip/error result is never one.
|
|
reads: _PerceptionStore = field(default_factory=_PerceptionStore)
|
|
perception: _PerceptionLedger = field(default_factory=_PerceptionLedger)
|
|
# Net-progress ledger (additive shadow); None disables it, mirroring the guard's *_after knobs.
|
|
progress: _ProgressLedger | None = None
|
|
canonical: _CanonicalProgressTracker = field(default_factory=_CanonicalProgressTracker)
|
|
revisit_memory: _RevisitMemory = field(default_factory=_RevisitMemory)
|
|
# The action-loop counter: (repeat count, first turn of the streak) per billable action
|
|
# identity, cleared whenever evidence of page change arrives. action_warned holds the streaks
|
|
# whose warning was actually DELIVERED — termination is gated on it, so the model always gets
|
|
# the warning (and a chance to self-correct) at least one turn before the verdict.
|
|
# (repeats, first turn, page fingerprint before the first attempt, moved since then): the nudge
|
|
# may only claim the page is unchanged when no sample in the streak left that fingerprint.
|
|
action_counts: dict[tuple[str, str], tuple[int, int, str | None, bool]] = field(default_factory=dict)
|
|
action_warned: set[tuple[str, str]] = field(default_factory=set)
|
|
billable_actions: list[str] = field(default_factory=list)
|
|
# Credential re-submit guard (SKY-16594), keyed on the placeholder token rather than the field:
|
|
# the page re-renders between attempts, so an element-keyed rule is evaded by the very re-render
|
|
# that precedes the second submit. Tokens entered since the last submit, and how many times each
|
|
# has been submitted; entering one that has spent CREDENTIAL_SUBMIT_BUDGET is refused.
|
|
credentials_entered: set[str] = field(default_factory=set)
|
|
credential_submits: dict[str, int] = field(default_factory=dict)
|
|
# A single-action block's completion is offered through the finish tool at most once: a guard that
|
|
# holds a verdict only once would pass a second offer the model never saw it hold.
|
|
block_completion_offered: bool = False
|
|
|
|
|
|
async def run_agent_tool_loop(
|
|
*,
|
|
llm_caller: Any,
|
|
system_prompt: str,
|
|
user_prompt: str,
|
|
tools: list[ToolSpec],
|
|
max_turns: int,
|
|
max_tool_calls: int,
|
|
max_action_steps: int | None = None,
|
|
max_action_steps_ceiling: int | None = None,
|
|
prompt_name: str = "taskv3-agent-loop",
|
|
organization_id: str | None = None,
|
|
call_kwargs: dict[str, Any] | None = None,
|
|
should_cancel: Callable[[], Awaitable[bool]] | None = None,
|
|
on_action_round: Callable[[list[RoundAction], str | None], Awaitable[None]] | None = None,
|
|
on_pre_action: Callable[[str, dict[str, Any]], Awaitable[None]] | None = None,
|
|
max_tokens: int | None = None,
|
|
deadline_seconds: float | None = None,
|
|
retryable_call_exceptions: tuple[type[BaseException], ...] = (),
|
|
max_call_retries: int = 0,
|
|
call_retry_base_delay: float = 1.0,
|
|
# Makes this loop the owner of the LLM exhaustion receipt: calls are told not to emit one, and this
|
|
# fires once when the loop gives up, so a failure its own retry recovers is never counted.
|
|
on_llm_call_exhausted: Callable[[BaseException], None] | None = None,
|
|
stall_nudge_after: int | None = PERCEPTION_STALL_NUDGE_AFTER,
|
|
stall_terminate_after: int | None = PERCEPTION_STALL_TERMINATE_AFTER,
|
|
action_nudge_after: int | None = ACTION_LOOP_NUDGE_AFTER,
|
|
action_terminate_after: int | None = ACTION_LOOP_TERMINATE_AFTER,
|
|
progress_window: int | None = PROGRESS_LEDGER_WINDOW,
|
|
activity: ActivityRecency | None = None,
|
|
submit_watch: SubmitWatch | None = None,
|
|
telemetry_salt: str | None = None,
|
|
completion_probe: CompletionProbe | None = None,
|
|
verification_blocker: VerificationBlocker | None = None,
|
|
staged_downloads: set[str] | None = None,
|
|
initial_navigation_status: int | None = None,
|
|
# Only ever named in the dead-end verdict's text, so a caller that has the status but not the URL
|
|
# (or whose URL is unfit to print) still gets the same verdict, minus the place.
|
|
initial_navigation_url: str | None = None,
|
|
# Every URL the CALLER gave this run, normalized by `caller_known_published_urls`. A guard verdict
|
|
# publishes a landed URL's path only when it is one of these; every other path is elided to its
|
|
# host. Computed once by the caller (it owns the task config) and never read by a tool.
|
|
caller_known_urls: frozenset[str] = frozenset(),
|
|
# Resolves the run's drop-check secret values when a verdict is about to name a page-supplied
|
|
# element. Read at verdict time, not loop start: the registry grows as a run resolves credentials.
|
|
label_secret_values: Callable[[], Collection[str]] | None = None,
|
|
# Placeholder tokens minted for a login credential's username slot, held to
|
|
# LOGIN_IDENTIFIER_SUBMIT_BUDGET instead of CREDENTIAL_SUBMIT_BUDGET. Read per refusal.
|
|
login_identifier_tokens: Callable[[], Collection[str]] | None = None,
|
|
page_probe: Callable[[], Awaitable[str | None]] | None = None,
|
|
reload_page: Callable[[], Awaitable[None]] | None = None,
|
|
max_refresh_cycles: int = 3,
|
|
page_fingerprint: Callable[[], Awaitable[str | None]] | None = None,
|
|
# One model call's worth of token headroom, reserved so the granted final turn (below) is
|
|
# actually fundable rather than being immediately re-tripped by the same max_tokens check it
|
|
# was granted under. 0 (the default) reproduces today's unreserved check.
|
|
final_turn_token_reserve: int = 0,
|
|
# The caller's policy for sizing the runaway guards off an action-step budget — the same
|
|
# function that produced the max_turns/max_tool_calls/max_tokens above. Supplied so a granted
|
|
# budget extension can RE-DERIVE them from the extended cap instead of running on the leftover
|
|
# headroom of the pre-extension one; without it a bigger action budget merely converts a
|
|
# step-cap death into a token-cap death. None keeps the guards fixed for the whole run.
|
|
backstops_for_cap: Callable[[int], tuple[int, int, int]] | None = None,
|
|
semantic_commit_stats: SemanticCommitStats | None = None,
|
|
# Set for an extraction block: it reads, and may click to reveal what it reads, but it does not
|
|
# author input, so every FILL_TOOLS call is refused at dispatch.
|
|
refuse_input_entry: bool = False,
|
|
tool_trail: ToolTrail | None = None,
|
|
# A block whose contract is one action: once one succeeded, a follow-up the step cap refuses offers
|
|
# finish(completed) instead of failing the block, as the step engine completes it after that step.
|
|
single_action_block: bool = False,
|
|
) -> LoopOutcome:
|
|
tool_by_name = {tool.name: tool for tool in tools}
|
|
st = LoopState(
|
|
max_turns=max_turns,
|
|
max_tool_calls=max_tool_calls,
|
|
max_action_steps=max_action_steps,
|
|
max_tokens=max_tokens,
|
|
original_action_steps=max_action_steps,
|
|
messages=[
|
|
{"role": "system", "content": system_prompt},
|
|
{"role": "user", "content": user_prompt},
|
|
],
|
|
progress=_ProgressLedger(window=progress_window) if progress_window is not None else None,
|
|
)
|
|
# Per run, never logged: the hashes it keys are stable within this run (the only scope any guard
|
|
# decision spans) and uncorrelatable across runs, so page content and arguments cannot be
|
|
# fingerprinted across tenants from telemetry.
|
|
if telemetry_salt is None:
|
|
telemetry_salt = secrets.token_hex(16)
|
|
openai_tools = [tool.to_openai_tool() for tool in tools]
|
|
|
|
def _clear_action_state() -> None:
|
|
st.action_counts.clear()
|
|
st.action_warned.clear()
|
|
|
|
def _note_page_change_evidence() -> None:
|
|
st.last_change_evidence_step = st.action_steps
|
|
|
|
async def _consume_refresh_signal(
|
|
ctx: SkyvernContext, tool_name: str, remaining: list[Any], round_actions: list[Any], *, drop: bool
|
|
) -> bool:
|
|
"""Clear the page-refresh signal and, unless dropped or past the cap, reload and void `remaining`."""
|
|
ctx.refresh_working_page = False
|
|
st.refresh_cycles += 1
|
|
if drop:
|
|
LOG.info("taskv3 loop refresh signal dropped", tool=tool_name, turn=st.turns)
|
|
return False
|
|
if st.refresh_cycles > max_refresh_cycles:
|
|
# The queued calls were chosen on a page declared stale, so they are voided rather than
|
|
# run; a page that keeps demanding a reload cannot be stabilized, and the run ends there.
|
|
LOG.warning(
|
|
"taskv3 loop refresh signal past cap",
|
|
tool=tool_name,
|
|
turn=st.turns,
|
|
guard=PAGE_REFRESH_EXHAUSTED_GUARD,
|
|
refresh_cycles=st.refresh_cycles,
|
|
)
|
|
_append_skipped_tool_results(st.messages, remaining, "the page could not be stabilized")
|
|
st.outcome = _guard_verdict(
|
|
PAGE_REFRESH_EXHAUSTED_GUARD,
|
|
"The page kept having to be reloaded and never settled into a state the run could act "
|
|
"on, so the task could not continue there.",
|
|
)
|
|
return True
|
|
if reload_page is not None:
|
|
reload_record = ("reload_page", {"reason": "a page-level handler requested a refresh"})
|
|
try:
|
|
await reload_page()
|
|
except Exception:
|
|
# The page did not change, so nothing is re-baselined; the queued calls are still
|
|
# voided (they were chosen on a page declared stale), the signal is re-armed for
|
|
# another attempt (bounded by the cap), and the model is told the reload failed.
|
|
LOG.warning("taskv3 loop page reload failed after refresh signal", tool=tool_name, exc_info=True)
|
|
round_actions.append(RoundAction(*reload_record, False))
|
|
ctx.refresh_working_page = True
|
|
_append_skipped_tool_results(st.messages, remaining, "a page reload was requested but failed")
|
|
st.reload_failed_nudge_due = True
|
|
return True
|
|
round_actions.append(RoundAction(*reload_record, True))
|
|
LOG.info("taskv3 loop honored page refresh signal", tool=tool_name, turn=st.turns)
|
|
# The reloaded document is a new baseline for every ledger that described the old one, and a
|
|
# look taken before it would hand the model marks that no longer exist. That includes the
|
|
# budget-extension evidence stamp: pre-reload progress says nothing about the fresh document,
|
|
# so the run must re-demonstrate progress before it can earn an extension.
|
|
_clear_action_state()
|
|
st.last_change_evidence_step = None
|
|
st.trailing_page_state_stall_rounds = 0
|
|
st.page_state_nudge_delivered = False
|
|
st.page_state_prev_fp = None
|
|
st.perception.reset()
|
|
if activity is not None:
|
|
activity.perception_stall_imminent = False
|
|
st.pending_screenshots = []
|
|
st.pending_nav_dead_end = None
|
|
st.stall_nudges_due = []
|
|
if st.progress is not None:
|
|
st.progress.hard_progress()
|
|
st.canonical.progress(_ProgressEvidence.REFRESH_RELOAD)
|
|
if submit_watch is not None:
|
|
submit_watch.clear()
|
|
_append_skipped_tool_results(st.messages, remaining, "the page was refreshed")
|
|
st.refresh_nudge_due = True
|
|
return True
|
|
|
|
def _progress_observe_shadow(
|
|
observe_summary: dict[str, int], tool_name: str, attribution: dict[str, Any], baseline_before: int | None
|
|
) -> None:
|
|
"""Feeds a model-issued no-arg observe's invalid-fields count into the net-progress ledger."""
|
|
if st.progress is None or not observe_summary:
|
|
return
|
|
invalid_fields = observe_summary.get("invalid_fields")
|
|
# The ledger's True return is the SHADOW STALL verdict, not progress — the canonical clear
|
|
# keys on the ledger re-baselining (a new low, or a rise re-baseline: fresh context), which
|
|
# is visible as its baseline moving under an already-set baseline. The baseline is captured
|
|
# BEFORE _absorb_result_data runs: a replayed download notice hard-progresses the shadow
|
|
# ledger (nulling the baseline) on the same result whose summary carries the new low, and
|
|
# reading the post-absorb value would leave this clear dead for the rest of the page.
|
|
stalled = invalid_fields is not None and st.progress.observe(invalid_fields)
|
|
if baseline_before is not None and st.progress.invalid_baseline != baseline_before:
|
|
st.canonical.progress(_ProgressEvidence.INVALID_FIELDS_BASELINE_MOVE)
|
|
if stalled:
|
|
LOG.info(
|
|
PROGRESS_LEDGER_SHADOW_EVENT,
|
|
actions=st.progress.actions_since_progress,
|
|
invalid_fields=st.progress.last_invalid,
|
|
form_armed=st.progress.form_armed,
|
|
tool=tool_name,
|
|
turn=st.turns,
|
|
**attribution,
|
|
)
|
|
|
|
def _absorb_result_data(tool_name: str, spec: ToolSpec | None, result_data: dict[str, Any]) -> bool:
|
|
"""Absorbs a download/page-change signal in a tool's result.data (staged_downloads,
|
|
action-state clear, progress hard_progress). Returns whether marks were renumbered."""
|
|
if staged_downloads is not None and result_data.get("staged_download"):
|
|
staged_downloads.add(result_data["staged_download"])
|
|
if spec is not None and (result_data.get("download_notice") or result_data.get("page_state_changed")):
|
|
# A download landing or a navigation is progress no matter which tool witnessed it or
|
|
# whether that call itself errored: re-clicking the button that produces a file (a
|
|
# "download next" flow), or re-trying after navigating to a fresh page, is a healthy
|
|
# loop, not a stuck one. A same-URL reload is the exception: it resets the retry ledger
|
|
# like any reload but is a state WIPE, not progress — it clears the extension evidence
|
|
# exactly like the refresh-signal path.
|
|
_clear_action_state()
|
|
if result_data.get("same_url_reload") or result_data.get("nav_revisit"):
|
|
# A reload destroys the observed document and a revisit replaces it with a fresh
|
|
# instance of known territory: re-baseline the perception ledger (as the refresh
|
|
# path does) so the first post-navigation look cannot diff against a pre-navigation
|
|
# digest and read as progress — and clear the evidence stamp in both cases, since
|
|
# navigation is non-billable and a surviving stamp would stay maximally recent
|
|
# through any amount of oscillation.
|
|
st.perception.reset()
|
|
st.last_change_evidence_step = None
|
|
if activity is not None:
|
|
activity.perception_stall_imminent = False
|
|
elif result_data.get("download_new") or result_data.get("page_state_changed"):
|
|
# Only a download detected on THIS call is evidence — a compactable tool replaying
|
|
# a retained notice re-clears the retry ledger but earns no budget.
|
|
_note_page_change_evidence()
|
|
# The old document's perception streak cannot speak for the fresh page: clear the
|
|
# imminent flag exactly as the refresh path does.
|
|
if activity is not None:
|
|
activity.perception_stall_imminent = False
|
|
if st.progress is not None:
|
|
st.progress.hard_progress()
|
|
if (
|
|
result_data.get("download_new")
|
|
or result_data.get("page_state_changed")
|
|
or result_data.get("same_url_reload")
|
|
or result_data.get("nav_revisit")
|
|
):
|
|
# A REPLAYED notice (download_notice without download_new) re-clears the retry
|
|
# ledger but is not fresh progress: it must not keep wiping the canonical ring, or
|
|
# a post-download loop could never accumulate enough touches to emit telemetry.
|
|
st.canonical.progress(_ProgressEvidence.FRESH_DOWNLOAD_OR_NAVIGATION)
|
|
# A click that moved the URL is a real page transition (H1 hard progress) for the shadow
|
|
# ledger, but URL equality does NOT prove same-page (a URL-stable SPA form advance) — so only
|
|
# the positive direction is acted on, and kept OUT of the branch above so it never clears the
|
|
# action-loop guard's state; this signal is shadow-only and additive. It is a URL-only HINT
|
|
# (history.pushState moves the URL without changing the document), so it never stamps
|
|
# budget-extension evidence either — the content-confirmed signals are that bar.
|
|
if result_data.get("page_transitioned") is True:
|
|
if st.progress is not None:
|
|
st.progress.hard_progress()
|
|
st.canonical.progress(_ProgressEvidence.PAGE_TRANSITIONED)
|
|
return tool_name == "look" and bool(result_data.get("marks_renumbered"))
|
|
|
|
async def _completion_probe_outcome(
|
|
tool_name: str, spec: ToolSpec | None, result_data: dict[str, Any]
|
|
) -> LoopOutcome | None:
|
|
"""Consults the completion probe after a billable or download-signaling tool result."""
|
|
if not (
|
|
completion_probe is not None
|
|
and spec is not None
|
|
and (spec.billable or result_data.get("download_notice"))
|
|
# file_upload stages an http(s) source file into the same downloads dir; that landed
|
|
# file is not the run's OWN download unless the wrapper also flagged download_notice.
|
|
and not (result_data.get("staged_download") and not result_data.get("download_notice"))
|
|
):
|
|
return None
|
|
try:
|
|
completion_reason = await completion_probe(frozenset(staged_downloads or ()))
|
|
except Exception:
|
|
LOG.warning("taskv3 completion_probe failed; not treating it as complete", exc_info=True)
|
|
return None
|
|
if not completion_reason:
|
|
return None
|
|
# The probe is the second path to a completed outcome, and it never reaches the finish tool.
|
|
# Without this the verification gate would hold only one of the two, so a run whose code
|
|
# never arrived could still end `completed` on a file that happened to land.
|
|
if verification_blocker is not None:
|
|
try:
|
|
verification_message = await verification_blocker("completed")
|
|
except Exception:
|
|
# Fail closed, as the finish tool's completed side does: a broken gate must not let
|
|
# a blank verification step read as done.
|
|
LOG.warning("taskv3 completion probe verification gate failed; failing closed", exc_info=True)
|
|
return None
|
|
if verification_message:
|
|
LOG.info(
|
|
"taskv3 loop completion probe refused by the verification gate",
|
|
tool=tool_name,
|
|
turn=st.turns,
|
|
)
|
|
if not st.verification_refusal_nudged:
|
|
st.verification_refusal_nudge = verification_message
|
|
return None
|
|
LOG.info("taskv3 loop completion probe fired", tool=tool_name, turn=st.turns)
|
|
# No tracker-wide clear here: the probe firing is download progress for the COMPLETING
|
|
# touch only, and its call site drops that one target's pending rungs — a whole-generation
|
|
# bump would erase a sibling target's true-positive rung along with it.
|
|
return LoopOutcome("completed", completion_reason)
|
|
|
|
def _perception_stall_check(
|
|
ledger: _PerceptionLedger,
|
|
content_digest: str,
|
|
action_key: tuple[str, str],
|
|
tool_name: str,
|
|
attribution: dict[str, Any],
|
|
*,
|
|
content_only_digest: str | None = None,
|
|
refresh_pending: bool = False,
|
|
) -> tuple[LoopOutcome | None, list[tuple[str, int]]]:
|
|
"""Trips the nudge/terminate thresholds off ``ledger``'s own streak."""
|
|
stall_nudges: list[tuple[str, int]] = []
|
|
snap = ledger.record(action_key, content_digest)
|
|
if snap.progressed:
|
|
# This probe saw the page change since it last looked — fresh evidence of progress, so
|
|
# repeat counts for actions taken against the old state are stale. A first-time probe has
|
|
# no baseline and proves nothing, which is what keeps varied-selector probing from
|
|
# laundering repetition into progress. A multi-page wizard that clicks the same selector
|
|
# (e.g. "next") on every page relies on THIS clear to survive — page_transitioned alone
|
|
# deliberately does not clear the action-loop guard (see below), so only a progressed
|
|
# snapshot does.
|
|
ring = ledger.content_only.get(action_key) if content_only_digest is not None else None
|
|
# INVARIANT: this test and the budget-evidence test below must both refuse a return to
|
|
# the ring. They answer one question — did the run reach ground it has not already
|
|
# covered? — and relaxing either alone silently reopens SKY-14998: a page that CYCLES
|
|
# moves on every probe, so `progressed` holds every round, and an unconditional clear
|
|
# let the action DRIVING the oscillation reset its own counter forever (one production
|
|
# key ran 11 times against a threshold of 6 while the nudge fired once).
|
|
returned_to_known_ground = ring is not None and content_only_digest in ring
|
|
if not returned_to_known_ground:
|
|
_clear_action_state()
|
|
# Two landed digests that differ are positive evidence, exactly like a fingerprint
|
|
# mismatch — and the only movement evidence there is when page_fingerprint is absent.
|
|
st.canonical.progress(_ProgressEvidence.PERCEPTION_DIGEST)
|
|
# Evidence requires NEW content: a URL-only flip (history.pushState) still clears the
|
|
# repeat guards above but earns no budget, and neither does a return to a content state
|
|
# in the probe's recent ring (a panel toggling open and shut).
|
|
if ring and content_only_digest not in ring:
|
|
_note_page_change_evidence()
|
|
if content_only_digest is not None:
|
|
ring = ledger.content_only.get(action_key)
|
|
if ring is None:
|
|
ring = ledger.content_only.setdefault(action_key, deque(maxlen=PERCEPTION_RING))
|
|
ring.append(content_only_digest)
|
|
if activity is not None and stall_terminate_after is not None:
|
|
activity.perception_stall_imminent = ledger.next_snapshot_can_trip(stall_terminate_after)
|
|
# A refresh about to be honored re-baselines this ledger anyway, so a stall verdict raised on
|
|
# the stale page it is replacing would be wrong the instant the reload lands.
|
|
if stall_terminate_after is not None and snap.live >= stall_terminate_after and not refresh_pending:
|
|
LOG.info(
|
|
"taskv3 loop perception stalled",
|
|
tool=tool_name,
|
|
identical_count=snap.live,
|
|
turn=st.turns,
|
|
guard=PERCEPTION_STALL_GUARD,
|
|
**attribution,
|
|
)
|
|
return (
|
|
_guard_verdict(
|
|
PERCEPTION_STALL_GUARD,
|
|
"The page stopped changing in response to the run's actions — every fresh read of it "
|
|
"came back with the same content — so the task could not make progress there, "
|
|
"commonly because something on the page is in the way that the run cannot see or "
|
|
"operate, such as a prompt inside an embedded frame.",
|
|
),
|
|
stall_nudges,
|
|
)
|
|
if (
|
|
stall_terminate_after is not None
|
|
and snap.tool_identical == stall_terminate_after
|
|
and not ledger.suppressed_reported
|
|
):
|
|
ledger.suppressed_reported = True
|
|
LOG.info(
|
|
PERCEPTION_STALL_SUPPRESSED_EVENT,
|
|
tool=tool_name,
|
|
identical_count=snap.tool_identical,
|
|
turn=st.turns,
|
|
**attribution,
|
|
)
|
|
if (
|
|
stall_terminate_after is not None
|
|
and snap.probe_revisits >= stall_terminate_after
|
|
and not ledger.shadow_reported
|
|
):
|
|
ledger.shadow_reported = True
|
|
LOG.info(
|
|
PERCEPTION_STALL_SHADOW_EVENT,
|
|
snapshots=snap.probe_revisits,
|
|
tool=tool_name,
|
|
turn=st.turns,
|
|
**attribution,
|
|
)
|
|
if stall_nudge_after is not None and snap.tool_identical == stall_nudge_after:
|
|
# Warn off the per-tool counter, not ``live``: it moves by one per read, so it
|
|
# crosses the threshold exactly once per streak and before any live verdict.
|
|
stall_nudges.append((tool_name, snap.tool_identical))
|
|
return None, stall_nudges
|
|
|
|
# Mutable for the run: a provider that rejects tool_choice rejects it every turn, so a drop
|
|
# made once must stick.
|
|
active_call_kwargs = dict(call_kwargs or {})
|
|
if on_llm_call_exhausted is not None:
|
|
active_call_kwargs["caller_owns_exhaustion_receipt"] = True
|
|
|
|
def _degrade_tool_choice(exc: BaseException) -> bool:
|
|
"""Drop the optional call parameters (tool_choice, the reasoning-summary dict) and report
|
|
whether the turn is worth re-issuing.
|
|
|
|
Called only when the turn is otherwise about to end the run, so the cost is one extra call
|
|
on a run that was already failing. A context-window overflow is excluded because dropping a
|
|
parameter provably cannot fix it. Dropping reasoning_effort reverts to the config's own
|
|
value, so a provider that rejects the dict form cannot end the run on turn 1.
|
|
"""
|
|
if isinstance(exc, SkyvernContextWindowExceededError):
|
|
return False
|
|
# One parameter per degrade, least-proven first: a provider that rejects only the summary
|
|
# dict keeps its independently supported tool_choice; a second rejection drops that too.
|
|
for key in ("reasoning_effort", "tool_choice"):
|
|
if active_call_kwargs.pop(key, None) is not None:
|
|
LOG.warning(
|
|
"taskv3 loop retrying without optional call param", dropped=key, turn=st.turns, exc_info=True
|
|
)
|
|
return True
|
|
return False
|
|
|
|
# Images produced by an on-demand `look` this turn, to show the model on the NEXT call only. Passed
|
|
# as the transient screenshots= arg once, then cleared, so a look costs one image on one turn and
|
|
# never enters `messages` (the transcript re-seeds message_history each turn, so it's structurally
|
|
# gone the turn after).
|
|
started_at = time.monotonic()
|
|
deadline_at = started_at + deadline_seconds if deadline_seconds is not None else None
|
|
|
|
# The task's starting URL is navigated during browser setup, before this loop runs, so a dead/removed
|
|
# starting posting never routes through the in-loop `navigate` tool — the model just observes the dead
|
|
# page and finishes (defaulting to failed). Classify that pre-loop navigation here so the dominant
|
|
# dead-posting case ends `terminated`, matching v1, without waiting on the model's finish discretion.
|
|
# Cancellation is checked first, exactly as the first loop turn would: a run canceled during setup must
|
|
# persist as `canceled` (and stay unbilled), not be pre-empted into `terminated` by this fast path.
|
|
if st.outcome is None and initial_navigation_status in NAVIGATION_DEAD_END_STATUSES:
|
|
if should_cancel is not None and await should_cancel():
|
|
st.outcome = LoopOutcome("canceled", "run canceled")
|
|
else:
|
|
LOG.info(
|
|
"taskv3 loop initial navigation dead end",
|
|
http_status=initial_navigation_status,
|
|
guard=NAV_DEAD_END_GUARD,
|
|
)
|
|
st.outcome = _guard_verdict(
|
|
NAV_DEAD_END_GUARD,
|
|
_dead_end_reason(
|
|
initial_navigation_status,
|
|
initial_navigation_url,
|
|
page_noun="The task's starting page",
|
|
caller_known_urls=caller_known_urls,
|
|
),
|
|
)
|
|
|
|
while st.outcome is None:
|
|
if should_cancel is not None and await should_cancel():
|
|
st.outcome = LoopOutcome("canceled", "run canceled")
|
|
break
|
|
# The final-turn grant (see final_turn_granted above): a budget trip here buys one more
|
|
# turn instead of ending the run, so the checks below are elif'd
|
|
# (only the first tripped cap matters this iteration). A spent grant ends the run BEFORE any
|
|
# counter is consulted: the granted turn may have been granted at a gate no top-of-turn
|
|
# check re-reads (the action-step gate), so waiting for a counter to re-trip would let a
|
|
# finish-less granted turn keep looping on other budgets — and the cap that granted the
|
|
# turn, not whichever counter happens to re-trip first, is the honest fact to report.
|
|
# final_turn_started (set below, right before the turn actually runs) distinguishes a spent
|
|
# grant from a mid-batch grant whose turn has not run yet. Cancellation above is exempt: it
|
|
# is not a budget cap.
|
|
if st.final_turn_granted and st.final_turn_started:
|
|
st.outcome = LoopOutcome(
|
|
"budget_exhausted",
|
|
_budget_exhausted_reason(st.cap_trip_pending or "budget"),
|
|
cap_trip=st.cap_trip_pending,
|
|
extracted_output=st.final_turn_staged_output,
|
|
)
|
|
break
|
|
# Once the grant fired no top-of-turn cap can trip again, so the checks only run pre-grant.
|
|
top_of_turn_trip: str | None = None
|
|
if not st.final_turn_granted:
|
|
if deadline_seconds is not None and time.monotonic() - started_at > deadline_seconds:
|
|
top_of_turn_trip = f"deadline ({deadline_seconds:.0f}s) reached"
|
|
elif st.max_tokens is not None and st.total_tokens >= max(0, st.max_tokens - final_turn_token_reserve):
|
|
top_of_turn_trip = f"max_tokens ({st.max_tokens}) reached"
|
|
elif st.turns >= st.max_turns:
|
|
top_of_turn_trip = f"max_turns ({st.max_turns}) reached"
|
|
elif st.total_tool_calls >= st.max_tool_calls:
|
|
top_of_turn_trip = f"max_tool_calls ({st.max_tool_calls}) reached"
|
|
if top_of_turn_trip is not None:
|
|
st.final_turn_granted = True
|
|
st.cap_trip_pending = top_of_turn_trip
|
|
LOG.info(
|
|
FINAL_TURN_GRANTED_EVENT,
|
|
cap=top_of_turn_trip,
|
|
turn=st.turns,
|
|
tool_calls_remaining=None if activity is None else activity.tool_calls_remaining,
|
|
tokens_remaining=None if activity is None else activity.tokens_remaining,
|
|
)
|
|
st.messages.append({"role": "user", "content": _budget_exhausted_observation(top_of_turn_trip, activity)})
|
|
if st.final_turn_granted:
|
|
st.final_turn_started = True
|
|
if activity is not None:
|
|
# Read by the finish tool's hold gates: a hold's retry turn no longer exists.
|
|
activity.final_turn_active = True
|
|
st.turns += 1
|
|
if activity is not None:
|
|
activity.turn = st.turns
|
|
activity.turns_remaining = st.max_turns - st.turns
|
|
activity.tool_calls_remaining = st.max_tool_calls - st.total_tool_calls
|
|
|
|
# Elide superseded perception results before re-sending the transcript, so a perception-heavy
|
|
# run can't balloon the context to the token backstop (the pre-compaction runaway mode).
|
|
st.reads.compact(st.messages)
|
|
llm_caller.message_history = list(st.messages)
|
|
# Consume any pending look image into THIS call only, then clear: the image rides one request
|
|
# and is never appended to `messages`, so the turn after carries zero image blocks.
|
|
screenshots_for_call = st.pending_screenshots or None
|
|
st.pending_screenshots = []
|
|
# Retry only the LLM call on transient provider errors. No browser tool has run this
|
|
# turn, so re-issuing the same call is side-effect-free — unlike a whole-task retry,
|
|
# which would re-execute prior clicks/types. This restores the step engine's transient
|
|
# resilience, which v3 otherwise loses by running as one non-retried unit.
|
|
response = None
|
|
call_attempt = 0
|
|
while True:
|
|
try:
|
|
response = await llm_caller.call(
|
|
prompt=None,
|
|
prompt_name=prompt_name,
|
|
organization_id=organization_id,
|
|
tools=openai_tools,
|
|
use_message_history=True,
|
|
raw_response=True,
|
|
screenshots=screenshots_for_call,
|
|
**active_call_kwargs,
|
|
)
|
|
break
|
|
except retryable_call_exceptions as exc:
|
|
call_attempt += 1
|
|
if call_attempt > max_call_retries:
|
|
# A provider rejecting the parameter surfaces here, not in the generic handler
|
|
# below: litellm's 400s subclass openai.APIError, which the LLM layer maps to
|
|
# the retryable type. Degrading only after the transient budget is spent keeps
|
|
# a passing blip from disabling the lever for the rest of the run.
|
|
if _degrade_tool_choice(exc):
|
|
# Spend the transient budget once, not once per parameter set: the degraded
|
|
# turn gets a single shot, which is what "last resort" is worth.
|
|
call_attempt = max_call_retries
|
|
continue
|
|
LOG.warning(
|
|
"taskv3 loop LLM call failed after retries", turn=st.turns, attempts=call_attempt, exc_info=True
|
|
)
|
|
if on_llm_call_exhausted is not None:
|
|
on_llm_call_exhausted(exc)
|
|
st.outcome = LoopOutcome("loop_error", f"llm_call_failed: {type(exc).__name__}: {exc}")
|
|
break
|
|
LOG.info("taskv3 loop retrying transient LLM error", turn=st.turns, attempt=call_attempt)
|
|
await asyncio.sleep(call_retry_base_delay * (2 ** (call_attempt - 1)))
|
|
except Exception as exc:
|
|
if _degrade_tool_choice(exc):
|
|
continue
|
|
LOG.warning("taskv3 loop LLM call failed", turn=st.turns, exc_info=True)
|
|
if on_llm_call_exhausted is not None:
|
|
on_llm_call_exhausted(exc)
|
|
st.outcome = LoopOutcome("loop_error", f"llm_call_failed: {type(exc).__name__}: {exc}")
|
|
break
|
|
if st.outcome is not None:
|
|
break
|
|
|
|
usage = _get(response, "usage") or {}
|
|
turn_tokens = _get(usage, "total_tokens")
|
|
if not turn_tokens:
|
|
turn_tokens = (_get(usage, "prompt_tokens") or 0) + (_get(usage, "completion_tokens") or 0)
|
|
st.total_tokens += int(turn_tokens or 0)
|
|
if activity is not None:
|
|
activity.last_turn_tokens = int(turn_tokens or 0)
|
|
activity.tokens_remaining = None if st.max_tokens is None else st.max_tokens - st.total_tokens
|
|
|
|
text = _extract_text(response)
|
|
reasoning_summary = _extract_reasoning_summary(response)
|
|
tool_calls = _extract_tool_calls(response)
|
|
|
|
assistant_message: dict[str, Any] = {"role": "assistant", "content": text or None}
|
|
if tool_calls:
|
|
assistant_message["tool_calls"] = [
|
|
{"id": tool_call_id, "type": "function", "function": {"name": name, "arguments": json.dumps(args)}}
|
|
for tool_call_id, name, args in tool_calls
|
|
]
|
|
st.messages.append(assistant_message)
|
|
|
|
if not tool_calls:
|
|
st.no_tool_call_turns += 1
|
|
LOG.info("taskv3 loop turn produced no tool call", turn=st.turns)
|
|
st.messages.append({"role": "user", "content": NO_TOOL_CALL_NUDGE})
|
|
continue
|
|
|
|
# Dispatched feeds the page-state stall detector; charged feeds the action-step budget and the
|
|
# rows' billable flag. They differ only by calls the tool refused before acting on the page.
|
|
turn_dispatched_billable = False
|
|
turn_charged = False
|
|
st.stall_nudges_due = []
|
|
st.refresh_nudge_due = False
|
|
st.budget_extended_notice = None
|
|
st.reload_failed_nudge_due = False
|
|
action_nudges_due: list[tuple[str, dict[str, Any], int]] = []
|
|
round_actions: list[RoundAction] = []
|
|
# A hard 404/410 from an in-loop navigate, applied only AFTER the batch so a same-turn fallback
|
|
# navigate can clear it — the model is told to batch aggressively, and terminating on the first
|
|
# of a batched [navigate(dead), navigate(live)] would discard the recovery it planned.
|
|
st.pending_nav_dead_end = None
|
|
# A same-selector dependent of a failed page-action call is skipped; any later click, Enter-shaped
|
|
# submit, or finish in the batch is skipped too -- the loop cannot tell a submit from the first two,
|
|
# and a verdict written before the failure was seen may be wrong or mis-reasoned.
|
|
failed_selectors: set[str] = set()
|
|
batch_had_failure = False
|
|
# A refusal leaves its value in the field, so no click after it -- toggle or not -- may run.
|
|
batch_had_refusal = False
|
|
marks_stale = False
|
|
# Loop events minted this batch, emitted only after every progress signal the batch can
|
|
# produce has been absorbed (see the end-of-batch emission below).
|
|
pending_canonical_fires: list[dict[str, Any]] = []
|
|
batch_page_change_reason: str | None = None
|
|
batch_fp_before: str | None = None
|
|
# Sample only when the batch can actually land a billable action -- the end-of-batch check
|
|
# below gates on turn_dispatched_billable, so a finish-only or perception-only batch has no use for
|
|
# this baseline and shouldn't pay its round-trip.
|
|
batch_has_billable_call = any(
|
|
tool_by_name.get(tool_name) is not None and tool_by_name[tool_name].billable
|
|
for _, tool_name, _ in tool_calls
|
|
)
|
|
# A cancellation that already landed makes this batch's baseline dead work: the per-call
|
|
# check below (before the first dispatch) ends the batch before anything it'd inform runs.
|
|
batch_will_sample_baseline = batch_has_billable_call and page_fingerprint is not None
|
|
batch_cancelled = batch_will_sample_baseline and should_cancel is not None and await should_cancel()
|
|
# The page-state stall detector (SKY-15265) reads the before/after fingerprint pair for
|
|
# every billable batch.
|
|
if page_fingerprint is not None and batch_has_billable_call and not batch_cancelled:
|
|
batch_fp_before = await _sample_probe(page_fingerprint, deadline_at=deadline_at)
|
|
if (
|
|
st.page_state_prev_fp is not None
|
|
and batch_fp_before is not None
|
|
and batch_fp_before != st.page_state_prev_fp
|
|
):
|
|
# The page moved BETWEEN batches (a delayed render landing after the prior
|
|
# after-sample): the touches the old samples described are stale, and this batch's
|
|
# dispatch and extension decisions must not read them. Canonical-only — the
|
|
# incumbent stall counters keep their end-of-batch turn_dispatched_billable gate.
|
|
st.canonical.progress(_ProgressEvidence.CROSS_BATCH_MOVEMENT)
|
|
batch_fp_after: str | None = None
|
|
if activity is not None:
|
|
# Scoped to THIS batch, and cleared here rather than only on the consume path: several
|
|
# branches between the hold and that path can `break` out of the batch first (the
|
|
# refresh signal is the reachable one -- a held finish makes `verdict_stands` false, so a
|
|
# pending signal takes that exit). A flag surviving into the next batch would drop
|
|
# everything queued behind its first tool, citing a hold that did not happen there.
|
|
activity.held_verdict_batch_skip = False
|
|
for idx, (tool_call_id, tool_name, args) in enumerate(tool_calls):
|
|
# Enforce the cap per tool call so one batched turn cannot overrun it, and honor a
|
|
# cancellation that arrives mid-batch before the next click/type/submit runs. Neither
|
|
# this call nor the rest of the batch executes, so answer them as skipped.
|
|
# Gated on `not final_turn_granted`: once the final turn is granted this same counter is
|
|
# already at (or past) the cap by construction, so re-enforcing it here would block the
|
|
# granted turn's very first call (including finish) before it ever ran. The top-of-turn
|
|
# check is what actually ends the run if this granted turn doesn't finish either.
|
|
if not st.final_turn_granted and st.total_tool_calls >= st.max_tool_calls:
|
|
mid_batch_trip = f"max_tool_calls ({st.max_tool_calls}) reached"
|
|
st.final_turn_granted = True
|
|
st.cap_trip_pending = mid_batch_trip
|
|
LOG.info(
|
|
FINAL_TURN_GRANTED_EVENT,
|
|
cap=mid_batch_trip,
|
|
turn=st.turns,
|
|
tool_calls_remaining=None if activity is None else activity.tool_calls_remaining,
|
|
tokens_remaining=None if activity is None else activity.tokens_remaining,
|
|
)
|
|
_append_skipped_tool_results(st.messages, tool_calls[idx:], "tool-call budget reached")
|
|
st.messages.append({"role": "user", "content": _budget_exhausted_observation(mid_batch_trip, activity)})
|
|
break
|
|
if should_cancel is not None and await should_cancel():
|
|
st.outcome = LoopOutcome("canceled", "run canceled")
|
|
_append_skipped_tool_results(st.messages, tool_calls[idx:], "run canceled")
|
|
break
|
|
spec = tool_by_name.get(tool_name)
|
|
call_selector = _call_selector(args)
|
|
if marks_stale and call_selector is not None and call_selector.startswith("mark="):
|
|
if activity is not None:
|
|
# Same predicate as the extraction and credential refusals below: the model
|
|
# emitted a well-formed action call and the harness declined to dispatch it over
|
|
# its OWN bookkeeping.
|
|
activity.action_attempts += 1
|
|
st.messages.append(
|
|
{
|
|
"role": "tool",
|
|
"tool_call_id": tool_call_id,
|
|
"name": tool_name,
|
|
"content": (
|
|
"skipped: an earlier look in this batch renumbered the marks, so this mark was chosen "
|
|
"from an old screenshot; pick it again from the new one"
|
|
),
|
|
}
|
|
)
|
|
continue
|
|
if call_selector is not None and call_selector in failed_selectors:
|
|
st.messages.append(
|
|
{
|
|
"role": "tool",
|
|
"tool_call_id": tool_call_id,
|
|
"name": tool_name,
|
|
"content": f"skipped: depends on an earlier call in this batch on {call_selector} that failed",
|
|
}
|
|
)
|
|
continue
|
|
if batch_had_failure and _is_finish(tool_name):
|
|
# Any verdict queued behind the failure was written before the model saw it: a completed
|
|
# one may be false, and a failed/terminated one carries a reason that predates the error.
|
|
st.messages.append(
|
|
{
|
|
"role": "tool",
|
|
"tool_call_id": tool_call_id,
|
|
"name": tool_name,
|
|
"content": (
|
|
"skipped: a field in this batch failed before this verdict was reached; "
|
|
"re-observe, then finish with a status that reflects the failure"
|
|
),
|
|
}
|
|
)
|
|
continue
|
|
if refuse_input_entry and tool_name in FILL_TOOLS:
|
|
LOG.info(EXTRACTION_ENTRY_REFUSED_EVENT, tool=tool_name, turn=st.turns)
|
|
if activity is not None:
|
|
# The model attempted; the harness declined. A refusal must not read as a run
|
|
# that never tried.
|
|
activity.action_attempts += 1
|
|
# A refused call did not do what the rest of the batch was planned around, so it marks the
|
|
# batch failed: a later click, Enter-shaped submit, or finish in the same batch is skipped.
|
|
batch_had_failure = True
|
|
batch_had_refusal = True
|
|
st.messages.append(
|
|
{
|
|
"role": "tool",
|
|
"tool_call_id": tool_call_id,
|
|
"name": tool_name,
|
|
"content": (
|
|
"refused: an extraction block does not type, select, or upload input. Extract what the "
|
|
"page shows now (clicking to reveal content is allowed), or finish with a status "
|
|
"that reflects it"
|
|
),
|
|
}
|
|
)
|
|
continue
|
|
spent_credentials = (
|
|
{
|
|
token
|
|
for token in _credential_placeholders(args)
|
|
if st.credential_submits.get(token, 0) >= CREDENTIAL_SUBMIT_BUDGET
|
|
}
|
|
if tool_name in CREDENTIAL_ENTRY_TOOLS
|
|
else set()
|
|
)
|
|
if spent_credentials and login_identifier_tokens is not None:
|
|
try:
|
|
identifiers = set(login_identifier_tokens())
|
|
except Exception:
|
|
LOG.warning("taskv3 loop could not resolve the run's login identifier tokens", tool=tool_name)
|
|
identifiers = set()
|
|
spent_credentials = {
|
|
token
|
|
for token in spent_credentials
|
|
if token not in identifiers or st.credential_submits.get(token, 0) >= LOGIN_IDENTIFIER_SUBMIT_BUDGET
|
|
}
|
|
if spent_credentials:
|
|
LOG.info(CREDENTIAL_RESUBMIT_REFUSED_EVENT, tool=tool_name, turn=st.turns)
|
|
if activity is not None:
|
|
activity.action_attempts += 1
|
|
# Same reason the extraction refusal marks the batch: a click queued behind this call
|
|
# would submit the value still sitting in the field, which is the act being refused.
|
|
batch_had_failure = True
|
|
batch_had_refusal = True
|
|
st.messages.append(
|
|
{
|
|
"role": "tool",
|
|
"tool_call_id": tool_call_id,
|
|
"name": tool_name,
|
|
"content": (
|
|
"refused: this secret value has already been entered and submitted in this task. The "
|
|
"page asking for it again means it was not accepted, and every further attempt spends a "
|
|
"real allowance on the account behind it — a sign-in lockout budget, a card's retry "
|
|
"limit — so it will not be entered again: retyping it, rewording it or putting it in a "
|
|
"different field will all be refused the same way. Finish with a status that reflects "
|
|
"the rejected value, or continue with whatever does not need it"
|
|
),
|
|
}
|
|
)
|
|
continue
|
|
toggle_exempted = False
|
|
if batch_had_failure and not batch_had_refusal and _may_submit(tool_name, args):
|
|
toggle_exempted = await _is_toggle_click(tool_name, spec, args)
|
|
if batch_had_failure and _may_submit(tool_name, args) and not toggle_exempted:
|
|
st.messages.append(
|
|
{
|
|
"role": "tool",
|
|
"tool_call_id": tool_call_id,
|
|
"name": tool_name,
|
|
"content": (
|
|
"skipped: a field in this batch failed and this call may submit the form; "
|
|
"fix the field, then re-queue it"
|
|
),
|
|
}
|
|
)
|
|
continue
|
|
# Once the action-step budget is spent, refuse a further page action — terminate, mirroring
|
|
# the step engine's max-steps stop — but let perception/finish through, since the cap bounds
|
|
# new action rounds, not the separate re-observe/finish turn the system prompt asks for.
|
|
# A run with recent positive page-change evidence earns an extension of half the original
|
|
# cap first (SKY-15264): the observed exhaustion population splits into genuinely long
|
|
# multi-page forms dying mid-progress (the extension's target) and stalled runs the gate
|
|
# refuses so they fail exactly as before. The grant repeats under that same predicate
|
|
# while the evidence keeps arriving (SKY-15666) — a form that was long enough to need one
|
|
# extension is routinely long enough to need another — up to a hard multiple of the
|
|
# original cap.
|
|
if (
|
|
spec is not None
|
|
and spec.billable
|
|
and st.max_action_steps is not None
|
|
and st.action_steps >= st.max_action_steps
|
|
):
|
|
base_cap = st.original_action_steps if st.original_action_steps else st.max_action_steps
|
|
raw_extension = base_cap // 2
|
|
extension_limit = base_cap * ACTION_BUDGET_EXTENSION_MAX_FACTOR
|
|
extension = min(raw_extension, max(0, extension_limit - st.max_action_steps))
|
|
if max_action_steps_ceiling is not None:
|
|
# The org's workflow-run-wide step pool is a HARD ceiling the extension must
|
|
# never breach: truncate the grant to what the pool can fund.
|
|
extension = min(extension, max_action_steps_ceiling - st.max_action_steps)
|
|
# Re-derive the runaway guards from the cap this grant would produce, and never
|
|
# below their live values. This is the whole difference between converting a
|
|
# step-cap death into a completion and merely relocating it onto the token guard:
|
|
# the guards were sized as functions of the action-step budget, so moving that
|
|
# budget without moving them leaves the extension running on leftovers.
|
|
grant_max_turns, grant_max_tool_calls, grant_max_tokens = st.max_turns, st.max_tool_calls, st.max_tokens
|
|
headroom_gain = (0, 0, 0)
|
|
token_clamped = False
|
|
if backstops_for_cap is not None and extension > 0:
|
|
next_turns, next_tool_calls, next_tokens = backstops_for_cap(st.max_action_steps + extension)
|
|
grant_max_turns = max(st.max_turns, next_turns)
|
|
grant_max_tool_calls = max(st.max_tool_calls, next_tool_calls)
|
|
token_gain = 0
|
|
if st.max_tokens is not None:
|
|
grant_max_tokens = max(st.max_tokens, next_tokens)
|
|
token_gain = grant_max_tokens - st.max_tokens
|
|
headroom_gain = (
|
|
grant_max_turns - st.max_turns,
|
|
grant_max_tool_calls - st.max_tool_calls,
|
|
token_gain,
|
|
)
|
|
# Turns still scale with the cap but tokens do not: the token guard is at its
|
|
# ceiling, and a cap raised past that point silently stops buying tokens.
|
|
# Reported from the grant branch below, never here — a refused extension leaves
|
|
# the cap where it was, so reporting on it would name a cap never in effect and
|
|
# would spend the once-per-run latch on a non-event.
|
|
token_clamped = st.max_tokens is not None and token_gain == 0 and grant_max_turns > st.max_turns
|
|
refresh_ctx = skyvern_context.current()
|
|
if st.max_action_steps >= extension_limit:
|
|
allowed, gate_reason = False, "extension_limit_reached"
|
|
elif raw_extension <= 0:
|
|
allowed, gate_reason = False, "cap_too_small"
|
|
elif extension <= 0:
|
|
allowed, gate_reason = False, "hard_step_ceiling"
|
|
elif refresh_ctx is not None and refresh_ctx.refresh_working_page:
|
|
# A pending refresh voids this very action and re-baselines the page: the grant
|
|
# must not race it and spend the extension on pre-reload evidence.
|
|
allowed, gate_reason = False, "refresh_pending"
|
|
else:
|
|
allowed, gate_reason = _budget_extension_gate(
|
|
st.action_steps,
|
|
st.last_change_evidence_step,
|
|
st.action_warned,
|
|
# CURRENT confirmed stalled-ness by the ledger's own rules: form_armed
|
|
# (the latest look showed a form) plus a window of fruitless actions. Not
|
|
# the one-shot telemetry latch, and never a bare counter — a form-less page
|
|
# increments the counter but must not be judged stuck by it.
|
|
st.progress is not None
|
|
and st.progress.form_armed
|
|
and st.progress.actions_since_progress >= st.progress.window,
|
|
activity,
|
|
deadline_at,
|
|
extension,
|
|
seconds_per_step=(time.monotonic() - started_at) / max(st.action_steps, 1),
|
|
headroom_gain=headroom_gain,
|
|
)
|
|
# This read cannot see unabsorbed in-batch movement: action_steps is frozen during
|
|
# a batch, so the cap check trips on the batch's FIRST billable call — before any
|
|
# page action dispatches — and between-batch movement was absorbed when
|
|
# batch_fp_before was sampled.
|
|
looping_now = st.canonical.looping_targets()
|
|
LOG.info(
|
|
CANONICAL_EXTEND_DELTA_EVENT,
|
|
current_decision=allowed,
|
|
gate_reason=gate_reason,
|
|
canonical_looping_targets=looping_now,
|
|
would_block_extension=bool(allowed and looping_now),
|
|
)
|
|
if allowed:
|
|
original_cap = st.max_action_steps
|
|
st.budget_extensions_granted += 1
|
|
st.max_action_steps += extension
|
|
st.max_turns, st.max_tool_calls, st.max_tokens = (
|
|
grant_max_turns,
|
|
grant_max_tool_calls,
|
|
grant_max_tokens,
|
|
)
|
|
if activity is not None:
|
|
# The guards moved mid-turn; the failure-evidence gates read these same
|
|
# fields later in this turn and would otherwise judge the run on the
|
|
# headroom it had before the grant.
|
|
activity.turns_remaining = st.max_turns - st.turns
|
|
activity.tool_calls_remaining = st.max_tool_calls - st.total_tool_calls
|
|
activity.tokens_remaining = None if st.max_tokens is None else st.max_tokens - st.total_tokens
|
|
if token_clamped and not st.token_clamp_reported:
|
|
st.token_clamp_reported = True
|
|
LOG.info(
|
|
"task_v3 token backstop clamped at its ceiling",
|
|
log_code="taskv3_token_backstop_clamped",
|
|
step_cap=st.max_action_steps,
|
|
max_tokens=st.max_tokens,
|
|
extensions_granted=st.budget_extensions_granted,
|
|
)
|
|
if st.final_turn_granted and _cap_trip_relieved(
|
|
st.cap_trip_pending,
|
|
st.max_tokens,
|
|
st.total_tokens,
|
|
final_turn_token_reserve,
|
|
st.max_turns,
|
|
st.turns,
|
|
st.max_tool_calls,
|
|
st.total_tool_calls,
|
|
):
|
|
# This run was granted its single wrap-up turn by a guard the grant above just
|
|
# raised past the trip. Before SKY-15666 no cap could be relieved mid-run, so
|
|
# the latch was permanent and reporting the remembered cap was honest; now it
|
|
# would end the run naming a budget it is nowhere near, with the extension it
|
|
# just earned unspent. Release it and let the top-of-turn checks resume.
|
|
LOG.info(
|
|
FINAL_TURN_RELEASED_EVENT,
|
|
cap=st.cap_trip_pending,
|
|
turn=st.turns,
|
|
extensions_granted=st.budget_extensions_granted,
|
|
)
|
|
released_cap = st.cap_trip_pending
|
|
st.final_turn_granted = False
|
|
st.final_turn_started = False
|
|
st.cap_trip_pending = None
|
|
# Staged under a latch that no longer holds: leaving it would stamp a stale
|
|
# extraction onto a terminal produced by some LATER cap trip.
|
|
st.final_turn_staged_output = None
|
|
if activity is not None:
|
|
activity.final_turn_active = False
|
|
# Staged, not appended: a user message may not sit between an assistant
|
|
# turn's tool calls and their results, and this fires mid-batch. It rides
|
|
# out with the other end-of-batch notes.
|
|
st.budget_extended_notice = _budget_extended_observation(released_cap or "budget", activity)
|
|
LOG.info(
|
|
ACTION_BUDGET_EXTENDED_EVENT,
|
|
original_cap=original_cap,
|
|
extension=extension,
|
|
extensions_granted=st.budget_extensions_granted,
|
|
base_cap=base_cap,
|
|
max_turns=st.max_turns,
|
|
max_tool_calls=st.max_tool_calls,
|
|
max_tokens=st.max_tokens,
|
|
action_steps=st.action_steps,
|
|
turn=st.turns,
|
|
tool=tool_name,
|
|
)
|
|
else:
|
|
LOG.info(
|
|
ACTION_BUDGET_EXTENSION_REFUSED_EVENT,
|
|
gate_reason=gate_reason,
|
|
max_action_steps=st.max_action_steps,
|
|
extensions_granted=st.budget_extensions_granted,
|
|
base_cap=base_cap,
|
|
action_steps=st.action_steps,
|
|
turn=st.turns,
|
|
tool=tool_name,
|
|
)
|
|
step_cap_trip = f"Reached the maximum steps ({st.max_action_steps})"
|
|
_append_skipped_tool_results(st.messages, tool_calls[idx:], "action-step budget reached")
|
|
finish_spec = tool_by_name.get("finish")
|
|
block_refusal: str | None = None
|
|
if (
|
|
single_action_block
|
|
and st.billable_actions
|
|
and finish_spec is not None
|
|
and not st.block_completion_offered
|
|
):
|
|
st.block_completion_offered = True
|
|
block_reason = (
|
|
f"performed the block's action ({st.billable_actions[0]}); "
|
|
"a further action was past the block's step limit"
|
|
)
|
|
# Through the real handler, so every guard on a completed verdict still applies.
|
|
try:
|
|
block_finish = await finish_spec.handler({"status": "completed", "reason": block_reason})
|
|
except Exception:
|
|
LOG.warning("taskv3 block completion finish raised", exc_info=True)
|
|
block_finish = ToolResult.error("")
|
|
if activity is not None:
|
|
activity.held_verdict_batch_skip = False
|
|
if block_finish.status == "ok" and (block_finish.data or {}).get("status") == "completed":
|
|
st.outcome = LoopOutcome("completed", block_reason)
|
|
break
|
|
if block_finish.status == "error":
|
|
block_refusal = block_finish.content
|
|
# Unlike the mid-batch max_tool_calls check above, the step gate is NOT special-cased
|
|
# away once the final turn is granted: a billable dispatch on the granted turn still
|
|
# hits it, which is the honest exit the grant exists to produce.
|
|
if st.final_turn_granted:
|
|
# A finish staged later in this skipped batch: its VERDICT is not honored
|
|
# (it presumed the refused action would run — accepting a completed there
|
|
# could claim a success that never happened), but the extraction it was
|
|
# already carrying is salvaged onto the honest exit.
|
|
staged_finish_output = next(
|
|
(
|
|
t_args.get("extracted_output")
|
|
for _t_id, t_name, t_args in tool_calls[idx:]
|
|
if t_name == "finish" and isinstance(t_args, dict)
|
|
),
|
|
None,
|
|
)
|
|
# The cap that granted the final turn is the honest fact to report — the
|
|
# step gate merely happens to be where the spent grant gets caught.
|
|
st.outcome = LoopOutcome(
|
|
"budget_exhausted",
|
|
_budget_exhausted_reason(st.cap_trip_pending or step_cap_trip),
|
|
cap_trip=st.cap_trip_pending or step_cap_trip,
|
|
extracted_output=staged_finish_output,
|
|
)
|
|
else:
|
|
st.final_turn_granted = True
|
|
st.cap_trip_pending = step_cap_trip
|
|
LOG.info(
|
|
FINAL_TURN_GRANTED_EVENT,
|
|
cap=step_cap_trip,
|
|
turn=st.turns,
|
|
tool_calls_remaining=None if activity is None else activity.tool_calls_remaining,
|
|
tokens_remaining=None if activity is None else activity.tokens_remaining,
|
|
)
|
|
observation = _budget_exhausted_observation(step_cap_trip, activity)
|
|
if block_refusal:
|
|
observation += f"\n\nThe block's completion was not accepted: {block_refusal}"
|
|
st.messages.append({"role": "user", "content": observation})
|
|
break
|
|
# Submit-shaped actions (the failure-evidence predicate, minus captcha) are reported BEFORE
|
|
# dispatch, since after it the page may be the confirmation page. A failure here never fails
|
|
# the action, and the time is not billed to the tool.
|
|
# Consumed before the side-effecting pre-submit capture, so a call chosen on a page that is
|
|
# gone leaves no artifacts; the recheck right before the handler covers the awaits below.
|
|
hook_ctx = skyvern_context.current()
|
|
if hook_ctx is not None and hook_ctx.refresh_working_page:
|
|
if await _consume_refresh_signal(
|
|
hook_ctx, tool_name, tool_calls[idx:], round_actions, drop=reload_page is None
|
|
):
|
|
break
|
|
if (
|
|
spec is not None
|
|
and on_pre_action is not None
|
|
and tool_name != "solve_captcha"
|
|
and _arms_failure_evidence(tool_name, args, True)
|
|
):
|
|
try:
|
|
await on_pre_action(tool_name, args)
|
|
except Exception:
|
|
LOG.warning("taskv3 on_pre_action callback failed", tool=tool_name, exc_info=True)
|
|
# Sampled before dispatch so an error below can be checked for having moved the page even
|
|
# when the tool set no flag (a raised exception carries none). Not billed to the tool's timing.
|
|
probe_before: str | None = None
|
|
if spec is not None and page_probe is not None:
|
|
probe_before = await _sample_probe(page_probe, deadline_at=deadline_at)
|
|
# A signal raised between calls (a route handler finishing during the model's turn, or
|
|
# during the pre-dispatch probes above) is honored right before this call runs, as the
|
|
# legacy per-action check does; the model chose it on a page that is gone.
|
|
pre_ctx = skyvern_context.current()
|
|
if pre_ctx is not None and pre_ctx.refresh_working_page:
|
|
if await _consume_refresh_signal(
|
|
pre_ctx, tool_name, tool_calls[idx:], round_actions, drop=reload_page is None
|
|
):
|
|
break
|
|
# Charged only once the call is really dispatched: a call voided by a refresh spent nothing.
|
|
st.total_tool_calls += 1
|
|
if activity is not None:
|
|
# Refreshed per call, not per turn: a batched action+finish turn must not defer on
|
|
# a stale turn-start snapshot (the conversion the headroom guard exists to prevent).
|
|
activity.tool_calls_remaining = st.max_tool_calls - st.total_tool_calls
|
|
selector_kind = _selector_kind(args)
|
|
# Cleared per call, so a value can never carry over from the previous one in the batch.
|
|
_RESOLVE_SECONDS.set(None)
|
|
_HIT_CLASS.set(None)
|
|
_COVERED_LAYER.set(None)
|
|
_TEXT_DELTA.set(None)
|
|
_TOOL_CALL_SEQ.set(st.total_tool_calls)
|
|
dispatch_ctx = skyvern_context.current()
|
|
runtime_secrets_before = len(dispatch_ctx.runtime_secret_values) if dispatch_ctx is not None else 0
|
|
# As in V1/V2, acting on the page again means an earlier navigation failure is no longer what the
|
|
# task ends on; reads and finish keep it, since a run that ends there ended on that navigation.
|
|
if spec is not None and spec.touches_page and dispatch_ctx is not None and dispatch_ctx.task_id:
|
|
clear_task_nav_error_code(dispatch_ctx.task_id)
|
|
tool_started_at = time.monotonic()
|
|
if spec is None:
|
|
result = ToolResult.error(f"unknown_tool: {tool_name}")
|
|
else:
|
|
if spec.billable:
|
|
turn_dispatched_billable = True
|
|
try:
|
|
result = await spec.handler(args)
|
|
except ToolRefusal as refusal:
|
|
result = refusal.as_result()
|
|
except Exception as exc:
|
|
LOG.warning(
|
|
"taskv3 tool handler raised",
|
|
tool=tool_name,
|
|
exc_info=True,
|
|
**_navigate_record_fields(tool_name, args, None),
|
|
)
|
|
raised_class = _raised_error_class(exc)
|
|
result = ToolResult.error(f"tool_error: {type(exc).__name__}: {exc}", error_class=raised_class)
|
|
tool_duration_seconds = time.monotonic() - tool_started_at
|
|
st.tool_seconds += tool_duration_seconds
|
|
# A dispatched page action consumes a step even if it errors (it may mutate before failing),
|
|
# unless the tool refused it before acting on the page, within the grace. Billing counts
|
|
# successes only.
|
|
call_charged = spec is not None and spec.billable
|
|
if call_charged and result.status == "error" and result.refused and not result.touched_page:
|
|
call_charged = st.uncharged_refusals >= UNCHARGED_REFUSAL_GRACE
|
|
st.uncharged_refusals += 0 if call_charged else 1
|
|
elif call_charged:
|
|
st.uncharged_refusals = 0
|
|
turn_charged = turn_charged or call_charged
|
|
# Entry is recorded only when the tool succeeded (a stale-ref failure put nothing in the
|
|
# field), but the submit is recorded on DISPATCH whatever the verdict: the loop ran the
|
|
# click itself, so nothing here asks the page whether a submission completed. Entry first,
|
|
# so a type that pressed Enter submits the value it just entered.
|
|
if spec is not None:
|
|
if tool_name in CREDENTIAL_ENTRY_TOOLS and result.status == "ok":
|
|
st.credentials_entered |= _credential_placeholders(args)
|
|
if _may_submit(tool_name, args):
|
|
# Cleared, so a second click on an untouched form is not a second submission of
|
|
# a credential that was only ever entered once.
|
|
for token in st.credentials_entered:
|
|
st.credential_submits[token] = st.credential_submits.get(token, 0) + 1
|
|
st.credentials_entered.clear()
|
|
# Observe's summary counters are the only trace a perception change leaves on this
|
|
# record; its content is deliberately never logged. Gated on the tool, not the payload,
|
|
# so every other tool's record keeps exactly today's fields.
|
|
observe_summary = _observe_summary_fields(result) if tool_name == "observe" else {}
|
|
# Which URL the call was about, so a navigate row is attributable to a target without the
|
|
# arguments themselves being logged.
|
|
navigate_fields = _navigate_record_fields(tool_name, args, result)
|
|
menu_note_fields = _menu_note_record_fields(tool_name, result)
|
|
# Conditional for the same reason observe's counters are: a record only carries a field
|
|
# the call actually produced, so an ok call's record keeps exactly the fields it has
|
|
# today and `resolve_seconds` is absent (not null) on tools with no address to resolve.
|
|
cost_fields: dict[str, Any] = {}
|
|
if result.status == "error":
|
|
# `tool_error_class` on the record, `error_class` on the result: the log key lands in a
|
|
# FLAT index where `error_class` is already taken -- cloud/webeye logs it as
|
|
# `type(exc).__name__` at ~17 sites, so an unprefixed key here would mix this closed
|
|
# vocabulary with Python exception names under one facet, and taskv3 emits on every
|
|
# erroring tool call so it would dominate the values.
|
|
cost_fields["tool_error_class"] = result.error_class or "other"
|
|
# Only on the rows it describes, like the fields below: a billable call that did not
|
|
# spend an action step because the tool refused it before acting on the page.
|
|
if spec is not None and spec.billable and not call_charged:
|
|
cost_fields["charged"] = False
|
|
# Only on the class they describe, so every other erroring row keeps exactly the
|
|
# fields it has today. Total over `covered` rows by construction: the helper that
|
|
# builds all three messages records before it returns any of them, so a covered row
|
|
# missing these means the class was emitted somewhere that is not that helper.
|
|
covered = _COVERED_LAYER.get()
|
|
if result.error_class == "covered" and covered is not None:
|
|
cost_fields["covered_branch"] = covered["branch"]
|
|
cost_fields["covered_controls"] = covered["controls"]
|
|
cost_fields["covered_layer_kind"] = covered["layer_kind"]
|
|
elif result.ok_class is not None:
|
|
# Prefixed for the same flat-index reason as `tool_error_class` above. NOT defaulted
|
|
# the way that field is: it is emitted only by tools whose `ok` spans distinct
|
|
# outcomes, so a fleet-wide default would put a field on every successful call to
|
|
# say nothing. WHAT A DENOMINATOR MEANS HERE, because a groupBy drops rows missing a
|
|
# facet: grouping a tool by this facet alone shows its `ok` calls ONLY and silently
|
|
# excludes its errors. `tool_status` is the total partition -- cut on it first, then
|
|
# read this within `ok` and `tool_error_class` within `error`. A tool that emits
|
|
# this at all must emit it on EVERY `ok` branch, or the buckets don't sum to `ok`.
|
|
cost_fields["tool_ok_class"] = result.ok_class
|
|
# Read off the context variable, not the result: on the raise path the loop built the
|
|
# result itself and the handler's own resolution time would otherwise be lost.
|
|
resolve_seconds = _RESOLVE_SECONDS.get()
|
|
if resolve_seconds is not None:
|
|
cost_fields["resolve_seconds"] = resolve_seconds
|
|
text_delta = _TEXT_DELTA.get()
|
|
if text_delta is not None:
|
|
cost_fields["text_delta_seconds"] = text_delta[0]
|
|
if text_delta[1] is not None:
|
|
cost_fields["text_delta_lines"] = text_delta[1]
|
|
cost_fields["text_delta_chars"] = text_delta[2]
|
|
if text_delta[3]:
|
|
cost_fields["text_delta_over_bound"] = True
|
|
if text_delta[4]:
|
|
cost_fields["text_delta_skipped"] = text_delta[4]
|
|
if text_delta[5]:
|
|
cost_fields["text_delta_pending_lines"] = text_delta[5]
|
|
# Every click row carries this, defaulting to `unknown`, because a groupBy DROPS rows
|
|
# missing a facet -- a partial field would read as a clean result rather than a gap.
|
|
# Gated on `spec is not None` for the same reason the `tool` field is: an unregistered
|
|
# tool logs as `unknown_tool`, and such a row must not also carry a click-only facet.
|
|
if spec is not None and tool_name == "click":
|
|
hit = _HIT_CLASS.get() or {}
|
|
cost_fields["hit_class"] = hit.get("hit_class") or "unknown"
|
|
# Known for every click whether or not the probe ran, so it stays total alongside the
|
|
# class. The two qualifiers below are NOT total by design: they describe a reading
|
|
# that happened, and a fabricated `false` would be indistinguishable from a measured
|
|
# one -- so group them WITHIN a hit_class bucket, never alone.
|
|
|
|
# What the probe CALL cost at the production call site. Not the marginal round trip:
|
|
# the first reading per realm also pays isolated-world construction, which is why the
|
|
# median within one `hit_probe_isolated` bucket is the readable figure.
|
|
# Present together, iff the probe was attempted: `hit_class` is the only total one.
|
|
probe_seconds = hit.get("probe_seconds")
|
|
if probe_seconds is not None:
|
|
cost_fields["hit_probe_seconds"] = probe_seconds
|
|
cost_fields["hit_needed"] = bool(hit.get("needed"))
|
|
cost_fields["hit_probe_raised"] = bool(hit.get("raised"))
|
|
if hit.get("isolated") is not None:
|
|
cost_fields["hit_probe_isolated"] = hit["isolated"]
|
|
# The action-loop guard's key and the perception ledger's digest, computed here (pure) so
|
|
# their hashes ride the record below; the ledger itself is updated further down, unchanged.
|
|
action_key = (tool_name, json.dumps(args, sort_keys=True, default=str))
|
|
attribution: dict[str, Any] = {"action_key_hash": telemetry_hash(telemetry_salt, *action_key)}
|
|
content_digest: str | None = None
|
|
# Whether this result is markup a window cut open — the only thing that can carry a
|
|
# marker fragment at its head. `rendered_text` is the tool's own statement that its
|
|
# content holds no start tags of the page's own; asking the ARGUMENTS instead would make
|
|
# the loop re-derive from a tool's argument conventions something the tool already said.
|
|
# Exact spans the TOOL reported: where it cut a marker open at the window's head, and
|
|
# where it appended its own notice. Neither is re-derived here, because page- and
|
|
# server-authored text can both wear those shapes and only the tool knows what it wrote.
|
|
reported = result.data or {}
|
|
head_fragment_len = int(reported.get("head_fragment_len") or 0)
|
|
notice_at = reported.get("notice_at")
|
|
clip_spans = reported.get("clip_spans")
|
|
# The text-delta section's span in `result.content`, read once for this result: the stall
|
|
# digest drops exactly that span, and the perception store below re-finds it after masking.
|
|
reported_delta_at = reported.get("delta_at")
|
|
reported_delta_end = reported.get("delta_end")
|
|
if spec is not None and spec.compactable and result.status == "ok":
|
|
content_digest = hashlib.sha256(
|
|
_canonical_perception_content(
|
|
result.content,
|
|
is_observe=tool_name == "observe",
|
|
head_fragment_len=head_fragment_len,
|
|
notice_at=notice_at,
|
|
clip_spans=clip_spans,
|
|
delta_at=reported_delta_at,
|
|
delta_end=reported_delta_end,
|
|
).encode()
|
|
).hexdigest()
|
|
attribution["snapshot_digest"] = telemetry_hash(telemetry_salt, content_digest)
|
|
attribution["probe_first_time"] = st.perception.first_time(action_key)
|
|
# Emitted on its own record, never folded into the one above: the tool-call record's
|
|
# fields are a stable contract and this signal must not perturb them.
|
|
seen_count, new_states_since = st.revisit_memory.record(content_digest)
|
|
# A revisit with fresh ground covered in between is a drill-down returning to its
|
|
# list, not a replay. Reporting those too would fire the signal identically on
|
|
# healthy and stuck runs and leave the precision read it exists to feed unable to
|
|
# tell them apart.
|
|
if seen_count >= PERCEPTION_REVISIT_LOG_AFTER and not new_states_since:
|
|
LOG.info(
|
|
PERCEPTION_REVISIT_EVENT,
|
|
revisit_count=seen_count,
|
|
distinct_states=st.revisit_memory.distinct_states,
|
|
peak_revisits=st.revisit_memory.peak_revisits,
|
|
progressed_revisits=st.revisit_memory.progressed_revisits,
|
|
capped=st.revisit_memory.capped,
|
|
tool=tool_name,
|
|
turn=st.turns,
|
|
**attribution,
|
|
)
|
|
# The only per-tool-call timing the engine has: tool execution is the majority of a v3
|
|
# run's wall-clock and otherwise emits nothing at all. Names, sizes and booleans only —
|
|
# argument values and result content carry end-user data and must not be logged.
|
|
LOG.info(
|
|
"taskv3 tool call finished",
|
|
# A hallucinated name would otherwise put unbounded model output into an indexed
|
|
# field on every call; the name itself stays in the tool result the model reads.
|
|
tool=tool_name if spec is not None else "unknown_tool",
|
|
tool_status=result.status,
|
|
duration_seconds=tool_duration_seconds,
|
|
result_chars=len(result.content),
|
|
# Truthiness, not presence: the tools treat a null or empty selector as absent and
|
|
# fall back to scanning the whole page, which is the case this field exists to find.
|
|
selector_present=bool(args.get("selector")),
|
|
selector_kind=selector_kind,
|
|
billable=bool(spec is not None and spec.billable),
|
|
turn=st.turns,
|
|
batch_size=len(tool_calls),
|
|
batch_index=idx,
|
|
**cost_fields,
|
|
**observe_summary,
|
|
**navigate_fields,
|
|
**menu_note_fields,
|
|
**attribution,
|
|
)
|
|
if spec is not None and spec.billable:
|
|
target_key = _call_selector(args) or f"tool:{tool_name}"
|
|
same_count, same_errors = st.canonical.record_touch(target_key, result.status == "error")
|
|
# Every rung of the distribution is logged (not one cliff): the Phase-B block
|
|
# threshold is tuned from this data.
|
|
if (
|
|
same_count in CANONICAL_LOOP_FIRE_COUNTS
|
|
and same_errors >= same_count - 1
|
|
and st.canonical.claim_rung(target_key, same_count)
|
|
):
|
|
pending_canonical_fires.append(
|
|
{
|
|
"gen": st.canonical.gen,
|
|
"target_key_hash": telemetry_hash(telemetry_salt, target_key),
|
|
"tool": tool_name,
|
|
"repeat_count": same_count,
|
|
"repeat_errors": same_errors,
|
|
"turn": st.turns,
|
|
}
|
|
)
|
|
|
|
model_facing_content = result.content
|
|
skyvern_ctx = skyvern_context.current()
|
|
if skyvern_ctx is not None:
|
|
model_facing_content = skyvern_ctx.hide_from_model(model_facing_content)
|
|
# Masking can change lengths, so the tool's offset is re-found as the masked delta's suffix
|
|
# position; if masking broke the suffix, the whole result counts as the read.
|
|
delta_at: int | None = None
|
|
if type(reported_delta_at) is int and 0 <= reported_delta_at <= len(result.content):
|
|
delta = result.content[reported_delta_at:]
|
|
if skyvern_ctx is not None:
|
|
delta = skyvern_ctx.hide_from_model(delta)
|
|
if model_facing_content.endswith(delta):
|
|
delta_at = len(model_facing_content) - len(delta)
|
|
transcript_content = model_facing_content
|
|
if spec is not None and spec.compactable and result.status == "ok":
|
|
transcript_content = st.reads.admit(
|
|
len(st.messages),
|
|
tool=tool_name,
|
|
args=_declared_args_key(spec, args),
|
|
page=probe_before or "",
|
|
handles=spec.issues_handles,
|
|
content=model_facing_content,
|
|
turn=st.turns,
|
|
delta_at=delta_at,
|
|
)
|
|
# A refresh raised by this call re-baselines the stall and repeat ledgers below, so neither
|
|
# guard may end the run on it first.
|
|
refresh_pending = skyvern_ctx is not None and skyvern_ctx.refresh_working_page
|
|
st.messages.append(
|
|
{"role": "tool", "tool_call_id": tool_call_id, "name": tool_name, "content": transcript_content}
|
|
)
|
|
if tool_trail is not None and spec is not None and not spec.terminal:
|
|
tool_trail.record(
|
|
TrailEntry(
|
|
tool=tool_name,
|
|
status=result.status,
|
|
content=model_facing_content,
|
|
perception=spec.compactable,
|
|
page_changing=spec.touches_page and (result.touched_page or not result.refused),
|
|
# The two ways a secret reaches the page: a credential placeholder typed by an
|
|
# entry tool, and a value a tool registered as secret (a delivered verification
|
|
# code) for the model to type next.
|
|
# `navigate` resolves placeholders in a URL too.
|
|
secret_entered=(
|
|
(tool_name in CREDENTIAL_ENTRY_TOOLS or tool_name == "navigate")
|
|
and bool(_credential_placeholders(args))
|
|
)
|
|
or (
|
|
dispatch_ctx is not None
|
|
and len(dispatch_ctx.runtime_secret_values) > runtime_secrets_before
|
|
),
|
|
entered=entered_values(tool_name, args),
|
|
# Masked like the content, so an unchanged URL compares equal to the one observe prints.
|
|
url_before=_trail_url_before(result, skyvern_ctx),
|
|
)
|
|
)
|
|
# A look's annotated screenshot is shown to the model on the next call only, never stored in
|
|
# the transcript. Consumed and cleared at the top of the next turn. Only the LATEST snapshot
|
|
# survives: a second look in the same turn supersedes the first (its marks replace the prior
|
|
# ones), so re-sending the stale image would just hand the model a dead numbering.
|
|
if result.screenshots:
|
|
st.pending_screenshots = list(result.screenshots)
|
|
result_data = result.data or {}
|
|
# The page-state stall detector (SKY-15265) re-baselines on these flags. Keeps the FIRST
|
|
# qualifying signal this batch -- the reason field is diagnostic (telemetry), not the
|
|
# decision itself, so a later call's signal never overwrites it.
|
|
if batch_page_change_reason is None or batch_page_change_reason == "page_transitioned":
|
|
# page_transitioned checks LAST and can be superseded: it is a URL-only hint the
|
|
# stall detector must not reset on, so a stronger same-batch signal outranks it.
|
|
if result_data.get("page_state_changed"):
|
|
batch_page_change_reason = "page_state_changed"
|
|
elif result_data.get("download_new"):
|
|
# A freshly detected download is progress even when the DOM never moves (a
|
|
# download-next flow); a replayed notice deliberately does not qualify.
|
|
batch_page_change_reason = "download_new"
|
|
elif tool_name == "hover" and result.status == "ok":
|
|
# A hover's only purpose is to reveal state (submenus, tooltips) that a
|
|
# CSS-only change leaves invisible to the innerHTML fingerprint and the
|
|
# document-identity probe alike -- always treat it as a change signal.
|
|
batch_page_change_reason = "hover"
|
|
elif tool_name == "navigate" and result.status == "ok":
|
|
batch_page_change_reason = "navigate"
|
|
elif batch_page_change_reason is None and result_data.get("page_transitioned") is True:
|
|
batch_page_change_reason = "page_transitioned"
|
|
shadow_baseline_before = st.progress.invalid_baseline if st.progress is not None else None
|
|
if _absorb_result_data(tool_name, spec, result_data):
|
|
# Every mark=N still queued in this batch was chosen before this look renumbered the
|
|
# marks, so it now names an arbitrary element; a look refused before rebuilding its
|
|
# manifest leaves the old marks (and their failed keys) live.
|
|
marks_stale = True
|
|
st.canonical.invalidate_marks()
|
|
_progress_observe_shadow(observe_summary, tool_name, attribution, shadow_baseline_before)
|
|
if content_digest is not None:
|
|
stall_outcome, nudges_due = _perception_stall_check(
|
|
st.perception,
|
|
content_digest,
|
|
action_key,
|
|
tool_name,
|
|
attribution,
|
|
content_only_digest=hashlib.sha256(
|
|
_content_only_perception(
|
|
result.content,
|
|
is_observe=tool_name == "observe",
|
|
head_fragment_len=head_fragment_len,
|
|
notice_at=notice_at,
|
|
clip_spans=clip_spans,
|
|
delta_at=reported_delta_at,
|
|
delta_end=reported_delta_end,
|
|
).encode()
|
|
).hexdigest(),
|
|
refresh_pending=refresh_pending,
|
|
)
|
|
st.stall_nudges_due.extend(nudges_due)
|
|
if stall_outcome is not None:
|
|
st.outcome = stall_outcome
|
|
_append_skipped_tool_results(st.messages, tool_calls[idx + 1 :], "perception stalled")
|
|
break
|
|
if spec is not None and spec.billable:
|
|
if st.progress is not None:
|
|
st.progress.on_billable()
|
|
# Errored and refused dispatches count too: a repeat-failing action is the same
|
|
# no-progress pathology. It does not bound refusals (a stale one resets it);
|
|
# UNCHARGED_REFUSAL_GRACE does.
|
|
repeat_count, first_turn, streak_fp, streak_moved = st.action_counts.get(
|
|
action_key, (0, st.turns, batch_fp_before, False)
|
|
)
|
|
repeat_count += 1
|
|
streak_moved = streak_moved or (
|
|
streak_fp is not None and batch_fp_before is not None and batch_fp_before != streak_fp
|
|
)
|
|
st.action_counts[action_key] = (repeat_count, first_turn, streak_fp, streak_moved)
|
|
# Terminate only when the streak spans more than one turn AND its warning was
|
|
# delivered: the system prompt commands batching identical clicks (steppers,
|
|
# arrows), so a single-batch streak has had no chance to see feedback yet, and a
|
|
# verdict must never arrive before the model saw the warning it could have acted on.
|
|
if (
|
|
action_terminate_after is not None
|
|
and repeat_count >= action_terminate_after
|
|
and first_turn < st.turns
|
|
and (action_nudge_after is None or action_key in st.action_warned)
|
|
and not refresh_pending
|
|
):
|
|
LOG.info(
|
|
"taskv3 loop action repeated",
|
|
tool=tool_name,
|
|
repeat_count=repeat_count,
|
|
turn=st.turns,
|
|
guard=ACTION_LOOP_GUARD,
|
|
**attribution,
|
|
)
|
|
# The customer-facing half names the control by the page's own accessible name, never
|
|
# the selector the model's nudge uses: that selector is usually v3's own `data-tv3`
|
|
# marker, which matches nothing in the customer's markup (SKY-16271).
|
|
named_target = _verdict_target(
|
|
tool_name,
|
|
result_data.get(TARGET_LABEL_DATA_KEY),
|
|
result_data.get(TARGET_KIND_DATA_KEY),
|
|
label_secret_values,
|
|
skyvern_ctx,
|
|
)
|
|
st.outcome = _guard_verdict(ACTION_LOOP_GUARD, _action_loop_reason(named_target))
|
|
_append_skipped_tool_results(st.messages, tool_calls[idx + 1 :], "action loop")
|
|
break
|
|
if (
|
|
action_nudge_after is not None
|
|
and repeat_count >= action_nudge_after
|
|
and action_key not in st.action_warned
|
|
):
|
|
action_nudges_due.append((tool_name, args, repeat_count))
|
|
if submit_watch is not None and tool_name == "navigate" and result.status == "ok":
|
|
# Outside the billable/recordable branch on purpose: the run left the page and the
|
|
# control it clicked went too, whatever navigate's spec flags happen to say.
|
|
submit_watch.clear()
|
|
if activity is not None and spec is not None:
|
|
# The round_actions branch below is deliberately NARROWER: this one adds
|
|
# `engages_page`, which is engagement without billing or an action row. Perception is
|
|
# counted separately off `compactable`, the flag observe / get_html / look already
|
|
# share, rather than a tool name -- `look` does not even exist when vision is off.
|
|
if spec.touches_page:
|
|
activity.action_attempts += 1
|
|
elif spec.compactable and result.status == "ok":
|
|
activity.perceptions += 1
|
|
if spec is not None and (spec.billable or spec.recordable):
|
|
# Dispatched page actions enter the round with their outcome: a failed charged round
|
|
# still consumed budget and must persist (else later blocks undercount the run
|
|
# budget); refused calls and recordable tools persist without claiming a budget unit.
|
|
round_outcome = result_data.get(ACTION_OUTCOME_DATA_KEY)
|
|
round_outcome = round_outcome if isinstance(round_outcome, dict) else None
|
|
round_actions.append(
|
|
RoundAction(
|
|
tool_name,
|
|
args,
|
|
result.status == "ok" and not _outcome_reports_failure(round_outcome),
|
|
result_data.get(TARGET_LABEL_DATA_KEY) or None,
|
|
result_data.get(TARGET_KIND_DATA_KEY) or None,
|
|
call_charged,
|
|
round_outcome,
|
|
result.content if result.status == "error" else None,
|
|
)
|
|
)
|
|
if spec.billable and result.status == "ok":
|
|
st.billable_actions.append(tool_name)
|
|
if activity is not None and _arms_failure_evidence(tool_name, args, result.status == "ok"):
|
|
activity.last_trigger_turn = st.turns
|
|
if submit_watch is not None:
|
|
submit_selector = _names_submit_control(tool_name, args, result.status == "ok")
|
|
if submit_selector is not None:
|
|
submit_watch.record(submit_selector)
|
|
|
|
# A page-level handler (e.g. an anti-bot bypass that exhausted its retries) can reload the
|
|
# page out from under any tool call, billable or not; honored after the call's own bookkeeping,
|
|
# never over a verdict the call itself just produced, mirroring the legacy per-action check.
|
|
if skyvern_ctx is not None and skyvern_ctx.refresh_working_page:
|
|
# Consumed whether or not it can be acted on: the context outlives this run, and a
|
|
# signal left set would fire on the first action of the next block.
|
|
verdict_stands = spec is not None and spec.terminal and result.status == "ok"
|
|
if await _consume_refresh_signal(
|
|
skyvern_ctx,
|
|
tool_name,
|
|
tool_calls[idx + 1 :],
|
|
round_actions,
|
|
drop=reload_page is None or verdict_stands,
|
|
):
|
|
break
|
|
|
|
completion_outcome = await _completion_probe_outcome(tool_name, spec, result_data)
|
|
if completion_outcome is not None:
|
|
st.outcome = completion_outcome
|
|
if spec is not None and spec.billable:
|
|
# The download landed while this call's probe waited, so this call's touch is
|
|
# the progressing one: drop ITS pending rungs only, never a sibling target's.
|
|
completing_hash = telemetry_hash(telemetry_salt, _call_selector(args) or f"tool:{tool_name}")
|
|
pending_canonical_fires[:] = [
|
|
fire for fire in pending_canonical_fires if fire["target_key_hash"] != completing_hash
|
|
]
|
|
_append_skipped_tool_results(st.messages, tool_calls[idx + 1 :], "completion probe fired")
|
|
break
|
|
|
|
dead_end_status = result_data.get("navigation_dead_end")
|
|
if dead_end_status is not None:
|
|
# A hard 404/410 landing is a non-capability dead-end (a dead/removed posting). Remember
|
|
# it but do NOT break the batch: a later navigate in the same turn can land the run on a
|
|
# live page and clear it below. Applied once the batch settles (after this for-loop).
|
|
st.pending_nav_dead_end = _NavDeadEnd(dead_end_status, result_data.get("navigation_dead_end_url"))
|
|
elif result_data.get("page_state_changed"):
|
|
# A successful navigate moved the run off any dead page seen earlier this batch.
|
|
st.pending_nav_dead_end = None
|
|
|
|
if activity is not None and activity.held_verdict_batch_skip:
|
|
# Raised only by the goal-check enforce hold. Its held finish is neither billable nor recordable,
|
|
# so the generic error branch below would let a click or type batched behind it dispatch.
|
|
activity.held_verdict_batch_skip = False
|
|
_append_skipped_tool_results(
|
|
st.messages,
|
|
tool_calls[idx + 1 :],
|
|
"queued behind a verdict that was held; re-observe the page before choosing again",
|
|
)
|
|
break
|
|
|
|
if spec is not None and spec.terminal and result.status == "ok":
|
|
data = result.data or {}
|
|
st.outcome = LoopOutcome(
|
|
status=data.get("status", "completed"),
|
|
reason=data.get("reason", ""),
|
|
extracted_output=data.get("extracted_output"),
|
|
goal_check_ended=bool(data.get("goal_check_ended")),
|
|
converted_from=_non_completed(data.get("converted_from")),
|
|
converted_from_reason=data.get("converted_from_reason", ""),
|
|
# The model's own verdict wins whether or not it landed on the granted final turn;
|
|
# cap_trip just records the fact that a cap forced this to be the last turn.
|
|
cap_trip=st.cap_trip_pending if st.final_turn_granted else None,
|
|
)
|
|
break
|
|
|
|
if result.status == "ok" and result_data.get("readiness_incomplete"):
|
|
# A navigate whose document committed but did not finish loading. The rest of the batch
|
|
# was queued against a loaded page, and before SKY-16278 the tool raised here and the
|
|
# error branch below skipped it; keep that, so the readiness the tool reported reaches
|
|
# the model before it chooses its next action.
|
|
_append_skipped_tool_results(
|
|
st.messages,
|
|
tool_calls[idx + 1 :],
|
|
"the page this batch navigated to had not finished loading — observe it before re-queuing these",
|
|
)
|
|
break
|
|
|
|
# A toggle can auto-advance a wizard, leaving the batch's failed field on the step behind it.
|
|
# Read off the click's own before/after URL comparison, so the check costs nothing extra.
|
|
if toggle_exempted and result.status == "ok" and result_data.get("page_transitioned") is True:
|
|
_append_skipped_tool_results(
|
|
st.messages,
|
|
tool_calls[idx + 1 :],
|
|
"an earlier field in this batch failed and was left on the previous step; the page moved "
|
|
"on — re-observe",
|
|
)
|
|
break
|
|
|
|
if result.status == "error":
|
|
# A live refusal needs a new model choice, so stop_batch skips pending calls without claiming progress.
|
|
poisoned = (
|
|
tool_name == "navigate"
|
|
or result.content == PAGE_UNAVAILABLE_ERROR
|
|
or bool(
|
|
result_data.get("page_transitioned")
|
|
or result_data.get("page_state_changed")
|
|
or result_data.get("navigation_dead_end")
|
|
or result_data.get("stop_batch")
|
|
)
|
|
)
|
|
# Whether the page moved is independent of the tool's kind: a wait that timed out because
|
|
# the site navigated poisons the batch just as a failed click would.
|
|
if not poisoned and spec is not None and page_probe is not None:
|
|
probe_after = await _sample_probe(page_probe, deadline_at=deadline_at)
|
|
poisoned = probe_before is None or probe_after is None or probe_after != probe_before
|
|
if probe_before is not None and probe_after is not None and probe_after != probe_before:
|
|
# Two landed identity samples that differ: the failed call still moved the
|
|
# document, and the fingerprint may render identically on the new one (a
|
|
# same-template step) — absorb it here or the rung survives that blindness.
|
|
st.canonical.progress(_ProgressEvidence.PROBE_MISMATCH)
|
|
if poisoned:
|
|
_append_skipped_tool_results(
|
|
st.messages,
|
|
tool_calls[idx + 1 :],
|
|
"earlier tool call in this batch failed and changed the page — re-observe before "
|
|
"re-queuing these",
|
|
)
|
|
break
|
|
if spec is not None and (spec.billable or spec.recordable):
|
|
batch_had_failure = True
|
|
if call_selector is not None:
|
|
failed_selectors.add(call_selector)
|
|
|
|
# Whatever skipped or refused a finish once a cap tripped — behind the very call that
|
|
# tripped it in the GRANTING batch, behind a failed call or fail-closed blocker on the
|
|
# granted turn, or under a guard that terminated the batch — the extraction it was already
|
|
# carrying is remembered; the post-loop stamp rides it out on whatever terminal the run
|
|
# ends with. Last write wins, so a granted-turn restatement supersedes the older capture.
|
|
if st.final_turn_granted:
|
|
staged = next(
|
|
(
|
|
t_args.get("extracted_output")
|
|
for _t_id, t_name, t_args in tool_calls
|
|
if t_name == "finish" and isinstance(t_args, dict) and t_args.get("extracted_output") is not None
|
|
),
|
|
None,
|
|
)
|
|
if staged is not None:
|
|
st.final_turn_staged_output = staged
|
|
|
|
# The batch settled on a dead page (an in-loop navigate hit a hard 404/410 and no later navigate
|
|
# recovered): end the run as terminated deterministically, matching v1, rather than leaving the
|
|
# failed/terminated choice to the model's finish tool (which does not converge on this class).
|
|
if st.outcome is None and st.pending_nav_dead_end is not None:
|
|
LOG.info(
|
|
"taskv3 loop navigation dead end",
|
|
http_status=st.pending_nav_dead_end.status,
|
|
turn=st.turns,
|
|
guard=NAV_DEAD_END_GUARD,
|
|
)
|
|
st.outcome = _guard_verdict(
|
|
NAV_DEAD_END_GUARD,
|
|
_dead_end_reason(
|
|
st.pending_nav_dead_end.status,
|
|
st.pending_nav_dead_end.url,
|
|
page_noun="The page the run opened",
|
|
caller_known_urls=caller_known_urls,
|
|
),
|
|
)
|
|
|
|
# Page-state stall detector (SKY-15265): tool-independent — any batch of billable work that
|
|
# leaves the rendered document byte-identical ticks the counter, whatever tools produced it.
|
|
# A missing sample is no evidence either way; any page-change flag or fingerprint movement
|
|
# re-baselines.
|
|
if (
|
|
st.outcome is None
|
|
and turn_dispatched_billable
|
|
and page_fingerprint is not None
|
|
and batch_fp_before is not None
|
|
):
|
|
if st.page_state_prev_fp is not None and batch_fp_before != st.page_state_prev_fp:
|
|
# The page moved BETWEEN batches (a delayed render landing after the prior
|
|
# after-sample): the streak the old samples described is stale.
|
|
st.trailing_page_state_stall_rounds = 0
|
|
st.page_state_nudge_delivered = False
|
|
page_state_changed: bool | None
|
|
if batch_page_change_reason is not None and batch_page_change_reason != "page_transitioned":
|
|
page_state_changed = True
|
|
else:
|
|
if batch_fp_after is None and not (deadline_at is not None and deadline_at - time.monotonic() <= 0):
|
|
batch_fp_after = await _sample_probe(page_fingerprint, deadline_at=deadline_at)
|
|
page_state_changed = None if batch_fp_after is None else batch_fp_after != batch_fp_before
|
|
st.page_state_prev_fp = batch_fp_after if batch_fp_after is not None else batch_fp_before
|
|
if batch_fp_after is not None:
|
|
for key, (count, first_turn, streak_fp, moved) in st.action_counts.items():
|
|
if not moved and streak_fp is not None and streak_fp != batch_fp_after:
|
|
st.action_counts[key] = (count, first_turn, streak_fp, True)
|
|
st.page_state_ever_judged = st.page_state_ever_judged or page_state_changed is not None
|
|
if page_state_changed is True:
|
|
st.trailing_page_state_stall_rounds = 0
|
|
st.page_state_nudge_delivered = False
|
|
st.canonical.progress(_ProgressEvidence.PAGE_STATE_VERDICT)
|
|
elif page_state_changed is False:
|
|
st.trailing_page_state_stall_rounds += 1
|
|
st.peak_page_state_stall_rounds = max(
|
|
st.peak_page_state_stall_rounds, st.trailing_page_state_stall_rounds
|
|
)
|
|
if (
|
|
st.trailing_page_state_stall_rounds == PAGE_STATE_STALL_TERMINATE_AFTER
|
|
and st.page_state_nudge_delivered
|
|
):
|
|
# Shadow-only verdict: measured, not enforced (see PAGE_STATE_STALL_SHADOW_EVENT).
|
|
LOG.info(
|
|
PAGE_STATE_STALL_SHADOW_EVENT,
|
|
rounds=st.trailing_page_state_stall_rounds,
|
|
turn=st.turns,
|
|
)
|
|
elif (
|
|
st.trailing_page_state_stall_rounds >= PAGE_STATE_STALL_NUDGE_AFTER
|
|
and not st.page_state_nudge_delivered
|
|
):
|
|
st.page_state_nudge_due = True
|
|
|
|
if (
|
|
st.outcome is not None
|
|
# An acknowledged cancellation must not wait on a possibly-hung renderer just to
|
|
# decide telemetry; and a batch whose pending fires are all generation-stale already
|
|
# has nothing left to absorb for.
|
|
and st.outcome.status != "canceled"
|
|
and page_fingerprint is not None
|
|
and batch_fp_before is not None
|
|
and any(fire["gen"] == st.canonical.gen for fire in pending_canonical_fires)
|
|
):
|
|
# A terminal outcome mid-batch (a finish, a fired completion probe) skips the detector
|
|
# above, so the batch's own movement is unabsorbed here: take the after-sample now and
|
|
# clear on a positive mismatch only, before deciding the pending events below.
|
|
if batch_fp_after is None:
|
|
batch_fp_after = await _sample_probe(page_fingerprint, deadline_at=deadline_at)
|
|
if batch_fp_after is not None and batch_fp_after != batch_fp_before:
|
|
st.canonical.progress(_ProgressEvidence.TERMINAL_BATCH_FINGERPRINT)
|
|
# A rung completed by — or followed in this batch by — absorbed progress describes a
|
|
# progressing run, not a loop: only events whose generation survived every clear above
|
|
# (same-call result data, invalid-fields baseline, batch fingerprint verdict) are emitted.
|
|
for pending_fire in pending_canonical_fires:
|
|
if pending_fire.pop("gen") == st.canonical.gen:
|
|
LOG.info(CANONICAL_LOOP_EVENT, **pending_fire)
|
|
pending_canonical_fires.clear()
|
|
|
|
# Warn only after the batch completes: a user message may not sit between an assistant
|
|
# turn's tool results, and the model reads it with the snapshot that tripped it. Every note
|
|
# due this turn shares ONE user message so the transcript keeps alternating roles.
|
|
nudge_parts: list[str] = []
|
|
if st.budget_extended_notice is not None:
|
|
# First: it retracts a standing claim that the run is ending, which every other note
|
|
# this turn is written as if untrue.
|
|
nudge_parts.append(st.budget_extended_notice)
|
|
if st.outcome is None and st.refresh_nudge_due:
|
|
nudge_parts.append(_refresh_nudge_text())
|
|
elif st.outcome is None and st.reload_failed_nudge_due:
|
|
nudge_parts.append(_reload_failed_nudge_text())
|
|
if st.outcome is None and st.stall_nudges_due:
|
|
nudge_parts.append(_stall_nudge_text(st.stall_nudges_due, set(tool_by_name)))
|
|
if st.outcome is None and st.page_state_nudge_due:
|
|
st.page_state_nudge_due = False
|
|
st.page_state_nudge_delivered = True
|
|
LOG.info("taskv3 loop page state stall nudged", rounds=st.trailing_page_state_stall_rounds, turn=st.turns)
|
|
nudge_parts.append(_page_state_nudge_text(st.trailing_page_state_stall_rounds))
|
|
if st.outcome is None and st.verification_refusal_nudge is not None:
|
|
nudge_parts.append(st.verification_refusal_nudge)
|
|
st.verification_refusal_nudge = None
|
|
st.verification_refusal_nudged = True
|
|
if st.outcome is None and action_nudges_due:
|
|
# Deliver only warnings whose streak survived the batch AND spans turns: a later call in
|
|
# the same batch (an observe showing the page changed, a download) may have cleared it,
|
|
# and a streak born entirely this turn has had no feedback yet — the message's "the
|
|
# state you last observed is unchanged" would be false for it. An undelivered warning
|
|
# stays unmarked, so it re-queues (and termination stays blocked) until the model has
|
|
# actually seen it. Counts read live, not the threshold-crossing snapshot, and logged
|
|
# here so the warn-then-recovered metric counts only warnings the model saw.
|
|
still_stuck = []
|
|
movements: list[bool | None] = []
|
|
for name, warn_args, _count in action_nudges_due:
|
|
key = (name, json.dumps(warn_args, sort_keys=True, default=str))
|
|
entry = st.action_counts.get(key)
|
|
if entry is not None and entry[1] < st.turns and key not in st.action_warned:
|
|
st.action_warned.add(key)
|
|
still_stuck.append((name, warn_args, entry[0]))
|
|
streak_fp, latest_fp = entry[2], st.page_state_prev_fp
|
|
comparable = streak_fp is not None and latest_fp is not None
|
|
if entry[3] or (comparable and latest_fp != streak_fp):
|
|
movements.append(True)
|
|
else:
|
|
# No pair to compare means no reading, which _sample_probe defines as
|
|
# evidence of nothing — never a claim that the page held still.
|
|
movements.append(False if comparable else None)
|
|
if still_stuck:
|
|
page_moved = True if any(movements) else (None if None in movements else False)
|
|
for name, _warn_args, count in still_stuck:
|
|
LOG.info(
|
|
"taskv3 loop action repeat nudged",
|
|
tool=name,
|
|
repeat_count=count,
|
|
turn=st.turns,
|
|
page_moved=page_moved,
|
|
)
|
|
nudge_parts.append(_action_nudge_text(still_stuck, set(tool_by_name), page_moved=page_moved))
|
|
if nudge_parts:
|
|
st.messages.append({"role": "user", "content": "\n\n".join(nudge_parts)})
|
|
|
|
# A "step" is one action round: a turn that ran >=1 page-mutating action. Perception-only
|
|
# turns (observe/get_html) don't consume the caller's step budget — the step engine bundles
|
|
# perception into each step, so counting v3's perception rounds against the same budget
|
|
# under-counts equivalent work.
|
|
if turn_charged:
|
|
st.action_steps += 1
|
|
# Hand the round's executed actions to the caller so it can persist per-action artifacts
|
|
# (screenshot, DB rows) — kept out of this transport-agnostic core, like should_cancel. A
|
|
# persistence hiccup must not abort an otherwise-good run, so failures are contained here.
|
|
if round_actions and on_action_round is not None:
|
|
try:
|
|
await on_action_round(round_actions, text or reasoning_summary or None)
|
|
except Exception:
|
|
LOG.warning("taskv3 on_action_round callback failed", turn=st.turns, exc_info=True)
|
|
|
|
if st.outcome is None:
|
|
st.outcome = LoopOutcome("loop_error", "loop exited without an outcome")
|
|
|
|
# Every model- or loop-produced terminal once the cap tripped — a finish verdict, a guard
|
|
# termination (even in the granting batch itself), a stall exit, the spent-grant exit —
|
|
# happened UNDER the cap: it carries the cap fact, and a missing extraction is filled from
|
|
# what a skipped finish had staged (the model's own earlier data), so no exit path can
|
|
# re-discard the partial output. A completed verdict that didn't restate its output would
|
|
# otherwise be demoted for missing extraction.
|
|
if st.final_turn_granted and st.outcome.status in (
|
|
"completed",
|
|
"failed",
|
|
"terminated",
|
|
"budget_exhausted",
|
|
"loop_error",
|
|
):
|
|
if st.outcome.cap_trip is None:
|
|
st.outcome.cap_trip = st.cap_trip_pending
|
|
if st.outcome.extracted_output is None and not st.outcome.goal_check_ended:
|
|
st.outcome.extracted_output = st.final_turn_staged_output
|
|
|
|
ledger_fields: LedgerTerminalFields | None = None
|
|
if st.progress is not None and st.progress.ever_armed:
|
|
# The ledger is precision-biased (see _ProgressLedger), so read its fire precision as
|
|
# trustworthy but its recall as a FLOOR: few fires is not few stuck runs.
|
|
ledger_fields = LedgerTerminalFields(
|
|
peak_actions_since_progress=st.progress.peak_actions_since_progress,
|
|
actions_since_progress=st.progress.actions_since_progress,
|
|
form_armed=st.progress.form_armed,
|
|
would_fire=st.progress.shadow_reported,
|
|
)
|
|
st.outcome.telemetry = TerminalTelemetry(
|
|
# The sticky flag, not `form_armed`: the latter is the CURRENT look and is cleared by
|
|
# progress, so a run that saw a form early and lost it by the end reads False on it. This is
|
|
# the partition the two collapsed records used to encode by which one of them fired.
|
|
form_ever_armed=st.progress is not None and st.progress.ever_armed,
|
|
survival=st.canonical.survival_fields(),
|
|
ledger=ledger_fields,
|
|
# Present only where the detector actually judged the page at least once, so the field means
|
|
# "this was the worst streak" and never "nothing ever looked".
|
|
peak_page_state_stall_rounds=st.peak_page_state_stall_rounds if st.page_state_ever_judged else None,
|
|
# Present for any run that RE-READ a probe, which is the population the counter is defined
|
|
# on. It ships unconditionally there because it is the calibration input for a pending
|
|
# threshold: the shadow line fires only above the current cutoff, so nothing below it is
|
|
# observable from logs today.
|
|
peak_probe_revisits=st.perception.peak_probe_revisits if st.perception.revisit_chances else None,
|
|
semantic_commit=semantic_commit_stats,
|
|
# Present only for a run that took a perception read, the population the counters describe.
|
|
perception_reads=st.reads.log_fields() if st.reads.entries else None,
|
|
)
|
|
|
|
st.outcome.turns = st.turns
|
|
st.outcome.no_tool_call_turns = st.no_tool_call_turns
|
|
st.outcome.tool_choice_in_effect = "tool_choice" in active_call_kwargs
|
|
st.outcome.tool_calls = st.total_tool_calls
|
|
st.outcome.tool_seconds = st.tool_seconds
|
|
st.outcome.action_steps = st.action_steps
|
|
st.outcome.billable_actions = st.billable_actions
|
|
st.outcome.messages = st.messages
|
|
return st.outcome
|