mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-03 12:18:39 +00:00
Merge managed/ouroboros (v6.113.1 + PRs #371-373) into the revsub line
One conflict: probe_node_version NODE_OPTIONS scrub — the same fix landed
independently on both sides (a2c2f15c here, managed's hermetic-lane commit);
resolution keeps managed's comment wording, identical code.
# Conflicts:
# ouroboros/platform_layer.py
This commit is contained in:
commit
c3d141ff1c
38 changed files with 2237 additions and 222 deletions
|
|
@ -59,6 +59,7 @@ from devtools.benchmarks.cybergym.cybergym_wire import (
|
|||
_unwrap_http_json,
|
||||
_unwrap_http_payload,
|
||||
)
|
||||
from ouroboros.openrouter_attribution import OPENROUTER_APP_HEADERS
|
||||
from devtools.benchmarks.cybergym.cybergym_docker import (
|
||||
_EXPECTED_MODEL,
|
||||
_GATEWAY_TASK_ID,
|
||||
|
|
@ -442,7 +443,8 @@ class _LifecycleMixin:
|
|||
}
|
||||
response = self.config.http_runner(
|
||||
"POST", self.config.provider_url, body=body,
|
||||
headers={"Authorization": f"Bearer {key}"}, timeout=60,
|
||||
headers={"Authorization": f"Bearer {key}", **OPENROUTER_APP_HEADERS},
|
||||
timeout=60,
|
||||
)
|
||||
response = _unwrap_http_json(response, operation="provider probe")
|
||||
observed = str(response.get("model") or "").strip()
|
||||
|
|
|
|||
|
|
@ -51,7 +51,6 @@ if str(pathlib.Path(__file__).resolve().parents[3]) not in sys.path:
|
|||
|
||||
# SSOT for leak-target hosts/URL/query patterns (shared with tests). See
|
||||
# leak_targets.py for the pattern-design constraints.
|
||||
from devtools.benchmarks.gaia.leak_targets import LEAK_QUERY_RE, LEAK_URL_RE # noqa: E402
|
||||
# The solvers' SSOT prompt instructions are stripped from full-text traces before
|
||||
# scanning, so an echoed prompt cannot self-trip LEAK_QUERY_RE (the anti-leak text
|
||||
# names the answer-source concept; the format text contains "final answer").
|
||||
|
|
@ -60,6 +59,8 @@ from devtools.benchmarks.gaia.inspect_solver import ( # noqa: E402
|
|||
GAIA_EPISTEMIC_INSTRUCTION,
|
||||
GAIA_FORMAT_INSTRUCTION,
|
||||
)
|
||||
from devtools.benchmarks.gaia.leak_targets import LEAK_QUERY_RE, LEAK_URL_RE # noqa: E402
|
||||
from ouroboros.openrouter_attribution import OPENROUTER_APP_HEADERS # noqa: E402
|
||||
|
||||
# Web-ish Ouroboros tools whose args/results can carry URLs or retrieved text.
|
||||
# Every network-capable tool that stays ENABLED in the bench profiles must be
|
||||
|
|
@ -342,8 +343,15 @@ def _judge(sample_id: str, gold: str, acts: list, model: str, api_key: str) -> d
|
|||
)
|
||||
body = json.dumps({"model": model, "messages": [{"role": "user", "content": prompt}],
|
||||
"max_tokens": 200, "temperature": 0}).encode()
|
||||
req = urllib.request.Request("https://openrouter.ai/api/v1/chat/completions", data=body,
|
||||
headers={"Authorization": f"Bearer {api_key}", "Content-Type": "application/json"})
|
||||
req = urllib.request.Request(
|
||||
"https://openrouter.ai/api/v1/chat/completions",
|
||||
data=body,
|
||||
headers={
|
||||
"Authorization": f"Bearer {api_key}",
|
||||
"Content-Type": "application/json",
|
||||
**OPENROUTER_APP_HEADERS,
|
||||
},
|
||||
)
|
||||
try:
|
||||
with urllib.request.urlopen(req, timeout=90) as r:
|
||||
txt = json.load(r)["choices"][0]["message"]["content"]
|
||||
|
|
|
|||
|
|
@ -245,6 +245,7 @@ server.py (Starlette+uvicorn) ← HTTP + WebSocket on configurable host:port (de
|
|||
├── tool_access.py ← Tool API v2 policy matrix: ToolProfile × ResourceRoot × Operation; also projects the side-effect-free filesystem affordance map injected into runtime context and checks closed-enum subagent required_capabilities against the selected profile
|
||||
├── tools/deliverables_shell.py ← Direct cp/mv/ln Deliverables target and symlink-payload checks, before generic workspace-root admission
|
||||
├── tools/shell_audit.py ← Post-execution user_files/Deliverables custody audit and declared-output root projection used by process tools
|
||||
├── tools/write_shape.py ← Write-shape classification SSOT (extracted from shell_guards.py at the module-size gate; shell_guards re-exports every historical name): the indicator vocabularies, the mode-aware `interpreter_write_shape`, and `non_interpreter_write_shape` (membership floor for unconditional writers, real-channel evidence for the pure-filter utilities sort/uniq/sed/tar/gzip, prose words yielding to the caller's read-carve). ONE seam: no deterministic write/owner-control guard may consume a coarser write fact than this composition's
|
||||
├── skill_payload_binding.py ← Exact payload binding projection: preserves the physical package root while `.seed-origin` distinguishes launcher-owned native from markerless user-managed logical `external`, admits native `read`/`list`/`search` only for the settled direct/read-only profiles, and reuses selected-candidate inventory for bounded manifestless `skill_publish` recovery
|
||||
├── tool_policy.py ← Round-one tool visibility policy (tool sets live in tool_capabilities)
|
||||
├── utils.py ← Shared utilities; v5.8.3-rc.2 SSOT for JSON atomic writes/reads, UTC timestamps, hashes, log sanitization, and subprocess helpers
|
||||
|
|
@ -680,7 +681,7 @@ Primary navigation exposes Chat (Main), a collapsible Projects group, Files, Ski
|
|||
|
||||
Each active or deleting Project has a sidebar row. Active rows can be opened, renamed, or deleted through pointer- and keyboard-operable controls; the backend owns the 80-character name limit and lifecycle truth. Unread Projects sort ahead of read Projects and then by durable activity. A deleting Project becomes non-openable and visibly remains in the transitional state until the server publishes authoritative registry state. On narrow screens navigation becomes an explicit drawer and the Project chat becomes a full-width overlay with a backdrop. There is no gesture-only navigation layer competing with message scroll, text selection, or the software keyboard.
|
||||
|
||||
Shared frontend primitives prevent pages from acquiring competing contracts. `page_header.js` owns page headers and tab strips; `page_icons.js` owns navigation/header icons; `api_client.js` owns browser API calls and typed error propagation; `api_types.js` mirrors the browser-facing contract shapes; `ui_helpers.js` owns shared status, safe-field, host-bridge, keyboard menu-lock suppression behavior, and the design-system action button for host-stamped system chat rows (`createSystemMessageAction`) (both top-level documents — the SPA and the onboarding wizard iframe — install its Alt guard on their own windows); `skill_card_renderer.js` owns installed-skill cards; `hub_sync.js` owns the one catalog×listing hub-card verdict (actions install/installed/update/adopt/wait_pr/none plus badges) consumed by the OuroborosHub tab and the My-skills hub badges — the hub tab joins the catalog with the global `/api/extensions` listing by canonical name and no longer reads the bucket-scoped installed endpoint; `client_surface.js` owns the send-time sending-surface snapshot (raw observables, no device taxonomy) that `chat.js` spreads into each chat frame; `log_events.js` owns event classification and the shared technical outcome reducers plus one factual task-presentation projection consumed by Chat and Logs. The projection translates task truth only into `Working` / `Done` / `Done with warnings` / `Failed` / `Cancelled`; it does not own actions, incidents, or notifications, and compact headlines never expose raw reason codes. `toast.js`, `masonry.js`, `widget_frame.js`, `widget_job.js`, and CSS tokens own common notifications, framed-widget bootstrap/lifecycle, bounded widget request/job policy, and layout; `task_control_menu.js` owns the S3 three-action task stop/hurry dropdown ("Wrap up" / "Hurry up" / "Stop now" — frozen owner wording) shared verbatim by Chat live cards and the Activity tab: eligibility gates differ per surface, but the actions, endpoint bindings (`stop_policy` mapping, stable per-task `request_id` retry), in-flight locking, and typed refusals do not; a pending cancel collapses the menu to the hard escalation only, dismissing the menu continues the run, and "Hurry up" acknowledges via LOCAL toast only — never a chat message (HQ1). The reason is dependency control: frontend work should not require reimplementing supervisor, review, marketplace, extension, and provider semantics in each page.
|
||||
Shared frontend primitives prevent pages from acquiring competing contracts. `page_header.js` owns page headers and tab strips; `page_icons.js` owns navigation/header icons; `api_client.js` owns browser API calls and typed error propagation; `api_types.js` mirrors the browser-facing contract shapes; `ui_helpers.js` owns shared status, safe-field, host-bridge, keyboard menu-lock suppression behavior, and the design-system action button for host-stamped system chat rows (`createSystemMessageAction`) (both top-level documents — the SPA and the onboarding wizard iframe — install its Alt guard on their own windows); `skill_card_renderer.js` owns installed-skill cards; `hub_sync.js` owns the one catalog×listing hub-card verdict (actions install/installed/update/adopt/wait_pr/none plus badges) consumed by the OuroborosHub tab and the My-skills hub badges — the hub tab joins the catalog with the global `/api/extensions` listing by canonical name and no longer reads the bucket-scoped installed endpoint; `client_surface.js` owns the send-time sending-surface snapshot (raw observables, no device taxonomy) that `chat.js` spreads into each chat frame; `log_events.js` owns event classification and the shared technical outcome reducers plus one factual task-presentation projection consumed by Chat and Logs. The projection translates task truth only into `Working` / `Done` / `Done with warnings` / `Failed` / `Cancelled`; it does not own actions, incidents, or notifications, and compact headlines never expose raw reason codes. `toast.js`, `masonry.js`, `widget_frame.js`, `widget_job.js`, and CSS tokens own common notifications, framed-widget bootstrap/lifecycle, bounded widget request/job policy, and layout; `task_control_menu.js` owns the S3 three-action task stop/hurry dropdown ("Wrap up" / "Hurry up" / "Stop now" — frozen owner wording) shared verbatim by Chat live cards and the Activity tab: eligibility gates differ per surface, but the actions, endpoint bindings (`stop_policy` mapping, stable per-task `request_id` retry), in-flight locking, and typed refusals do not; a pending cancel collapses the menu to the hard escalation only, dismissing the menu continues the run, and "Hurry up" acknowledges via LOCAL toast only — never a chat message (HQ1); because both consumers live inside clipped or scrolling containers, the temporary menu is page-owned, uses viewport-fixed flip/clamp placement, closes on ancestor/page scroll, window resize, or trigger visibility loss, and disposes its listeners and observer with the body portal instead of weakening container overflow. The reason is dependency control: frontend work should not require reimplementing supervisor, review, marketplace, extension, and provider semantics in each page.
|
||||
|
||||
`confirm_dialog.js::openConfirmDialog` is the one browser-dialog authority. Confirm mode resolves a strict boolean; input mode resolves `{confirmed, value}` and returns an empty value on cancellation; alert mode renders one acknowledgement button. Cancel, Close, backdrop, Escape, and supersession by a newer dialog all resolve as non-confirmation. Native `window.prompt`, `window.confirm`, and `window.alert` are forbidden in `web/modules`: they are visually and behaviorally inconsistent across shells, block the browser event loop, and `window.prompt` silently returns `null` in the macOS PyWebView shell because that backend has no prompt delegate. Critical controls therefore act only on the exact confirmed result; Panic's confirm-and-send sequence is one testable operation rather than a confirmation call detached from the command it guards.
|
||||
|
||||
|
|
@ -774,9 +775,11 @@ Each provider card has one compact **Test** action backed by `POST /api/provider
|
|||
|
||||
Desktop onboarding and the blocking web overlay are the same served `/onboarding` page — same provider, agents, model, review/runtime, budget, and summary steps, same backend normalization; context mode remains a separate owner setting. Startup readiness is structural: a recognized non-empty remote configuration or a task-capable local-routing flag is sufficient. An agent subscription strengthens Ouroboros but never satisfies that gate on its own. Credential validity, entitlement, model availability, and local-process health remain runtime status, not onboarding admission. Linux browser fallback therefore does not maintain a second setup flow. Every host completes through the single `POST /api/onboarding/complete` transaction described in the Startup / Onboarding Flow section, so a completed onboarding is all-or-nothing and install-time agent defaults are part of the same save.
|
||||
|
||||
Accounts is the owner-facing projection of Ouroboros's owned Claudexor daemon. The browser never receives its control token or interprets credentials. Status combines daemon/runtime readiness, login-capable harness discovery, credential profiles, honest vendor-live versus local-session verification, fresh quota windows, and optional model discovery. API-key-only adapters do not acquire fake Login buttons merely because they appear in a broader execution catalogue.
|
||||
Accounts is the owner-facing projection of Ouroboros's owned Claudexor daemon. The browser never receives its control token or interprets credentials. Status combines daemon/runtime readiness, login-capable harness discovery, credential profiles, honest vendor-live versus local-session verification, fresh quota windows, and optional model discovery. One `/v2/quota` envelope supplies both windows and typed per-subject absences to the status projection (`quota`, `quota_absences`), so a missing usage reading remains separate from login truth and route health never mixes two quota reads. API-key-only adapters do not acquire fake Login buttons merely because they appear in a broader execution catalogue.
|
||||
|
||||
Accounts are grouped into one card per agent family. Each card header carries the family name, an aggregate status that counts the accounts rotation can actually use — signed-in AND enabled, with all-disabled its own state — a fail-safe "Next up" badge naming who an unpinned run would take (read through the store's one dual-wire reader: the unified engine's `accountPools` first, the legacy per-harness `next_up` second; unknown kinds render as unknown, never a crash), and that card's own add action. Rows are ONE type on both engine generations: on a UNIFIED engine (the server-stamped `unified_accounts` feature fact) every account — migrated default logins included, under the reserved `<harness>-default` registry ids — is a named row carrying the same name, Enabled toggle (the engine's own per-profile PATCH) and Remove; on a LEGACY engine the native pseudo-row keeps the same two-line layout, is named by the identity the daemon observed (or "Default account"), and only its ACTIONS differ — no Remove and no toggle, because that engine has no route for either and a dead button would claim an effect this process cannot have. A row's first line is the account and its status; `verification=not_run` is neutral unknown/not verified while `failed` remains an error. When the engine explicitly reports `availability=unknown` with `verification=not_run`, the row says "Login status unknown" and its action re-runs the shared Refresh instead of starting a new sign-in; an auth probe failure is never treated as proof of logout, and the action remains available if the next read needs a real login. The second line is muted metadata in human words, including humanized quota (a migrated default row may inherit its pre-migration legacy-keyed window until the next refresh re-keys it; the exact subject always wins), a disabled row's own exclusion from rotation, and the humanized time of the last verification rather than a raw instant. Two migration-window residuals are accepted rather than patched: the legacy ''-keyed quota alias is granted only to the literal reserved `<harness>-default` registry id — a collision-suffixed migrated row (e.g. `codex-default-2`) does not inherit the legacy window until the next quota refresh re-keys it, and a pre-existing unrelated row that happens to bear the reserved name could borrow it (exact-keyed readings always win) — and pinned route health applies no legacy alias at all, so immediately after migration a pinned default may read UNKNOWN and fail open to the engine's own authoritative typed refusal (consistent with the strict-pin decision D-U6). Removing a named account is a request to the daemon's own credential-profile contract, and the complete deletion receipt is preserved: a refusal remains a refusal, while the exact vendor-owned / left-unchanged / OS-user disposition becomes a successful retained-credential warning rather than a false sign-out claim. A single service banner at the top of the tab explains a daemon or runtime problem once, per facet, instead of decorating rows with unavailability claims. Per-facet independence — a refused quota read leaving the catalogue and account facets authoritative — is real on both sides of the wire: `claudexor_accounts.py` fans the catalog, account and quota reads out independently, classifies each on its own, and stamps the result into the payload's `reads` block, so one refusal no longer collapses its siblings into the global `unreachable` verdict; the client's shared status store reads that stamp through its one facet reader. Only a legacy payload without the stamp is still read coarsely — a global refusal makes every facet indeterminate together rather than one of them being blamed.
|
||||
Claudexor owns snapshot-versus-absence coverage at its response boundary. Ouroboros keeps only fresh snapshots eligible for percentages or exhaustion and reads typed absences by exact subject; a malformed optional absence list becomes empty rather than breaking the whole status response. Already-redacted absence detail is displayed as text but never selects semantics, and legacy subject aliases apply only to snapshots, never to credential absences.
|
||||
|
||||
Accounts are grouped into one card per agent family. Each card header carries the family name, an aggregate status that counts the accounts rotation can actually use — signed-in AND enabled, with all-disabled its own state — a fail-safe "Next up" badge naming who an unpinned run would take (read through the store's one dual-wire reader: the unified engine's `accountPools` first, the legacy per-harness `next_up` second; unknown kinds render as unknown, never a crash), and that card's own add action. Rows are ONE type on both engine generations: on a UNIFIED engine (the server-stamped `unified_accounts` feature fact) every account — migrated default logins included, under the reserved `<harness>-default` registry ids — is a named row carrying the same name, Enabled toggle (the engine's own per-profile PATCH) and Remove; on a LEGACY engine the native pseudo-row keeps the same two-line layout, is named by the identity the daemon observed (or "Default account"), and only its ACTIONS differ — no Remove and no toggle, because that engine has no route for either and a dead button would claim an effect this process cannot have. A row's first line is the account and its status; `verification=not_run` is neutral unknown/not verified while `failed` remains an error. When the engine explicitly reports `availability=unknown` with `verification=not_run`, the row says "Login status unknown" and its action re-runs the shared Refresh instead of starting a new sign-in; an auth probe failure is never treated as proof of logout, and the action remains available if the next read needs a real login. The second line is muted metadata in human words, including humanized quota (a migrated default row may inherit its pre-migration legacy-keyed window until the next refresh re-keys it; the exact subject always wins), a disabled row's own exclusion from rotation, and the humanized time of the last verification rather than a raw instant. The row also appends only the exact subject's typed quota absence: refresh/rate-limit/pacing gaps stay neutral and never offer login, genuine `not_logged_in`/`auth_revoked` remains distinct in words, unknown future reasons degrade to neutral usage-unavailable copy, and prose detail never selects semantics. Two migration-window residuals are accepted rather than patched: the legacy ''-keyed quota alias is granted only to the literal reserved `<harness>-default` registry id — a collision-suffixed migrated row (e.g. `codex-default-2`) does not inherit the legacy window until the next quota refresh re-keys it, and a pre-existing unrelated row that happens to bear the reserved name could borrow it (exact-keyed readings always win) — and pinned route health applies no legacy alias at all, so immediately after migration a pinned default may read UNKNOWN and fail open to the engine's own authoritative typed refusal (consistent with the strict-pin decision D-U6). Removing a named account is a request to the daemon's own credential-profile contract, and the complete deletion receipt is preserved: a refusal remains a refusal, while the exact vendor-owned / left-unchanged / OS-user disposition becomes a successful retained-credential warning rather than a false sign-out claim. A single service banner at the top of the tab explains a daemon or runtime problem once, per facet, instead of decorating rows with unavailability claims. Per-facet independence — a refused quota read leaving the catalogue and account facets authoritative — is real on both sides of the wire: `claudexor_accounts.py` fans the catalog, account and quota reads out independently, classifies each on its own, and stamps the result into the payload's `reads` block, so one refusal no longer collapses its siblings into the global `unreachable` verdict; the client's shared status store reads that stamp through its one facet reader. Only a legacy payload without the stamp is still read coarsely — a global refusal makes every facet indeterminate together rather than one of them being blamed.
|
||||
|
||||
Connect is link-first and harness-agnostic. A typed disclosure renders the sign-in URL and any one-time code; flows that may need a pasted callback code keep that optional field visible while active because the browser callback may complete without it. Current engines publish optional-without-default `setupLogin` on each exact harness row: `{mode: in_app}` maps an omitted browser request to an omitted setup transport, `{mode: external_terminal}` maps it to `client_pty`, and malformed present data is a capability gap. Explicit null is ambiguous while the vendor CLI is absent, so support is delegated to the exact pinned engine's typed setup/profile admission and the response is stamped `setup_job_admission`; an omitted transport stays omitted and explicit `client_pty` stays exact. An explicit `client_pty` recovery request remains explicit even when the normal mode is `in_app`; Codex pairs that transport with `browser_redirect`, and non-Codex requests never acquire `loginFlow`. Only genuine key absence on a legacy engine consults the older global operation signal, and the create response discloses that compatibility source instead of presenting it as per-harness host evidence. Typed `credential_profile_required` plus `add_named_account` selects the name-the-account face; unrelated 400/409 and prose never do. A typed duplicate profile is idempotent directly; a 3.6.0 generic 409 (`internal_error`, or the transport's `http_409` fallback) is idempotent only after the exact harness/profile row is read back. Typed pre-job refusals prove setup custody absent/released, while unmarked discovery and transport failures remain unknown. The exact `terminal_transport_unavailable`, `terminal_transport_unsupported`, `terminal_transport_probe_failed`, or `terminal_transport_failed` code plus its required action (pre-job) or durable `job.nativeCommand.errorCode` (post-create) offers an explicit external-terminal continuation through the same release guard; no prose selects it. Unsupported/unavailable does not repeat a retry the engine did not offer, while probe failure may offer both actions. If the exact pinned engine answers the synchronous first create with its structural pre-command missing-vendor-binary terminal job, the same owner action also consents to one hidden local install through the exact managed Claudexor CLI and one retry. No message text, harness name, PATH executable, system npm, later poll, or command-bearing `not_supported` job can trigger it. The installer is a hard-timeout new process group, stderr is discarded, and stdout is capped and parsed as exactly one strict JSON success object. Success requires Claudexor's post-install proof: an absolute non-empty `installedBinary` plus a non-empty `installedVersion` bounded to 256 characters; process exit zero without that proof is refused. A second login refusal is returned rather than looped. A newly created explicit `client_pty` job exposes its labelled POSIX-shell or PowerShell attach command immediately in both full and compact cards; ordinary delayed/legacy attach remains collapsed under Advanced in the full card. There is no embedded terminal login surface or cmd formatter. Terminal job state and the current account row are reconciled so a stale verification read during login cannot claim failure after the account actually connected.
|
||||
|
||||
|
|
@ -1178,6 +1181,8 @@ authority check in this change.
|
|||
|
||||
Every tool call first crosses deterministic `ToolRegistry` and resource-root guards; policy-based LLM safety is added where `OUROBOROS_SAFETY_MODE` requires it. The deterministic layers run in every safety mode. `runtime_mode_policy.py` owns protected self-repo paths, frozen contracts, release/build and managed-repo invariants. Light blocks Ouroboros self-repo and control-plane mutation, not normal user deliverables under `user_files`, `task_drive`, or `artifact_store`. Advanced may evolve ordinary app code; Pro may leave protected edits on disk, but publication still requires the reviewed commit path. Runtime mode is a self-modification boundary, not an OS sandbox.
|
||||
|
||||
Every deterministic write/owner-control guard consumes ONE mode-aware write-shape seam (`ouroboros/tools/write_shape.py`) — the registry no longer even imports the coarse legacy scan, so a guard structurally cannot judge on a coarser fact. Interpreter argv (including an `sh -c` wrap) takes `interpreter_write_shape`: a read-only `open(p, 'rb')` is not a write shape. Non-interpreter argv takes `non_interpreter_write_shape`: unconditional writers (cp/mv/rm/mkdir/touch/chmod/…) keep the membership floor, the pure-filter utilities (sort/uniq/sed/tar/gzip) are write-shaped only through a REAL channel (`sort -o` in both spellings, `sed -i` in any spelling PLUS sed's in-script `w`/`W`/`e` commands and `-f` script files — a script not provably free of those fails closed, exotic non-`/` substitute delimiters are the disclosed fail-open residual — a second uniq operand, tar create/extract, a redirect, a reported writer target), and the bare prose words ('delete'/'trash'/'truncate') yield to the same read-carve the owner-control detectors use (`_is_pure_read_inspection`, the v6.80.0 scope-floor contract applied family-wide) — a provably read-only `grep -n delete ouroboros/safety.py` reads, an unprovable head (`osascript -e 'delete …'`) stays fail-closed. The protected-core lane's mention branch consumes the same composed fact, so a pure read that merely mentions a protected filename is no longer refused as a "modification". The coarse bare `open(` token used to feed pure interpreter reads into the workspace write guard, which refused them with a false "write-like" reason and no route — the same class the light-mode runtime_data lane has re-judged since v6.54.3 ("the original GAIA class"). Write-mode opens (python `[wax+]` modes AND perl `'>'`/`'>>'` spellings), pathlib `.open('w')`, library save-APIs, ruby's `File.delete`/`FileUtils.*`/`IO.binwrite`, opaque subprocess/exec escapes, and every shell-level indicator (redirects — token-initial or glued into an operand, `tee`, writer utilities) still classify as writes, and literal write targets stay covered by `writer_target_tokens`; the disclosed residuals (`open(p, m)` with the mode in a variable; a writer reached through an alias the regex cannot follow, e.g. `from os import remove as delete`; parenless perl builtins such as `rename $a, $b`) are covered for external workspaces by the runtime/secret read guard below plus the LLM safety supervisor — chasing them with more spellings is the arms race BIBLE P5/P13 forbids. Workspace write-guard block messages name the resolved offending path and the sanctioned route (gated `read_file`/`write_file`, `root=skill_payload` with bucket/skill_name, or writing inside the selected process root) instead of one byte-identical reasonless string across five return sites.
|
||||
|
||||
Read-only shell git is allowed everywhere; mutating shell git is allowed only when its resolved target is outside the Ouroboros system repository and runtime data drives. The target-aware `git_shell_policy` enforces that boundary, while network-disabled tasks still fence network git operations. Acting `self_worktree` children remain read-only because patch capture requires an unmoved HEAD. `git init` and `git clone` are judged by their destination rather than the current directory, including relative destinations and path-valued retargeting flags. In external-workspace mode, the runtime/secret read guard exempts only an all-read-only git command: mixed shell segments, `--no-index`, or a nominally read-only command with a writing `--output` path lose the exemption. `resolve_shell_cwd` canonicalizes the cwd once and every guard consumes that same path. These composition rules preserve ordinary local git power without turning git into a runtime-data read or write escape.
|
||||
|
||||
The generic Tool API VCS family (`vcs_status`, `vcs_diff`, `vcs_pull_ff`, `vcs_restore`, `vcs_revert`) defaults to `root=active_workspace` and accepts explicit `root=system_repo`; every result names the logical root and physical repository. Protected Ouroboros path names constrain generic restore/revert only on the explicit system target, so a project's own `BIBLE.md` or `contracts/` remains ordinary project content. `preflight_review`, `commit_reviewed`/`vcs_commit_reviewed`, `vcs_rollback`, and promotion remain system-repository lifecycles even when the calling task is focused on a project.
|
||||
|
|
@ -2629,6 +2634,12 @@ never classified as a mask; `prepare_settings_for_persist()` applies the same
|
|||
top-level repair at the common writer boundary. Password, token, and MCP masks
|
||||
remain context-specific rather than sharing a suffix heuristic.
|
||||
|
||||
`ouroboros/openrouter_attribution.py` is the application-identity SSOT for every
|
||||
first-party paid OpenRouter request, including runtime, review probes, and benchmark
|
||||
diagnostics. Its canonical public URL is the primary OpenRouter application id and
|
||||
`X-OpenRouter-Title` supplies the display name. A fork or another product must use
|
||||
its own URL rather than sharing this identity and competing to rename one app record.
|
||||
|
||||
### LLM output token budgets
|
||||
|
||||
Ouroboros uses provider-specific names for the same output-token budget:
|
||||
|
|
|
|||
|
|
@ -1421,6 +1421,11 @@ Before every commit, verify the following:
|
|||
refusals are `route_not_in_capability_catalog`, `route_disabled` (unpinned),
|
||||
access-profile mismatch, `engine_rejects_delegated_marker`, and positive
|
||||
quota exhaustion for the route's own model.
|
||||
- Any reader that needs quota snapshots plus typed absences must call
|
||||
`ClaudexorGateway.quota_state()` once and project both from that envelope.
|
||||
The list helpers are compatibility projections, not permission to perform
|
||||
two `/v2/quota` reads and mix evidence epochs. Optional absence metadata
|
||||
from older engines fails to an empty neutral value.
|
||||
- The acceptance packet carries a host-attested `substrate_execution` section —
|
||||
`actual_substrate`, `delegated_runs_*` counters, zero-run facts — read from
|
||||
durable custody rows at packet-build time
|
||||
|
|
|
|||
|
|
@ -261,6 +261,7 @@ def _status_payload(include_models: bool) -> Dict[str, Any]:
|
|||
"harnesses": [],
|
||||
"profiles": {},
|
||||
"quota": [],
|
||||
"quota_absences": [],
|
||||
# PROVENANCE, per independent facet (BIBLE P1: missing data is a GAP,
|
||||
# never a value). `[]`/`{}` alone cannot say WHETHER the daemon was
|
||||
# asked: the owner's panel printed "no account connected" for three
|
||||
|
|
@ -299,6 +300,15 @@ def _status_payload(include_models: bool) -> Dict[str, Any]:
|
|||
with ClaudexorGateway(endpoint) as gateway:
|
||||
gateway.handshake()
|
||||
payload["daemon"]["engine_version"] = gateway.engine_version
|
||||
|
||||
def _quota_state() -> Dict[str, Any]:
|
||||
reader = getattr(gateway, "quota_state", None)
|
||||
if callable(reader):
|
||||
return reader()
|
||||
# Compatibility for old embedded gateway doubles. The shipped
|
||||
# gateway has quota_state, so the live status path always uses
|
||||
# one physical GET and one evidence epoch.
|
||||
return {"snapshots": gateway.quota_snapshots(), "absences": []}
|
||||
# The catalog, manifest, profile and quota reads are INDEPENDENT GETs
|
||||
# over one thread-safe httpx client, and each costs SECONDS daemon-side
|
||||
# (it probes the real coding-agent CLIs on every read: binary, version,
|
||||
|
|
@ -312,7 +322,7 @@ def _status_payload(include_models: bool) -> Dict[str, Any]:
|
|||
catalog_call = pool.submit(gateway.agent_capabilities)
|
||||
manifests_call = pool.submit(gateway.harnesses)
|
||||
profiles_call = pool.submit(gateway.credential_profiles)
|
||||
quota_call = pool.submit(gateway.quota_snapshots)
|
||||
quota_call = pool.submit(_quota_state)
|
||||
# Deferred lookup on purpose: the failure of THIS read (or a
|
||||
# transport double that lacks the method) must land inside the
|
||||
# future, where the absorbed fail-closed handling below owns it.
|
||||
|
|
@ -325,7 +335,11 @@ def _status_payload(include_models: bool) -> Dict[str, Any]:
|
|||
catalog_outcome = _facet_outcome(catalog_call, envelope=("harnesses",))
|
||||
profiles_outcome = _facet_outcome(
|
||||
profiles_call, envelope=("profiles", "harnessAccounts"))
|
||||
quota_outcome = _facet_outcome(quota_call)
|
||||
quota_outcome = _facet_outcome(
|
||||
quota_call,
|
||||
envelope=("snapshots",),
|
||||
list_fields=("snapshots",),
|
||||
)
|
||||
payload["reads"] = {
|
||||
"catalog": catalog_outcome[0],
|
||||
"accounts": profiles_outcome[0],
|
||||
|
|
@ -413,7 +427,17 @@ def _status_payload(include_models: bool) -> Dict[str, Any]:
|
|||
and str(w["profile"].get("harness_id") or "") in (capable | vouched)
|
||||
]
|
||||
payload["profiles"] = profiles
|
||||
payload["quota"] = quota_outcome[1] if quota_outcome[0] == READ_OK else []
|
||||
quota = quota_outcome[1] if quota_outcome[0] == READ_OK else {}
|
||||
raw_snapshots = quota.get("snapshots") if isinstance(quota, dict) else None
|
||||
raw_absences = quota.get("absences") if isinstance(quota, dict) else None
|
||||
payload["quota"] = [
|
||||
row for row in (raw_snapshots if isinstance(raw_snapshots, list) else [])
|
||||
if isinstance(row, dict)
|
||||
]
|
||||
payload["quota_absences"] = [
|
||||
row for row in (raw_absences if isinstance(raw_absences, list) else [])
|
||||
if isinstance(row, dict)
|
||||
]
|
||||
if first_error is not None:
|
||||
# At least one facet refused while others landed. The daemon is
|
||||
# disclosed as unreachable AND the surviving facets keep their
|
||||
|
|
@ -429,7 +453,9 @@ def _status_payload(include_models: bool) -> Dict[str, Any]:
|
|||
return payload
|
||||
|
||||
|
||||
def _facet_outcome(call: "Future", *, envelope: tuple = ()) -> tuple:
|
||||
def _facet_outcome(
|
||||
call: "Future", *, envelope: tuple = (), list_fields: tuple = (),
|
||||
) -> tuple:
|
||||
"""Classify ONE fanned-out read: (state, value, error).
|
||||
|
||||
Independent by construction — a sibling's exception can never downgrade a
|
||||
|
|
@ -450,9 +476,9 @@ def _facet_outcome(call: "Future", *, envelope: tuple = ()) -> tuple:
|
|||
transport and arrives intact. Either would otherwise be published as an
|
||||
AUTHORITATIVE empty — exactly the lie the read block exists to stop — and
|
||||
both land on the same verdict here. An envelope carrying none of its keys is
|
||||
a read that did not answer, not an account store that is empty. Quota passes
|
||||
``()``: its reader already filters to a list of rows, and an empty quota
|
||||
renders as a neutral absence rather than a verdict.
|
||||
a read that did not answer, not an account store that is empty. Quota requires
|
||||
its canonical ``snapshots`` member as an array while retaining the typed
|
||||
absences from that same evidence envelope.
|
||||
"""
|
||||
from ouroboros.gateways.claudexor import ClaudexorUnavailable
|
||||
|
||||
|
|
@ -464,6 +490,8 @@ def _facet_outcome(call: "Future", *, envelope: tuple = ()) -> tuple:
|
|||
return (READ_FAILED, None, None)
|
||||
if envelope and not (isinstance(value, dict) and all(key in value for key in envelope)):
|
||||
return (READ_FAILED, None, None)
|
||||
if list_fields and not all(isinstance(value.get(key), list) for key in list_fields):
|
||||
return (READ_FAILED, None, None)
|
||||
return (READ_OK, value, None)
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -1042,6 +1042,7 @@ class ClaudexorStatusResponse(TypedDict, total=False):
|
|||
harnesses: List[Dict[str, Any]]
|
||||
profiles: Dict[str, Any]
|
||||
quota: List[Dict[str, Any]]
|
||||
quota_absences: List[Dict[str, Any]]
|
||||
reads: ClaudexorStatusReads
|
||||
# UNIFIED ACCOUNT MODEL feature fact (additive-optional): True only when
|
||||
# the engine's own /v2/operations catalog was read and advertises
|
||||
|
|
|
|||
|
|
@ -60,14 +60,15 @@ def _build_model_catalog_entry(
|
|||
) -> dict[str, str]:
|
||||
raw_id = str(model_id or "").strip()
|
||||
name = str(display_name or "").strip() or raw_id
|
||||
source_label = source or provider_label
|
||||
return {
|
||||
"provider_id": provider_id,
|
||||
"provider": provider_label,
|
||||
"source": source or provider_label,
|
||||
"source": source_label,
|
||||
"id": raw_id,
|
||||
"name": name,
|
||||
"value": _tagged_model_value(provider_id, raw_id),
|
||||
"label": f"{provider_label} · {name}",
|
||||
"label": f"{source_label} · {name}",
|
||||
}
|
||||
|
||||
|
||||
|
|
@ -394,7 +395,7 @@ async def api_model_catalog(_request: Request) -> JSONResponse:
|
|||
seen_values.add(value)
|
||||
items.append(item)
|
||||
|
||||
items.sort(key=lambda item: (item.get("provider", "").lower(), item.get("label", "").lower()))
|
||||
items.sort(key=lambda item: (item.get("provider", "").lower(), item.get("name", "").lower()))
|
||||
return JSONResponse({
|
||||
"items": items,
|
||||
"errors": errors,
|
||||
|
|
|
|||
|
|
@ -423,20 +423,24 @@ class ClaudexorGateway:
|
|||
rows = body.get("harnesses") if isinstance(body, dict) else None
|
||||
return [row for row in (rows or []) if isinstance(row, dict)]
|
||||
|
||||
def quota_snapshots(self) -> List[Dict[str, Any]]:
|
||||
def quota_state(self) -> Dict[str, Any]:
|
||||
"""GET /v2/quota once, retaining its one-epoch evidence envelope."""
|
||||
body = self._request("GET", "/v2/quota")
|
||||
return body if isinstance(body, dict) else {}
|
||||
|
||||
def quota_snapshots(self) -> List[Dict[str, Any]]:
|
||||
body = self.quota_state()
|
||||
snapshots = body.get("snapshots") if isinstance(body, dict) else None
|
||||
return [row for row in (snapshots or []) if isinstance(row, dict)]
|
||||
|
||||
def quota_absences(self) -> List[Dict[str, Any]]:
|
||||
"""Profiles whose quota could NOT be read (a 429/failed refresh, no login).
|
||||
|
||||
A separate reader on purpose: exhaustion needs POSITIVE evidence, and an
|
||||
absence is the typed record that evidence is missing for a profile — the
|
||||
route-health predicate treats any absence on a route as "unknown, so
|
||||
usable" rather than letting the readable minority speak for the whole route.
|
||||
Legacy compatibility projection of the same one-epoch quota envelope.
|
||||
An absence is typed evidence that quota could not be read for a profile;
|
||||
route health treats that state as unknown and therefore fail-open.
|
||||
"""
|
||||
body = self._request("GET", "/v2/quota")
|
||||
body = self.quota_state()
|
||||
absences = body.get("absences") if isinstance(body, dict) else None
|
||||
return [row for row in (absences or []) if isinstance(row, dict)]
|
||||
|
||||
|
|
|
|||
|
|
@ -14,7 +14,6 @@ import threading
|
|||
import time
|
||||
from typing import Any, Dict, List, Optional, Set, Tuple
|
||||
|
||||
from ouroboros.provider_models import OPENROUTER_DEFAULTS, PROVIDER_PREFIXES, normalize_anthropic_model_id, normalize_model_identity, resolve_minimax_base_url
|
||||
from ouroboros.anthropic_native_custody import (
|
||||
anthropic_replay_scoped,
|
||||
custody_private_key,
|
||||
|
|
@ -24,6 +23,8 @@ from ouroboros.anthropic_native_custody import (
|
|||
retain_native_assistant_content,
|
||||
scrub_native_custody,
|
||||
)
|
||||
from ouroboros.openrouter_attribution import OPENROUTER_APP_HEADERS
|
||||
from ouroboros.provider_models import OPENROUTER_DEFAULTS, PROVIDER_PREFIXES, normalize_anthropic_model_id, normalize_model_identity, resolve_minimax_base_url
|
||||
from ouroboros.request_wire_recovery import (
|
||||
finalize_wire_response,
|
||||
note_provider_metadata_drop_fields,
|
||||
|
|
@ -1434,10 +1435,7 @@ class LLMClient:
|
|||
"usage_model": usage_model,
|
||||
"api_key": current_api_key,
|
||||
"base_url": "https://openrouter.ai/api/v1" if explicit_settings else self._base_url,
|
||||
"default_headers": {
|
||||
"HTTP-Referer": "https://ouroboros.local/",
|
||||
"X-Title": "Ouroboros",
|
||||
},
|
||||
"default_headers": dict(OPENROUTER_APP_HEADERS),
|
||||
"supports_openrouter_extensions": True,
|
||||
"supports_generation_cost": True,
|
||||
}
|
||||
|
|
@ -4358,6 +4356,7 @@ def openrouter_web_search_server_tool(
|
|||
client_kwargs: Dict[str, Any] = dict(
|
||||
api_key=api_key,
|
||||
base_url="https://openrouter.ai/api/v1",
|
||||
default_headers=dict(OPENROUTER_APP_HEADERS),
|
||||
max_retries=0,
|
||||
)
|
||||
if timeout is not None:
|
||||
|
|
|
|||
10
ouroboros/openrouter_attribution.py
Normal file
10
ouroboros/openrouter_attribution.py
Normal file
|
|
@ -0,0 +1,10 @@
|
|||
"""Canonical OpenRouter application attribution for Ouroboros traffic."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
OPENROUTER_APP_URL = "https://ouroboros-agent.ai/"
|
||||
OPENROUTER_APP_TITLE = "Ouroboros"
|
||||
OPENROUTER_APP_HEADERS = {
|
||||
"HTTP-Referer": OPENROUTER_APP_URL,
|
||||
"X-OpenRouter-Title": OPENROUTER_APP_TITLE,
|
||||
}
|
||||
|
|
@ -970,9 +970,11 @@ def node_distribution_platform() -> str:
|
|||
|
||||
def probe_node_version(node_path: str) -> str:
|
||||
"""Return a normalized bundled-Node version, or ``""`` on probe failure."""
|
||||
# An inherited NODE_OPTIONS carrying test-mode flags (--test-name-pattern,
|
||||
# --test-only) makes `node --version` itself exit non-zero, so a healthy
|
||||
# runtime read as missing. The probe scrubs it exactly as the suite run does.
|
||||
# A metadata probe must not inherit runtime/test hooks. In particular,
|
||||
# NODE_OPTIONS can contain test filters or preload modules that either make
|
||||
# `node --version` fail before the hermetic lane gets a chance to scrub the
|
||||
# variable or execute arbitrary operator code during a supposedly inert
|
||||
# version check.
|
||||
probe_env = dict(os.environ)
|
||||
probe_env.pop("NODE_OPTIONS", None)
|
||||
try:
|
||||
|
|
|
|||
|
|
@ -159,7 +159,6 @@ BAND_PATHS = {
|
|||
"ouroboros/task_status.py": None,
|
||||
"ouroboros/tools/browser.py": None,
|
||||
"ouroboros/tools/plan_review_runtime.py": "Entered the band from 986 lines: timeout custody synthesis joined the existing plan-review runtime owner while preserving profile-continuity disclosures and typed health facts during target integration.",
|
||||
"ouroboros/tools/shell_guards.py": None,
|
||||
"ouroboros/tools/skill_exec.py": None,
|
||||
"ouroboros/tools/skill_publish.py": "Entered the band from 952 lines: publish now writes the OuroborosHub publication receipt at pr_opened through the shared locked-update seam and maps the receipt from the validated serialized form (hubflow sprint, receipt-as-only-stored-fact design).",
|
||||
"ouroboros/utils.py": None,
|
||||
|
|
|
|||
|
|
@ -571,10 +571,30 @@ def _exhausted_window(gateway: Any, route_id: str, route_model: str = "",
|
|||
return False
|
||||
return not pinned or str(subject.get("subject_id") or "") == pinned
|
||||
|
||||
quota_state = getattr(gateway, "quota_state", None)
|
||||
if callable(quota_state):
|
||||
envelope = quota_state()
|
||||
snapshots = envelope.get("snapshots") if isinstance(envelope, dict) else []
|
||||
absences = envelope.get("absences") if isinstance(envelope, dict) else []
|
||||
else:
|
||||
# Compatibility for older gateway doubles/embedders. Production
|
||||
# ClaudexorGateway owns quota_state and therefore performs one GET.
|
||||
snapshots = gateway.quota_snapshots()
|
||||
absence_reader = getattr(gateway, "quota_absences", None)
|
||||
absences = absence_reader() if callable(absence_reader) else []
|
||||
|
||||
# The current daemon schema guarantees arrays, but route health also supports
|
||||
# older embedders and test doubles. A malformed mandatory snapshot collection
|
||||
# is unknown (fail-open); malformed optional absences add no evidence.
|
||||
snapshots = snapshots if isinstance(snapshots, list) else []
|
||||
absences = absences if isinstance(absences, list) else []
|
||||
|
||||
resets: List[str] = []
|
||||
any_live = False
|
||||
any_spent = False
|
||||
for snapshot in gateway.quota_snapshots():
|
||||
for snapshot in snapshots or []:
|
||||
if not isinstance(snapshot, dict):
|
||||
continue
|
||||
subject = snapshot.get("subject") if isinstance(snapshot.get("subject"), dict) else {}
|
||||
if not _subject_matches(subject):
|
||||
continue
|
||||
|
|
@ -596,12 +616,16 @@ def _exhausted_window(gateway: Any, route_id: str, route_model: str = "",
|
|||
any_live = True
|
||||
if any_live or not any_spent:
|
||||
return False, ""
|
||||
absences = getattr(gateway, "quota_absences", None)
|
||||
if callable(absences):
|
||||
for row in absences() or []:
|
||||
subject = row.get("subject") if isinstance(row, dict) else None
|
||||
if isinstance(subject, dict) and _subject_matches(subject):
|
||||
return False, ""
|
||||
for row in absences or []:
|
||||
subject = row.get("subject") if isinstance(row, dict) else None
|
||||
if not isinstance(subject, dict) or not _subject_matches(subject):
|
||||
continue
|
||||
# Any explicit gap keeps a spent-looking route fail-open. The shipped
|
||||
# producer already removes absences covered by a snapshot for the same
|
||||
# quota subject; retaining the conservative check here also keeps old
|
||||
# gateway doubles and malformed future envelopes from authorizing a
|
||||
# fallback on contradictory evidence.
|
||||
return False, ""
|
||||
return True, min(resets) if resets else ""
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -38,13 +38,13 @@ from ouroboros.shell_parse import (
|
|||
unwrap_env_argv,
|
||||
)
|
||||
from ouroboros.tools.shell_guards import (
|
||||
LIGHT_SHELL_WRITER_COMMANDS,
|
||||
PROTECTED_RUNTIME_PATHS_LOWER,
|
||||
interpreter_family,
|
||||
interpreter_write_shape,
|
||||
light_shell_repo_mutation,
|
||||
non_interpreter_write_shape,
|
||||
parse_porcelain_paths,
|
||||
process_shell_guard_args,
|
||||
shell_has_write_indicator,
|
||||
runtime_data_guard_targets,
|
||||
shell_writer_targets_protected,
|
||||
workspace_executor_state_write_block,
|
||||
|
|
@ -164,12 +164,34 @@ def _executor_backend_candidate_path(ctx: Any, candidate: str) -> pathlib.Path |
|
|||
return None
|
||||
|
||||
|
||||
def _detect_runtime_mode_elevation(text_lower: str) -> bool:
|
||||
def _owner_control_mention_blocks(text_lower: str, detected: bool, writeish: bool) -> bool:
|
||||
"""Shared read-carve for the owner-control mention detectors.
|
||||
|
||||
The scope-floor guard adjudicated this contract at v6.80.0
|
||||
(``_detect_scope_review_floor_self_lowering``): naming an owner key/endpoint
|
||||
blocks UNLESS the whole command line is demonstrably read-only inspection —
|
||||
``grep OUROBOROS_RUNTIME_MODE data/settings.json`` and
|
||||
``rg /api/owner/safety-mode ouroboros/gateway`` read and do not act, and the
|
||||
product's own reuse-first duty (grep callers of ``save_settings``) depends on
|
||||
them. The other six family members stayed read-blind, blocking those exact
|
||||
inspections in every runtime mode — the same hazard class at a different
|
||||
strictness. Fail-closed like the precedent: ``writeish`` (any write shape)
|
||||
disqualifies the exemption, ``_is_pure_read_inspection`` is a HEAD allowlist
|
||||
where any interpreter, HTTP client, wrapper-with-flags, or nested execution
|
||||
is NOT an inspection, and the default ``writeish=True`` keeps a caller that
|
||||
cannot supply the fact fail-closed."""
|
||||
if not detected:
|
||||
return False
|
||||
return writeish or not _is_pure_read_inspection(text_lower)
|
||||
|
||||
|
||||
def _detect_runtime_mode_elevation(text_lower: str, *, writeish: bool = True) -> bool:
|
||||
"""Detect shell/script attempts to change ``OUROBOROS_RUNTIME_MODE``."""
|
||||
has_save = "save_settings" in text_lower
|
||||
has_mode_key = "ouroboros_runtime_mode" in text_lower
|
||||
has_dotted_path = "ouroboros.config.save_settings" in text_lower
|
||||
return (has_save and has_mode_key) or has_dotted_path
|
||||
detected = (has_save and has_mode_key) or has_dotted_path
|
||||
return _owner_control_mention_blocks(text_lower, detected, writeish)
|
||||
|
||||
|
||||
_SUBAGENT_SHELL_SECRET_MARKERS = (
|
||||
|
|
@ -229,6 +251,36 @@ def _command_mentions_protected_root(cmd_path_lower: str, root_text: str) -> boo
|
|||
start = end
|
||||
|
||||
|
||||
def _workspace_write_block_runtime_message(path_text: Any = "") -> str:
|
||||
"""Guard-B block for a write-shaped command reaching a protected runtime root.
|
||||
|
||||
Names the resolved offending path and the sanctioned route (the light-lane
|
||||
message at the runtime_data guard is the in-repo exemplar): five return sites
|
||||
used to emit one byte-identical reasonless string, so the log could not even
|
||||
say WHICH path fired, and the agent had no route to self-correct.
|
||||
"""
|
||||
path_note = f" Blocked path: {path_text}." if str(path_text or "").strip() else ""
|
||||
return (
|
||||
"⚠️ WORKSPACE_SHELL_BLOCKED: write-like shell command mentions Ouroboros system/data paths."
|
||||
+ path_note
|
||||
+ " Use the gated read_file/write_file tools for runtime data (installed skill"
|
||||
" payloads: root=skill_payload with bucket/skill_name, or run the command with"
|
||||
" cwd=skill_payload), and keep shell writes inside the selected process root."
|
||||
)
|
||||
|
||||
|
||||
def _workspace_write_block_outside_root_message(path_text: Any = "", work_dir: Any = "") -> str:
|
||||
"""Guard-B block for a write-shaped command targeting outside the process root."""
|
||||
path_note = f" Blocked path: {path_text}." if str(path_text or "").strip() else ""
|
||||
root_note = f" Selected process root: {work_dir}." if str(work_dir or "").strip() else ""
|
||||
return (
|
||||
"⚠️ WORKSPACE_SHELL_BLOCKED: write-like shell commands may not target paths"
|
||||
" outside the selected process root." + path_note + root_note
|
||||
+ " Write inside the process root, or use the file tools with an explicit root"
|
||||
" (user_files / task_drive / artifact_store)."
|
||||
)
|
||||
|
||||
|
||||
def _stray_skill_payload_failsoft(root_arg: str, workspace_mode: bool, task_constraint: Any) -> bool:
|
||||
"""Whether stray bucket/skill_name on a write tool should be DROPPED rather than
|
||||
surfaced as SKILL_PAYLOAD_ARG_ERROR. Fail-soft ONLY for a WORKSPACE edit that is
|
||||
|
|
@ -242,7 +294,7 @@ def _stray_skill_payload_failsoft(root_arg: str, workspace_mode: bool, task_cons
|
|||
return bool(workspace_mode and not skill_payload_intent)
|
||||
|
||||
|
||||
def _detect_mutative_toggle_self_change(text_lower: str) -> bool:
|
||||
def _detect_mutative_toggle_self_change(text_lower: str, *, writeish: bool = True) -> bool:
|
||||
"""Detect shell/script/CLI attempts to change the owner-only mutative-subagents toggle."""
|
||||
has_key = "ouroboros_allow_mutative_subagents" in text_lower
|
||||
has_write = (
|
||||
|
|
@ -252,7 +304,7 @@ def _detect_mutative_toggle_self_change(text_lower: str) -> bool:
|
|||
or "settings set" in text_lower # `ouroboros settings set <key> <value>` CLI path
|
||||
or "ouroboros.cli" in text_lower
|
||||
)
|
||||
return has_key and has_write
|
||||
return _owner_control_mention_blocks(text_lower, has_key and has_write, writeish)
|
||||
|
||||
|
||||
def _managed_update_code_tool_block(ctx: Any, name: str) -> str:
|
||||
|
|
@ -322,7 +374,7 @@ def _authorized_managed_update_resolver(ctx: Any) -> bool:
|
|||
return False
|
||||
|
||||
|
||||
def _detect_evolution_owner_control_self_change(text_lower: str) -> bool:
|
||||
def _detect_evolution_owner_control_self_change(text_lower: str, *, writeish: bool = True) -> bool:
|
||||
"""Detect shell/script/CLI attempts to set the owner-only self-evolution controls:
|
||||
the post-task evolution toggle OR the persistent evolution-objective steer (which
|
||||
biases every evolution campaign, so it is owner-only like the toggle)."""
|
||||
|
|
@ -337,10 +389,10 @@ def _detect_evolution_owner_control_self_change(text_lower: str) -> bool:
|
|||
or "settings set" in text_lower
|
||||
or "ouroboros.cli" in text_lower
|
||||
)
|
||||
return has_key and has_write
|
||||
return _owner_control_mention_blocks(text_lower, has_key and has_write, writeish)
|
||||
|
||||
|
||||
def _detect_context_mode_self_lowering(text_lower: str) -> bool:
|
||||
def _detect_context_mode_self_lowering(text_lower: str, *, writeish: bool = True) -> bool:
|
||||
"""Detect shell/script attempts to lower the owner-controlled context mode."""
|
||||
mentions_context_key = "ouroboros_context_mode" in text_lower
|
||||
mentions_owner_endpoint = "/api/owner/context-mode" in text_lower
|
||||
|
|
@ -351,13 +403,14 @@ def _detect_context_mode_self_lowering(text_lower: str) -> bool:
|
|||
)
|
||||
mentions_save = "save_settings" in text_lower or "settings.json" in text_lower
|
||||
mentions_owner_lowering_flag = "allow_context_lowering" in text_lower
|
||||
return (
|
||||
detected = (
|
||||
mentions_owner_endpoint
|
||||
or mentions_context_endpoint
|
||||
or mentions_context_cli
|
||||
or mentions_owner_lowering_flag
|
||||
or (mentions_context_key and mentions_save)
|
||||
)
|
||||
return _owner_control_mention_blocks(text_lower, detected, writeish)
|
||||
|
||||
|
||||
# Commands that can only READ. This is an ALLOWLIST on purpose: an unrecognised
|
||||
|
|
@ -397,6 +450,10 @@ _DENIED_READ_OPTIONS: dict = {
|
|||
"ripgrep": _SEARCH_TOOL_EXEC_OPTIONS,
|
||||
"ag": _SEARCH_TOOL_EXEC_OPTIONS,
|
||||
"ack": _SEARCH_TOOL_EXEC_OPTIONS,
|
||||
# yq edits the named file in place with -i/--inplace; without this the family
|
||||
# read-carve exempted `yq -i '.OUROBOROS_SAFETY_MODE="off"' settings.json` as
|
||||
# "pure inspection" (jq has no in-place edit and stays a stdout-only read).
|
||||
"yq": frozenset({"-i", "--inplace"}),
|
||||
"sort": frozenset({"-o", "--output", "--compress-program"}),
|
||||
"less": frozenset({"-o", "--log-file", "-k", "--lesskey-file"}),
|
||||
"more": frozenset({"-o"}),
|
||||
|
|
@ -501,6 +558,10 @@ def _is_pure_read_inspection(text_lower: str) -> bool:
|
|||
denied = _DENIED_READ_OPTIONS.get(head)
|
||||
if denied and any(_denied_read_option(token, denied) for token in tokens[1:]):
|
||||
return False
|
||||
if head == "uniq" and sum(1 for t in tokens[1:] if t == "-" or not t.startswith("-")) >= 2:
|
||||
# uniq's SECOND positional operand is its output file ('-' is the
|
||||
# stdin operand, not a flag): `... | uniq - settings.json` writes.
|
||||
return False
|
||||
return True
|
||||
|
||||
|
||||
|
|
@ -564,7 +625,7 @@ def _compose_execute_result(result: str, route_note: str, safety_msg: str) -> st
|
|||
return result
|
||||
|
||||
|
||||
def _detect_safety_mode_self_lowering(text_lower: str) -> bool:
|
||||
def _detect_safety_mode_self_lowering(text_lower: str, *, writeish: bool = True) -> bool:
|
||||
"""Detect shell/script attempts to lower the owner-controlled LLM-safety coverage
|
||||
(v6.54.3). OUROBOROS_SAFETY_MODE gates the LLM safety supervisor layer; the agent
|
||||
must not switch it to ``light``/``off`` to remove its own supervision friction
|
||||
|
|
@ -574,24 +635,25 @@ def _detect_safety_mode_self_lowering(text_lower: str) -> bool:
|
|||
# URL-decode so a percent-encoded endpoint (curl .../safety%2Dmode) cannot
|
||||
# slip the literal scan (review round 6; mirrors the attest-endpoint guard).
|
||||
decoded = urllib.parse.unquote(urllib.parse.unquote(text_lower)).lower()
|
||||
text_lower = f"{text_lower} {decoded}"
|
||||
mentions_mode_key = "ouroboros_safety_mode" in text_lower
|
||||
mentions_owner_endpoint = "/api/owner/safety-mode" in text_lower
|
||||
mentions_mode_endpoint = "safety-mode" in text_lower and "/api/owner" in text_lower
|
||||
mentions_mode_cli = "safety-mode" in text_lower and (
|
||||
"ouroboros settings" in text_lower
|
||||
or "ouroboros.cli" in text_lower
|
||||
text = f"{text_lower} {decoded}"
|
||||
mentions_mode_key = "ouroboros_safety_mode" in text
|
||||
mentions_owner_endpoint = "/api/owner/safety-mode" in text
|
||||
mentions_mode_endpoint = "safety-mode" in text and "/api/owner" in text
|
||||
mentions_mode_cli = "safety-mode" in text and (
|
||||
"ouroboros settings" in text
|
||||
or "ouroboros.cli" in text
|
||||
)
|
||||
mentions_save = "save_settings" in text_lower or "settings.json" in text_lower or "/api/settings" in text_lower
|
||||
return (
|
||||
mentions_save = "save_settings" in text or "settings.json" in text or "/api/settings" in text
|
||||
detected = (
|
||||
mentions_owner_endpoint
|
||||
or mentions_mode_endpoint
|
||||
or mentions_mode_cli
|
||||
or (mentions_mode_key and mentions_save)
|
||||
)
|
||||
return _owner_control_mention_blocks(text_lower, detected, writeish)
|
||||
|
||||
|
||||
def _detect_owner_skill_attest_self_call(text_lower: str) -> bool:
|
||||
def _detect_owner_skill_attest_self_call(text_lower: str, *, writeish: bool = True) -> bool:
|
||||
"""Detect agent attempts to loopback-call the OWNER-ONLY skill owner-attestation endpoint
|
||||
(C1, v6.39). Owner-attestation skips the expensive LLM skill review; it MUST be
|
||||
owner-issued, never agent self-callable — otherwise the agent could self-bypass the
|
||||
|
|
@ -603,7 +665,8 @@ def _detect_owner_skill_attest_self_call(text_lower: str) -> bool:
|
|||
import urllib.parse
|
||||
decoded = urllib.parse.unquote(urllib.parse.unquote(text_lower)).lower()
|
||||
text = f"{text_lower} {decoded}"
|
||||
return "/api/owner/skills/" in text and "attest-review" in text
|
||||
detected = "/api/owner/skills/" in text and "attest-review" in text
|
||||
return _owner_control_mention_blocks(text_lower, detected, writeish)
|
||||
|
||||
|
||||
def _task_constraint_path_allowed(path_text: str, constraint: Optional[TaskConstraint], drive_root: pathlib.Path) -> bool:
|
||||
|
|
@ -2269,7 +2332,11 @@ class ToolRegistry:
|
|||
)
|
||||
if targets_system:
|
||||
for cf in PROTECTED_RUNTIME_PATHS_LOWER:
|
||||
if cf in cmd_path_lower and shell_has_write_indicator(raw_cmd):
|
||||
# The MODE-AWARE composition fact, not the coarse legacy scan: a
|
||||
# pure read that merely mentions a protected name (`grep -n delete
|
||||
# ouroboros/safety.py`, `sed -n 1,40p BIBLE.md`, a python open('r'))
|
||||
# is not a modification; every write shape still blocks here.
|
||||
if cf in cmd_path_lower and writeish:
|
||||
return (
|
||||
"⚠️ CRITICAL SAFETY_VIOLATION: Shell command would modify "
|
||||
"a protected core/contract/release file. Protected: "
|
||||
|
|
@ -2658,7 +2725,7 @@ class ToolRegistry:
|
|||
_command_mentions_protected_root(cmd_path_lower, text)
|
||||
for text in allowed_texts
|
||||
):
|
||||
return "⚠️ WORKSPACE_SHELL_BLOCKED: write-like shell command mentions Ouroboros system/data paths."
|
||||
return _workspace_write_block_runtime_message(root_path)
|
||||
path_tokens = list(shell_argv_with_path_tokens(raw_cmd))
|
||||
path_tokens.extend(
|
||||
token
|
||||
|
|
@ -2702,7 +2769,7 @@ class ToolRegistry:
|
|||
for protected_path in protected_paths:
|
||||
try:
|
||||
mapped_executor.relative_to(protected_path)
|
||||
return "⚠️ WORKSPACE_SHELL_BLOCKED: write-like shell command mentions Ouroboros system/data paths."
|
||||
return _workspace_write_block_runtime_message(mapped_executor)
|
||||
except Exception:
|
||||
pass
|
||||
if _executor_backend_candidate_allowed(
|
||||
|
|
@ -2742,11 +2809,11 @@ class ToolRegistry:
|
|||
for protected_path in protected_paths:
|
||||
try:
|
||||
resolved.relative_to(protected_path)
|
||||
return "⚠️ WORKSPACE_SHELL_BLOCKED: write-like shell command mentions Ouroboros system/data paths."
|
||||
return _workspace_write_block_runtime_message(resolved)
|
||||
except Exception:
|
||||
pass
|
||||
if not pro_workspace_passthrough:
|
||||
return "⚠️ WORKSPACE_SHELL_BLOCKED: write-like shell commands may not target paths outside the selected process root."
|
||||
return _workspace_write_block_outside_root_message(resolved, work_dir)
|
||||
continue
|
||||
deliverables_decision = _deliverables_target_decision(pathlib.Path(candidate))
|
||||
if deliverables_decision is not None:
|
||||
|
|
@ -2759,9 +2826,9 @@ class ToolRegistry:
|
|||
continue
|
||||
for protected_path in protected_paths:
|
||||
if path_text_is_inside(candidate, protected_path):
|
||||
return "⚠️ WORKSPACE_SHELL_BLOCKED: write-like shell command mentions Ouroboros system/data paths."
|
||||
return _workspace_write_block_runtime_message(candidate)
|
||||
if not pro_workspace_passthrough:
|
||||
return "⚠️ WORKSPACE_SHELL_BLOCKED: write-like shell commands may not target paths outside the selected process root."
|
||||
return _workspace_write_block_outside_root_message(candidate, work_dir)
|
||||
continue
|
||||
resolved = (work_dir / pathlib.Path(candidate)).resolve(strict=False)
|
||||
# The lexical relative spelling is authoritative for detecting
|
||||
|
|
@ -2781,11 +2848,11 @@ class ToolRegistry:
|
|||
for protected_path in protected_paths:
|
||||
try:
|
||||
resolved.relative_to(protected_path)
|
||||
return "⚠️ WORKSPACE_SHELL_BLOCKED: write-like shell command mentions Ouroboros system/data paths."
|
||||
return _workspace_write_block_runtime_message(resolved)
|
||||
except Exception:
|
||||
pass
|
||||
if not pro_workspace_passthrough:
|
||||
return "⚠️ WORKSPACE_SHELL_BLOCKED: write-like shell commands may not target paths outside the selected process root."
|
||||
return _workspace_write_block_outside_root_message(resolved, work_dir)
|
||||
return None
|
||||
|
||||
def _run_shell_safety_check(
|
||||
|
|
@ -2839,6 +2906,7 @@ class ToolRegistry:
|
|||
argv_for_write = argv
|
||||
argv_executable = pathlib.PurePath(argv_for_write[0]).name.lower().removesuffix(".exe") if argv_for_write else ""
|
||||
write_target_argvs = [argv_for_write] if argv_for_write else []
|
||||
inline_argv: list = []
|
||||
if argv_executable in {"sh", "bash", "zsh"}:
|
||||
inline_cmd = next((str(argv_for_write[idx + 1] or "") for idx, token in enumerate(argv_for_write[1:], start=1) if str(token or "") in {"-c", "--command"} and idx + 1 < len(argv_for_write)), "")
|
||||
if not inline_cmd:
|
||||
|
|
@ -2860,12 +2928,52 @@ class ToolRegistry:
|
|||
explicit_write_targets.append(
|
||||
destination.rstrip("/\\") + "/" + source_name
|
||||
)
|
||||
# A located -e/-E/-c inline CODE BODY is not a write target of the command
|
||||
# line: writer_target_tokens' generic fallback reports every non-flag operand
|
||||
# of a LIGHT_SHELL_WRITER_COMMANDS member (ruby/perl) — including the code
|
||||
# string itself — which made every one-liner write-shaped here. The light
|
||||
# fence and the protected lane keep consuming the unfiltered SSOT (their
|
||||
# pinned XG-7B3.1 contracts rely on the body operand); only THIS lane's
|
||||
# write-shape/target set drops the bodies. FILE operands stay write-suspect
|
||||
# (`perl -pi -e s/a/b/ file` rewrites `file`), and literal in-code targets
|
||||
# still arrive via the dedicated inline extraction.
|
||||
from ouroboros.tools.shell_guards import interpreter_inline_code as _interp_inline_code
|
||||
inline_code_bodies: set = set()
|
||||
for target_argv in write_target_argvs:
|
||||
inline_code_bodies.update(_interp_inline_code([str(t) for t in target_argv]))
|
||||
if inline_code_bodies:
|
||||
explicit_write_targets = [t for t in explicit_write_targets if t not in inline_code_bodies]
|
||||
explicit_write_targets = list(dict.fromkeys(explicit_write_targets))
|
||||
executable_path_tokens = {str(target_argv[0]) for target_argv in write_target_argvs if target_argv}
|
||||
# Writer-command membership canonicalizes versioned interpreter spellings to
|
||||
# their family (`ruby3.2` is `ruby`), so a versioned basename is exactly as
|
||||
# write-suspect as the unversioned one (XG-2R.2).
|
||||
writeish = shell_has_write_indicator(raw_cmd) or (bool(argv_for_write) and (interpreter_family(argv_executable) or argv_executable) in LIGHT_SHELL_WRITER_COMMANDS) or bool(explicit_write_targets)
|
||||
# Interpreter argv (a direct interpreter, or one inside an sh -c wrap) takes the
|
||||
# MODE-AWARE write-shape classifier: the coarse bare `open(` token classified a
|
||||
# read-only `open(p, 'rb')` as a write and fed pure reads into the workspace
|
||||
# write guard — the same class the light-mode runtime_data lane already
|
||||
# re-judges ("the original GAIA class"). Write-mode opens, pathlib `.open('w')`,
|
||||
# save-APIs, opaque subprocess escapes, and every shell-level indicator still
|
||||
# classify as writes; `writer_target_tokens` keeps covering literal write
|
||||
# targets via `explicit_write_targets` below.
|
||||
write_shape_interpreter = bool(interpreter_family(argv_executable)) or (
|
||||
bool(inline_argv)
|
||||
and bool(interpreter_family(pathlib.PurePath(str(inline_argv[0])).name.lower().removesuffix(".exe")))
|
||||
)
|
||||
# ONE mode-aware write-shape seam (write_shape.py) for BOTH halves:
|
||||
# interpreter argv takes interpreter_write_shape; everything else takes
|
||||
# non_interpreter_write_shape, where unconditional writers keep the
|
||||
# membership floor, pure-filter utilities (sort/uniq/sed/tar/gzip) need a
|
||||
# real write channel, and prose words yield to the same read-carve the
|
||||
# owner-control detectors use. No guard below consumes a coarser fact.
|
||||
coarse_write_shape = (
|
||||
interpreter_write_shape(raw_cmd)
|
||||
if write_shape_interpreter
|
||||
else non_interpreter_write_shape(
|
||||
raw_cmd, argv_for_write, argv_executable, is_pure_read=_is_pure_read_inspection,
|
||||
)
|
||||
)
|
||||
writeish = coarse_write_shape or bool(explicit_write_targets)
|
||||
work_dir = self._resolved_shell_cwd(args, binding)
|
||||
if isinstance(work_dir, str):
|
||||
return work_dir
|
||||
|
|
@ -2899,20 +3007,23 @@ class ToolRegistry:
|
|||
if workspace_write_block:
|
||||
return workspace_write_block
|
||||
|
||||
# Elevation pattern: blocked in all modes.
|
||||
if _detect_runtime_mode_elevation(cmd_lower):
|
||||
# Elevation pattern: blocked in all modes. Every owner-control mention
|
||||
# detector takes the shared read-carve (pure read-only inspection of the
|
||||
# key/endpoint names is allowed; the write shape or any non-inspection
|
||||
# head still blocks) — the scope-floor precedent applied family-wide.
|
||||
if _detect_runtime_mode_elevation(cmd_lower, writeish=writeish):
|
||||
return "⚠️ ELEVATION_BLOCKED: shell command pattern looks like an OUROBOROS_RUNTIME_MODE elevation attempt (mentions ``save_settings`` together with ``OUROBOROS_RUNTIME_MODE``, or invokes ``ouroboros.config.save_settings`` directly). Runtime mode is owner-controlled — change it by stopping the agent and editing settings.json directly, then restart."
|
||||
if _detect_context_mode_self_lowering(cmd_lower):
|
||||
if _detect_context_mode_self_lowering(cmd_lower, writeish=writeish):
|
||||
return "⚠️ CONTEXT_MODE_SELF_LOWERING_BLOCKED: shell command pattern looks like an attempt to lower OUROBOROS_CONTEXT_MODE to low through settings.json or /api/owner/context-mode. Context mode is owner-controlled — ask the owner to change the Low/Max toggle or edit settings while the agent is stopped."
|
||||
if _detect_scope_review_floor_self_lowering(cmd_lower, writeish=writeish):
|
||||
return "⚠️ SCOPE_REVIEW_FLOOR_SELF_LOWERING_BLOCKED: shell command pattern reaches OUROBOROS_SCOPE_REVIEW_FLOOR through settings.json, /api/settings, or /api/owner/scope-review-floor from something other than a pure read. The floor is a deprecated, enforcement-inert owner setting (BIBLE P3 scope-review applicability follows the owner context mode) — it stays owner-only, and the agent must not write owner settings through any channel. Ask the owner to change it via the dedicated /api/owner/scope-review-floor endpoint, or stop the agent and edit settings.json directly. Pure source inspection (grep/rg/cat/jq/git grep) is allowed; an interpreter or HTTP client naming the endpoint is not, whatever verb it spells."
|
||||
if _detect_safety_mode_self_lowering(cmd_lower):
|
||||
if _detect_safety_mode_self_lowering(cmd_lower, writeish=writeish):
|
||||
return "⚠️ SAFETY_MODE_SELF_LOWERING_BLOCKED: shell command pattern looks like an attempt to change OUROBOROS_SAFETY_MODE (e.g. to ``light``/``off``) through settings.json, /api/settings, or /api/owner/safety-mode. LLM-safety coverage is owner-controlled (BIBLE P3) — the agent must not reduce its own supervision. Ask the owner to change it via the dedicated /api/owner/safety-mode endpoint, or stop the agent and edit settings.json directly."
|
||||
if _detect_owner_skill_attest_self_call(cmd_lower):
|
||||
if _detect_owner_skill_attest_self_call(cmd_lower, writeish=writeish):
|
||||
return "⚠️ OWNER_SKILL_ATTESTATION_SELF_CALL_BLOCKED: shell command pattern looks like an attempt to loopback-POST /api/owner/skills/<skill>/attest-review. Owner-attestation skips the expensive LLM skill review and is OWNER-ONLY — the agent must not self-attest its own skill to bypass the immune system's review. Ask the owner to attest it from the Skills UI."
|
||||
if _detect_mutative_toggle_self_change(cmd_lower):
|
||||
if _detect_mutative_toggle_self_change(cmd_lower, writeish=writeish):
|
||||
return "⚠️ ELEVATION_BLOCKED: OUROBOROS_ALLOW_MUTATIVE_SUBAGENTS is owner-controlled (it grants subagents write power against the live body). Change it by stopping the agent and editing settings.json directly, then restart — the agent must not self-enable mutative subagents."
|
||||
if _detect_evolution_owner_control_self_change(cmd_lower):
|
||||
if _detect_evolution_owner_control_self_change(cmd_lower, writeish=writeish):
|
||||
return "⚠️ ELEVATION_BLOCKED: the self-evolution controls (OUROBOROS_POST_TASK_EVOLUTION and OUROBOROS_EVOLUTION_PERSISTENT_OBJECTIVE) are owner-controlled — they enable or steer self-modification cycles. Change them via the owner Settings UI, or stop the agent and edit settings.json directly — the agent must not self-set evolution controls."
|
||||
if _mentions_skill_owner_state(cmd_lower):
|
||||
return (
|
||||
|
|
|
|||
|
|
@ -42,7 +42,6 @@ if TYPE_CHECKING:
|
|||
_UNDECLARED_OUTPUTS_MARKER = "⚠️ ARTIFACT_OUTPUT_UNDECLARED"
|
||||
_UNDECLARED_OUTPUT_SCAN_MAX_FILES = 5000
|
||||
_UNDECLARED_OUTPUT_METADATA_COMMANDS = frozenset({"chmod", "chown", "mkdir", "rm"})
|
||||
_UNDECLARED_OUTPUT_READ_ONLY_COMMANDS = frozenset({"sed"})
|
||||
_SHELL_WRAPPER_COMMANDS = frozenset({"sh", "bash", "zsh"})
|
||||
_SHELL_REDIRECT_TARGET_TOKENS = frozenset({
|
||||
">", ">>", "1>", "1>>", "2>", "2>>", "&>", "&>>",
|
||||
|
|
@ -83,16 +82,11 @@ def _writer_targets_for_output_audit(argv: list[str]) -> set[str]:
|
|||
segment_targets = set(writer_target_tokens(command_argv))
|
||||
if command in _UNDECLARED_OUTPUT_METADATA_COMMANDS:
|
||||
segment_targets = _redirect_targets_for_audit(segment)
|
||||
elif command in _UNDECLARED_OUTPUT_READ_ONLY_COMMANDS:
|
||||
in_place = any(
|
||||
token == "--in-place"
|
||||
or token.startswith("--in-place=")
|
||||
or (token.startswith("-i") and token != "--ignore-case")
|
||||
for token in command_argv[1:]
|
||||
)
|
||||
if not in_place:
|
||||
segment_targets = _redirect_targets_for_audit(segment)
|
||||
else:
|
||||
# sed's own second in-place matcher is gone: writer_target_tokens is
|
||||
# now channel-aware itself (in-place in any spelling, in-script
|
||||
# w/W/e, -f scripts), so the audit consumes the ONE parser and only
|
||||
# adds syntactic redirects — a non-writing sed reports no targets.
|
||||
segment_targets.update(_redirect_targets_for_audit(segment))
|
||||
targets.update(str(target) for target in segment_targets if str(target).strip())
|
||||
return targets
|
||||
|
|
|
|||
|
|
@ -24,75 +24,27 @@ PROTECTED_RUNTIME_PATHS_LOWER = frozenset(
|
|||
p.lower() for p in PROTECTED_RUNTIME_PATHS
|
||||
) | frozenset(prefix.lower() for prefix in FROZEN_CONTRACT_PATH_PREFIXES)
|
||||
|
||||
SHELL_WRITE_INDICATORS = (
|
||||
"rm ", "rm\t", ">", "sed -i", "tee ", "truncate",
|
||||
"mv ", "cp ", "chmod ", "chown ", "unlink ", "delete", "trash",
|
||||
"rsync ", "write_text", ".write(", ".writelines(",
|
||||
"os.remove(", "os.unlink(", "os.mkdir(", "os.makedirs(", "sort -o",
|
||||
"writefilesync", "appendfilesync", "createwritestream",
|
||||
# Write-shape classification lives in its own leaf (extracted at the module-size
|
||||
# gate); every historical name stays importable from here.
|
||||
from ouroboros.tools.write_shape import ( # noqa: E402,F401
|
||||
INTERPRETER_WRITE_RE,
|
||||
PURE_FILTER_WRITER_COMMANDS,
|
||||
SHELL_WRITE_INDICATORS,
|
||||
_INTERPRETER_ANY_WRITE_RE,
|
||||
_INTERPRETER_LANE_EXCLUDED_INDICATORS,
|
||||
_OPEN_CALL_WRITE_INDICATOR_RE,
|
||||
_SAFE_STDIO_REDIRECT_TOKENS,
|
||||
_shell_write_indicator_scan,
|
||||
interpreter_write_shape,
|
||||
non_interpreter_write_shape,
|
||||
shell_has_write_indicator,
|
||||
)
|
||||
# Preserve the longstanding coarse ``open(`` signal without matching it as the
|
||||
# suffix of another callable such as ``urlopen(``.
|
||||
_OPEN_CALL_WRITE_INDICATOR_RE = re.compile(r"(?<![A-Za-z0-9_])open\(")
|
||||
_SAFE_STDIO_REDIRECT_TOKENS = frozenset({
|
||||
">/dev/null",
|
||||
"1>/dev/null",
|
||||
"2>/dev/null",
|
||||
"2>&1",
|
||||
"1>&2",
|
||||
"2>&-",
|
||||
})
|
||||
|
||||
LIGHT_SHELL_WRITER_COMMANDS = frozenset({
|
||||
"chmod", "chown", "cp", "gunzip", "gzip", "ln", "mkdir", "mv",
|
||||
"perl", "rm", "ruby", "sed", "sort", "tar", "touch", "truncate", "uniq", "unzip",
|
||||
})
|
||||
|
||||
INTERPRETER_WRITE_RE = re.compile(
|
||||
r"""(?is)(?:\.write\(|write_text\(|write_bytes\(|fs\.write|fs\.append|"""
|
||||
r"""createwritestream|unlink\(|rename\(|mkdir\(|rmtree\(|remove\(|"""
|
||||
r"""open\s*\([^)]*,\s*['"][^'"]*[wax+])"""
|
||||
)
|
||||
# Wider write-indicator net for the read-vs-write runtime_data scan (v6.54.3):
|
||||
# includes filesystem-mutating calls the base write regex misses (shutil.copy*/move,
|
||||
# touch, symlink/link, chmod/chown, makedirs/removedirs, truncate) — a hit here
|
||||
# without AST-resolved targets stays on the conservative full mention scan instead
|
||||
# of being treated as a pure read. This list and the AST walker are NOT one
|
||||
# vocabulary and no invariant ties them: `<mod>.open(p,"w")` (io/codecs/gzip/bz2/lzma)
|
||||
# matches here, while `_python_path_open_target` reads arg 0 as the MODE — right for
|
||||
# `Path(p).open("w")`, wrong here — so the walker answers "no targets" and the fence
|
||||
# reads that as a proven read. Measured, not hypothetical: it truncates a repo source
|
||||
# file. DISCLOSED, not detected (owner direction: weaken, never strengthen; a false
|
||||
# invariant is removed rather than made true). NB: the leading
|
||||
# (?is) of INTERPRETER_WRITE_RE.pattern already applies globally to the whole
|
||||
# concatenated expression — a second mid-pattern global flag is a hard
|
||||
# re.error on Python 3.11+ (review round 2).
|
||||
_INTERPRETER_ANY_WRITE_RE = re.compile(
|
||||
INTERPRETER_WRITE_RE.pattern
|
||||
+ r"""|(?:makedirs\(|removedirs\(|rmdir\(|copyfile\(|copy2\(|copytree\(|os\.replace\(|"""
|
||||
+ r"""shutil\.(?:copy|move)\(|\.touch\(|symlink\(|os\.link\(|\.link_to\(|hardlink_to\(|"""
|
||||
+ r"""chmod\(|chown\(|truncate\(|"""
|
||||
# OPAQUE / unmodeled write-capable calls (adversarial review r2 #1): an
|
||||
# external process (subprocess/os.system/popen) can `rm`/`mv`/`dd` anything,
|
||||
# and archive-extract / db-open write to a directory the AST never resolves.
|
||||
# A hit here has no AST-resolvable target, so it falls through to the
|
||||
# conservative full mention scan (blocks drive paths OUTSIDE the task roots)
|
||||
# instead of being mis-classified as a pure read. Pure reads (open()/read_text
|
||||
# with no write token) still match nothing and stay allowed.
|
||||
+ r"""subprocess\.|os\.system\(|os\.popen\(|Popen\(|check_call\(|check_output\(|"""
|
||||
+ r"""\.extractall\(|unpack_archive\(|make_archive\(|sqlite3\.connect\(|"""
|
||||
# LIBRARY save-APIs (fable-5 cumulative review F1): to_csv/savefig/.save &co
|
||||
# write files while carrying no base write-token, so an interpreter command
|
||||
# using them was classified as a PURE READ and skipped the runtime_data
|
||||
# mention scan entirely. A false positive here only re-applies the
|
||||
# conservative pre-v6.54.3 always-scan behavior (fail-closed direction).
|
||||
# The mode-shaped single-arg .open("w"/"ab"/"x+") is the pathlib positional
|
||||
# form the comma-anchored open() token above cannot see; the tight 1-3 char
|
||||
# mode lookahead keeps .open("<path>") reads out.
|
||||
+ r"""\.save\(|\.to_csv\(|\.to_excel\(|\.to_parquet\(|\.to_json\(|\.to_pickle\(|"""
|
||||
+ r"""savefig\(|np\.save|imwrite\(|pickle\.dump\(|json\.dump\(|"""
|
||||
+ r"""\.open\(\s*(?:mode\s*=\s*)?['"](?=[a-z+]{1,3}['"])[a-z+]*[wax+][a-z+]*['"])"""
|
||||
)
|
||||
EMBEDDED_RELATIVE_PATH_RE = re.compile(r"(?<![A-Za-z0-9_.-])(?:\.\.?/)+[^\s'\"\\),;\]]+")
|
||||
_REDIRECT_TARGET_TOKENS = frozenset({">", ">>", "1>", "1>>", "2>", "2>>", "&>", "&>>"})
|
||||
# ONE structural owner of "is this executable a script interpreter, and of which
|
||||
|
|
@ -215,7 +167,12 @@ _SCRIPT_LITERAL_WRITE_RE = {
|
|||
r"""(?:writeFileSync|appendFileSync|createWriteStream|mkdirSync|rmSync|rmdirSync|unlinkSync)\s*\(\s*(['"])(.*?)\1"""
|
||||
),
|
||||
"ruby": re.compile(
|
||||
r"""(?is)(?:File\.write|File\.open|FileUtils\.(?:touch|mkdir_p|rm|rm_rf|remove|copy|cp|mv))\s*\(\s*(['"])(.*?)\1"""
|
||||
# File.write / FileUtils writers always write; File.open / File.new name a
|
||||
# write TARGET only with a write-mode 2nd arg (sol review: the mode-blind
|
||||
# form reported File.open('/x','r') and blocked a read outside the root).
|
||||
r"""(?is)(?:File\.write|FileUtils\.(?:touch|mkdir_p|rm|rm_rf|remove|copy|cp|mv)|"""
|
||||
r"""File\.(?:open|new)(?=\s*\([^)]*,\s*['"][^'"]*[wax+])"""
|
||||
r""")\s*\(\s*(['"])(.*?)\1"""
|
||||
),
|
||||
}
|
||||
|
||||
|
|
@ -851,35 +808,6 @@ def runtime_data_guard_targets(
|
|||
)
|
||||
|
||||
|
||||
def shell_has_write_indicator(raw_cmd: Any) -> bool:
|
||||
if isinstance(raw_cmd, list):
|
||||
text = " ".join(str(x) for x in raw_cmd).lower()
|
||||
else:
|
||||
text = str(raw_cmd).lower()
|
||||
tokens = [str(token).lower() for token in shell_argv_with_inline(raw_cmd)]
|
||||
filtered_tokens: List[str] = []
|
||||
i = 0
|
||||
while i < len(tokens):
|
||||
token = tokens[i]
|
||||
if token in _SAFE_STDIO_REDIRECT_TOKENS:
|
||||
i += 1
|
||||
continue
|
||||
if token in {">", "1>", "2>"} and i + 1 < len(tokens) and tokens[i + 1] == "/dev/null":
|
||||
i += 2
|
||||
continue
|
||||
filtered_tokens.append(token)
|
||||
i += 1
|
||||
filtered_text = " ".join(filtered_tokens)
|
||||
for token in _SAFE_STDIO_REDIRECT_TOKENS:
|
||||
text = text.replace(token, " ")
|
||||
return (
|
||||
any(indicator in filtered_text for indicator in SHELL_WRITE_INDICATORS)
|
||||
or bool(_OPEN_CALL_WRITE_INDICATOR_RE.search(filtered_text))
|
||||
or any(indicator in text for indicator in SHELL_WRITE_INDICATORS if indicator != ">")
|
||||
or bool(_OPEN_CALL_WRITE_INDICATOR_RE.search(text))
|
||||
)
|
||||
|
||||
|
||||
def process_shell_guard_args(name: str, args: Dict[str, Any], *, ctx: Any = None, runtime_mode: str = "") -> Dict[str, Any]:
|
||||
"""Normalize process-tool arguments into the command shape inspected by shell guards."""
|
||||
|
||||
|
|
@ -1004,6 +932,16 @@ def repo_target_mentioned(
|
|||
|
||||
_COMMAND_SEPARATOR_TOKENS = frozenset({"&&", "||", ";", "|", "&"})
|
||||
|
||||
# A sed SCRIPT that can write or execute: the `w FILE`/`W FILE` command shape
|
||||
# (addressed `1w FILE` included — digits stay out of the lookbehind), the GNU
|
||||
# `e`/`e cmd` execute command, or a substitute's trailing flag run carrying w/e
|
||||
# after the closing `/` (`s/a/b/gw f`, `s/x/y/e` — the flag class [gpimM0-9]
|
||||
# keeps replacement words like `raw ` out). Word-embedded letters (`/delete/p`,
|
||||
# `s/e/x/`) stay reads; exotic non-`/` delimiters are a disclosed residual.
|
||||
_SED_SCRIPT_WRITE_RE = re.compile(
|
||||
r"(?<![A-Za-z_])[wW]\s+\S|(?<![A-Za-z_])e(?:\s*(?:$|;)|\s+\S)|/[gpimM0-9]*[we](?=\s|$|;)"
|
||||
)
|
||||
|
||||
|
||||
def writer_target_tokens(argv: List[str]) -> List[str]:
|
||||
"""Write TARGETS of a (possibly compound) command line.
|
||||
|
|
@ -1174,7 +1112,9 @@ def _writer_target_tokens_single(argv: List[str]) -> List[str]:
|
|||
if not argv:
|
||||
return []
|
||||
cmd = pathlib.PurePath(argv[0]).name.lower().removesuffix(".exe")
|
||||
operands = [arg for arg in argv[1:] if arg and not arg.startswith("-")]
|
||||
# A literal '-' is the STDIN OPERAND, not a flag: dropping it hid uniq's
|
||||
# output operand (`uniq - OUT` writes OUT) from every consumer (sol-max r2).
|
||||
operands = [arg for arg in argv[1:] if arg and (arg == "-" or not arg.startswith("-"))]
|
||||
targets: List[str] = []
|
||||
if cmd == "cp":
|
||||
targets.extend(operands[-1:] if len(operands) >= 2 else [])
|
||||
|
|
@ -1185,13 +1125,104 @@ def _writer_target_tokens_single(argv: List[str]) -> List[str]:
|
|||
elif cmd in {"chmod", "chown"}:
|
||||
targets.extend(operands[1:] if len(operands) >= 2 else [])
|
||||
elif cmd == "sed":
|
||||
targets.extend(operands[1:] if len(operands) >= 2 else operands)
|
||||
# sed's write channels are -i (any spelling, incl. GNU attached `-ibak`)
|
||||
# AND the in-script `w`/`W` file commands and GNU `e` execute (fable-5
|
||||
# round-2: POSIX `sed 'w f' in` writes f with no -i at all). A pure
|
||||
# filter is only a script PROVABLY free of those; a -f script file or a
|
||||
# single-letter w/W/e command shape fails closed to the operand fallback.
|
||||
sed_args = [str(a) for a in argv[1:]]
|
||||
inplace = any(
|
||||
# -i in ANY short spelling, clustered included (`-ni.bak`, `-nibak`):
|
||||
# 'i' anywhere in the leading cluster letters means in-place.
|
||||
(
|
||||
t.startswith("-")
|
||||
and not t.startswith("--")
|
||||
and "i" in t.split(".", 1)[0][1:]
|
||||
)
|
||||
or t == "--in-place"
|
||||
or t.startswith("--in-place=")
|
||||
for t in sed_args
|
||||
)
|
||||
scripts: list = []
|
||||
script_unprovable = False
|
||||
expect_expr = False
|
||||
for t in sed_args:
|
||||
if expect_expr:
|
||||
scripts.append(t)
|
||||
expect_expr = False
|
||||
elif t in ("-e", "--expression"):
|
||||
expect_expr = True
|
||||
elif t.startswith("--expression="):
|
||||
scripts.append(t.split("=", 1)[1])
|
||||
elif t in ("-f", "--file") or t.startswith("--file="):
|
||||
script_unprovable = True
|
||||
if not scripts and operands:
|
||||
scripts.append(operands[0])
|
||||
writing_scripts = [s for s in scripts if _SED_SCRIPT_WRITE_RE.search(s)]
|
||||
if inplace or script_unprovable or writing_scripts:
|
||||
# The `w FILE` filename lives INSIDE the script operand; reporting the
|
||||
# script text as a target lets the cwd-joining consumers (light fence,
|
||||
# protected lane) see where it lands, exactly like the old operand
|
||||
# fallback did.
|
||||
targets.extend(writing_scripts)
|
||||
targets.extend(operands[1:] if len(operands) >= 2 else operands)
|
||||
elif cmd == "tar":
|
||||
# Mode letters are the LEADING cluster letters only (`-cf/o.tar` is
|
||||
# create+file with an attached path — the 't' inside the path is not
|
||||
# list mode; sol-max r2). Old-style `tar tf a.tar` carries the letters
|
||||
# in the first operand. Write modes (c/x/r/u/A/d, --extract/--create/…)
|
||||
# keep the operand fallback plus the attached/long file and -C/--directory
|
||||
# values; pure list (`t` with no write letter) reads.
|
||||
tar_args = [str(a) for a in argv[1:]]
|
||||
mode_letters = ""
|
||||
attached_value = ""
|
||||
for t in tar_args:
|
||||
m = re.match(r"^-([A-Za-z]+)(.*)$", t)
|
||||
if m:
|
||||
mode_letters += m.group(1)
|
||||
if attached_value == "" and m.group(2):
|
||||
attached_value = m.group(2)
|
||||
old_style = ""
|
||||
if not mode_letters and operands and re.fullmatch(r"[A-Za-z]+", operands[0] or ""):
|
||||
old_style = operands[0]
|
||||
mode_letters = old_style
|
||||
long_write = any(
|
||||
t in ("--create", "--extract", "--get", "--append", "--update", "--delete", "--concatenate", "--catenate")
|
||||
for t in tar_args
|
||||
)
|
||||
write_mode = long_write or any(ch in mode_letters for ch in "cxruAd")
|
||||
listing = ("t" in mode_letters or "--list" in tar_args) and not write_mode
|
||||
if not listing:
|
||||
targets.extend(op for op in operands if op != old_style)
|
||||
if attached_value:
|
||||
targets.append(attached_value)
|
||||
for t in tar_args:
|
||||
if t.startswith(("--file=", "--directory=")):
|
||||
targets.append(t.split("=", 1)[1])
|
||||
elif cmd in {"gzip", "gunzip"}:
|
||||
# Read modes by LEADING cluster letters only (`-S.tgz` is a suffix value,
|
||||
# not test mode): -l/--list, -t/--test read; -c/--stdout writes stdout
|
||||
# only. The default invocation replaces its operand (file <-> file.gz).
|
||||
readonly_mode = False
|
||||
for t in (str(a) for a in argv[1:]):
|
||||
if t in ("--list", "--test", "--stdout", "--to-stdout"):
|
||||
readonly_mode = True
|
||||
break
|
||||
m = re.match(r"^-([A-Za-z]+)", t) if not t.startswith("--") else None
|
||||
if m and any(ch in m.group(1) for ch in "ltc"):
|
||||
readonly_mode = True
|
||||
break
|
||||
if not readonly_mode:
|
||||
targets.extend(operands)
|
||||
elif cmd == "sort":
|
||||
for idx, arg in enumerate(argv[1:], start=1):
|
||||
if arg == "-o" and idx + 1 < len(argv):
|
||||
if arg in ("-o", "--output") and idx + 1 < len(argv):
|
||||
targets.append(argv[idx + 1])
|
||||
if arg.startswith("--output="):
|
||||
elif arg.startswith("--output="):
|
||||
targets.append(arg.split("=", 1)[1])
|
||||
elif arg.startswith("-o") and len(arg) > 2 and not arg.startswith("--"):
|
||||
# Attached GNU spelling: `sort -oFILE`.
|
||||
targets.append(arg[2:])
|
||||
elif cmd == "uniq":
|
||||
targets.extend(operands[1:2] if len(operands) >= 2 else [])
|
||||
elif _light_writer_command(cmd):
|
||||
|
|
|
|||
296
ouroboros/tools/write_shape.py
Normal file
296
ouroboros/tools/write_shape.py
Normal file
|
|
@ -0,0 +1,296 @@
|
|||
"""Write-shape classification SSOT for shell commands (extracted from
|
||||
shell_guards.py at the module-size gate; shell_guards re-exports every name).
|
||||
|
||||
ONE seam decides whether a command line is WRITE-SHAPED before any deterministic
|
||||
write/owner-control guard acts on it: interpreter argv takes the mode-aware
|
||||
``interpreter_write_shape``, everything else takes ``non_interpreter_write_shape``
|
||||
(membership floor for unconditional writers, real-channel evidence for the
|
||||
pure-filter utilities, prose words yield to the caller's read-carve). No guard
|
||||
may consume a coarser write fact than this composition's — that is the
|
||||
family-wide application of the v6.80.0 scope-floor read-carve contract.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import re
|
||||
from typing import Any, Callable, List, Optional
|
||||
|
||||
from ouroboros.shell_parse import shell_argv, shell_argv_with_inline
|
||||
|
||||
SHELL_WRITE_INDICATORS = (
|
||||
"rm ", "rm\t", ">", "sed -i", "tee ", "truncate",
|
||||
"mv ", "cp ", "chmod ", "chown ", "unlink ", "delete", "trash",
|
||||
"rsync ", "write_text", ".write(", ".writelines(",
|
||||
"os.remove(", "os.unlink(", "os.mkdir(", "os.makedirs(", "sort -o",
|
||||
"writefilesync", "appendfilesync", "createwritestream",
|
||||
)
|
||||
# Preserve the longstanding coarse ``open(`` signal without matching it as the
|
||||
# suffix of another callable such as ``urlopen(``.
|
||||
_OPEN_CALL_WRITE_INDICATOR_RE = re.compile(r"(?<![A-Za-z0-9_])open\(")
|
||||
_SAFE_STDIO_REDIRECT_TOKENS = frozenset({
|
||||
">/dev/null",
|
||||
"1>/dev/null",
|
||||
"2>/dev/null",
|
||||
"2>&1",
|
||||
"1>&2",
|
||||
"2>&-",
|
||||
})
|
||||
|
||||
INTERPRETER_WRITE_RE = re.compile(
|
||||
# The open() clause anchors on the MODE argument (python short quoted [a-z+]
|
||||
# write flag, or perl '>' / '>>' / '+>' / '+<'), not any quoted arg containing
|
||||
# w/a/x/+ — a read `open(my $fh, '<', '/tmp/f.txt')` must not match on the 'x'
|
||||
# in the filename.
|
||||
r"""(?is)(?:\.write\(|write_text\(|write_bytes\(|fs\.write|fs\.append|"""
|
||||
r"""createwritestream|unlink\(|rename\(|mkdir\(|rmtree\(|remove\(|"""
|
||||
r"""open\s*\([^)]*,\s*(?:mode\s*=\s*)?['"](?=[a-z+]{1,3}['"])[a-z+]*[wax+][a-z+]*['"]|"""
|
||||
r"""open\s*\([^)]*,\s*(?:mode\s*=\s*)?['"]\s*(?:\+\s*[<>]|>{1,2}))"""
|
||||
)
|
||||
# Wider write-indicator net for the read-vs-write runtime_data scan (v6.54.3):
|
||||
# includes filesystem-mutating calls the base write regex misses (shutil.copy*/move,
|
||||
# touch, symlink/link, chmod/chown, makedirs/removedirs, truncate) — a hit here
|
||||
# without AST-resolved targets stays on the conservative full mention scan instead
|
||||
# of being treated as a pure read. This list and the AST walker are NOT one
|
||||
# vocabulary and no invariant ties them: `<mod>.open(p,"w")` (io/codecs/gzip/bz2/lzma)
|
||||
# matches here, while `_python_path_open_target` reads arg 0 as the MODE — right for
|
||||
# `Path(p).open("w")`, wrong here — so the walker answers "no targets" and the fence
|
||||
# reads that as a proven read. Measured, not hypothetical: it truncates a repo source
|
||||
# file. DISCLOSED, not detected (owner direction: weaken, never strengthen; a false
|
||||
# invariant is removed rather than made true). NB: the leading
|
||||
# (?is) of INTERPRETER_WRITE_RE.pattern already applies globally to the whole
|
||||
# concatenated expression — a second mid-pattern global flag is a hard
|
||||
# re.error on Python 3.11+ (review round 2).
|
||||
_INTERPRETER_ANY_WRITE_RE = re.compile(
|
||||
INTERPRETER_WRITE_RE.pattern
|
||||
+ r"""|(?:makedirs\(|removedirs\(|rmdir\(|copyfile\(|copy2\(|copytree\(|os\.replace\(|"""
|
||||
+ r"""shutil\.(?:copy|move)\(|\.touch\(|symlink\(|os\.link\(|\.link_to\(|hardlink_to\(|"""
|
||||
+ r"""chmod\(|chown\(|truncate\(|"""
|
||||
# OPAQUE / unmodeled write-capable calls (adversarial review r2 #1): an
|
||||
# external process (subprocess/os.system/popen) can `rm`/`mv`/`dd` anything,
|
||||
# and archive-extract / db-open write to a directory the AST never resolves.
|
||||
# A hit here has no AST-resolvable target, so it falls through to the
|
||||
# conservative full mention scan (blocks drive paths OUTSIDE the task roots)
|
||||
# instead of being mis-classified as a pure read. Pure reads (open()/read_text
|
||||
# with no write token) still match nothing and stay allowed.
|
||||
+ r"""subprocess\.|os\.system\(|os\.popen\(|Popen\(|check_call\(|check_output\(|"""
|
||||
+ r"""\.extractall\(|unpack_archive\(|make_archive\(|sqlite3\.connect\(|"""
|
||||
# LIBRARY save-APIs (fable-5 cumulative review F1): to_csv/savefig/.save &co
|
||||
# write files while carrying no base write-token, so an interpreter command
|
||||
# using them was classified as a PURE READ and skipped the runtime_data
|
||||
# mention scan entirely. A false positive here only re-applies the
|
||||
# conservative pre-v6.54.3 always-scan behavior (fail-closed direction).
|
||||
# The mode-shaped single-arg .open("w"/"ab"/"x+") is the pathlib positional
|
||||
# form the comma-anchored open() token above cannot see; the tight 1-3 char
|
||||
# mode lookahead keeps .open("<path>") reads out.
|
||||
+ r"""\.save\(|\.to_csv\(|\.to_excel\(|\.to_parquet\(|\.to_json\(|\.to_pickle\(|"""
|
||||
+ r"""savefig\(|np\.save|imwrite\(|pickle\.dump\(|json\.dump\(|"""
|
||||
# RUBY native write idioms (fable-5 review): with the membership floor gone the
|
||||
# vocabulary must see ruby's spellings — File.delete/Dir.delete, IO.binwrite/
|
||||
# syswrite, FileUtils.* (variable args the literal-target regex cannot see; rare
|
||||
# read helpers like compare_file just fall back to the conservative scan).
|
||||
+ r"""file\.delete\(|dir\.delete\(|binwrite\(|syswrite\(|fileutils\.[a-z_]+|"""
|
||||
+ r"""file\.new\s*\([^)]*,\s*['"][^'"]*[wax+]|"""
|
||||
+ r"""\.open\(\s*(?:mode\s*=\s*)?['"](?=[a-z+]{1,3}['"])[a-z+]*[wax+][a-z+]*['"])"""
|
||||
)
|
||||
|
||||
# Interpreter-lane / prose refinements (see interpreter_write_shape): bare
|
||||
# ENGLISH WORDS excluded so "count deleted rows" is not a write (structural
|
||||
# deletion — os.remove/unlink/rmtree/File.delete/fileutils/boundary-worded rm —
|
||||
# stays covered); command words take a left word boundary so 'cp ' misses 'scp ';
|
||||
# 'truncate' takes a right boundary ("results truncated" reads); '>' counts only as
|
||||
# a real redirect (token-initial or glued into an operand), never inside a located
|
||||
# code body (`<$fh>`, `a > b`, '=>', '>='). Parenless perl builtins (`rename $a, $b`)
|
||||
# are a disclosed residual.
|
||||
_INTERPRETER_LANE_EXCLUDED_INDICATORS = frozenset({"delete", "trash"})
|
||||
# For NON-interpreter argv the same three bare words are PROSE, not channels: the
|
||||
# real channels are head membership (`truncate -s 0 f` blocks on its head), the
|
||||
# per-segment writer targets, and the option indicators ('sed -i', 'sort -o').
|
||||
_PROSE_WORD_INDICATORS = frozenset({"delete", "trash", "truncate"})
|
||||
_COMMAND_WORD_INDICATORS = frozenset({
|
||||
"rm ", "rm\t", "tee ", "mv ", "cp ", "chmod ", "chown ", "unlink ", "rsync ",
|
||||
})
|
||||
_COMMAND_WORD_BOUNDARY_RES = {
|
||||
indicator: re.compile(r"(?<![A-Za-z0-9_.-])" + re.escape(indicator))
|
||||
for indicator in _COMMAND_WORD_INDICATORS
|
||||
}
|
||||
_TRUNCATE_BOUNDARY_RE = re.compile(r"truncate(?![a-z])")
|
||||
_REDIRECT_SHAPE_TOKEN_RE = re.compile(r"^(?:(?:&|\d?)>>?(?=$|[^&|-])|>&.)")
|
||||
_MIDTOKEN_REDIRECT_RE = re.compile(r"(?<![<>=&|'\"-])>{1,2}(?![>=&])")
|
||||
|
||||
# LIGHT_SHELL_WRITER_COMMANDS members that are PURE FILTERS in their default
|
||||
# invocation: they write only through an explicit channel (sed -i, sort -o, a
|
||||
# second uniq operand, tar create/extract, gzip without -l/-t/-c, a redirect) —
|
||||
# every one of which the target parser or an option indicator reports on its own.
|
||||
# Bare membership made `sort /etc/hosts` "write-like" (EXT-3). Unconditional
|
||||
# writers (cp/mv/rm/mkdir/touch/chmod/...) keep the membership floor, and the
|
||||
# interpreter members (ruby/perl) take interpreter_write_shape instead.
|
||||
PURE_FILTER_WRITER_COMMANDS = frozenset({"gunzip", "gzip", "sed", "sort", "tar", "uniq"})
|
||||
|
||||
|
||||
def _shell_write_indicator_scan(
|
||||
raw_cmd: Any,
|
||||
*,
|
||||
include_bare_open: bool,
|
||||
interpreter_lane: bool = False,
|
||||
exclude_prose_words: bool = False,
|
||||
) -> bool:
|
||||
if isinstance(raw_cmd, list):
|
||||
text = " ".join(str(x) for x in raw_cmd).lower()
|
||||
else:
|
||||
text = str(raw_cmd).lower()
|
||||
tokens = [str(token).lower() for token in shell_argv_with_inline(raw_cmd)]
|
||||
filtered_tokens: List[str] = []
|
||||
i = 0
|
||||
while i < len(tokens):
|
||||
token = tokens[i]
|
||||
if token in _SAFE_STDIO_REDIRECT_TOKENS:
|
||||
i += 1
|
||||
continue
|
||||
if token in {">", "1>", "2>"} and i + 1 < len(tokens) and tokens[i + 1] == "/dev/null":
|
||||
i += 2
|
||||
continue
|
||||
filtered_tokens.append(token)
|
||||
i += 1
|
||||
filtered_text = " ".join(filtered_tokens)
|
||||
for token in _SAFE_STDIO_REDIRECT_TOKENS:
|
||||
text = text.replace(token, " ")
|
||||
|
||||
if interpreter_lane:
|
||||
from ouroboros.shell_parse import shell_command_string
|
||||
from ouroboros.tools.shell_guards import interpreter_inline_code
|
||||
|
||||
argv = shell_argv(raw_cmd)
|
||||
bodies = list(interpreter_inline_code(argv))
|
||||
# An sh -c wrap hides the interpreter one level down; locate the inner
|
||||
# bodies too so a '>' comparison inside them is not read as a redirect.
|
||||
if argv and str(argv[0]).rsplit("/", 1)[-1].lower() in {"sh", "bash", "zsh"}:
|
||||
inner = shell_command_string(argv)
|
||||
if inner:
|
||||
bodies.extend(interpreter_inline_code(shell_argv(inner)))
|
||||
inline_bodies = frozenset(body.lower() for body in bodies)
|
||||
else:
|
||||
inline_bodies = frozenset()
|
||||
|
||||
def _in_located_body(tok: str) -> bool:
|
||||
# Joined flags carry the body INSIDE the token (`-cBODY`, `--eval=BODY`).
|
||||
return any(body and body in tok for body in inline_bodies)
|
||||
|
||||
def _indicator_hits(scan_text: str, *, allow_bare_redirect: bool) -> bool:
|
||||
for indicator in SHELL_WRITE_INDICATORS:
|
||||
if indicator == ">":
|
||||
if not allow_bare_redirect:
|
||||
continue
|
||||
if interpreter_lane:
|
||||
# Token-level only: a real redirect is its own shell token or
|
||||
# glued into an operand; a '>' inside a located inline-code
|
||||
# body is not a write channel.
|
||||
if any(
|
||||
_REDIRECT_SHAPE_TOKEN_RE.match(tok)
|
||||
or (not _in_located_body(tok) and _MIDTOKEN_REDIRECT_RE.search(tok))
|
||||
for tok in filtered_tokens
|
||||
):
|
||||
return True
|
||||
continue
|
||||
if exclude_prose_words and indicator in _PROSE_WORD_INDICATORS:
|
||||
continue
|
||||
if interpreter_lane:
|
||||
if indicator in _INTERPRETER_LANE_EXCLUDED_INDICATORS:
|
||||
continue
|
||||
if indicator == "truncate":
|
||||
if _TRUNCATE_BOUNDARY_RE.search(scan_text):
|
||||
return True
|
||||
continue
|
||||
boundary = _COMMAND_WORD_BOUNDARY_RES.get(indicator)
|
||||
if boundary is not None:
|
||||
if boundary.search(scan_text):
|
||||
return True
|
||||
continue
|
||||
if indicator in scan_text:
|
||||
return True
|
||||
return False
|
||||
|
||||
if _indicator_hits(filtered_text, allow_bare_redirect=True) or _indicator_hits(
|
||||
text, allow_bare_redirect=False
|
||||
):
|
||||
return True
|
||||
if include_bare_open and (
|
||||
_OPEN_CALL_WRITE_INDICATOR_RE.search(filtered_text) or _OPEN_CALL_WRITE_INDICATOR_RE.search(text)
|
||||
):
|
||||
return True
|
||||
return False
|
||||
|
||||
|
||||
def shell_has_write_indicator(raw_cmd: Any) -> bool:
|
||||
return _shell_write_indicator_scan(raw_cmd, include_bare_open=True)
|
||||
|
||||
|
||||
def interpreter_write_shape(raw_cmd: Any) -> bool:
|
||||
"""Mode-aware write-shape classification for an INTERPRETER command line.
|
||||
|
||||
The coarse ``open(`` token marks a read-only ``open(p, 'rb')`` as writeish —
|
||||
the class the light-mode runtime_data lane already re-judges ("the original
|
||||
GAIA class", review round 8). For interpreter argv the bare token is replaced
|
||||
by ``_INTERPRETER_ANY_WRITE_RE`` (write-mode opens incl. perl '>' spellings,
|
||||
pathlib ``.open('w')``, save-APIs, ruby File.delete/FileUtils, opaque
|
||||
subprocess/exec escapes); every shell-level indicator still counts. Disclosed
|
||||
residual: ``open(p, m)`` with the mode in a variable is not write-shaped — the
|
||||
external-workspace runtime/secret read guard and the LLM safety supervisor
|
||||
stay the covering controls.
|
||||
"""
|
||||
if _shell_write_indicator_scan(raw_cmd, include_bare_open=False, interpreter_lane=True):
|
||||
return True
|
||||
if isinstance(raw_cmd, list):
|
||||
text = " ".join(str(x) for x in raw_cmd)
|
||||
else:
|
||||
text = str(raw_cmd)
|
||||
return bool(_INTERPRETER_ANY_WRITE_RE.search(text))
|
||||
|
||||
|
||||
def non_interpreter_write_shape(
|
||||
raw_cmd: Any,
|
||||
argv: List[str],
|
||||
executable: str,
|
||||
*,
|
||||
is_pure_read: Optional[Callable[[str], bool]] = None,
|
||||
) -> bool:
|
||||
"""Mode-aware write shape for NON-interpreter argv (the composition's other half).
|
||||
|
||||
Unconditional writers keep the membership floor. Pure-filter utilities
|
||||
(``PURE_FILTER_WRITER_COMMANDS``) are write-shaped only through a real channel
|
||||
— an option indicator, a redirect, or a writer target the parser reports (the
|
||||
caller ORs ``explicit_write_targets`` in). The bare prose words yield to the
|
||||
caller's read-carve: on a provably read-only inspection line (`grep -n delete
|
||||
ouroboros/safety.py`, ``rg truncate …``) a word is not a write channel; any
|
||||
head the carve cannot prove stays fail-closed on the full legacy scan.
|
||||
"""
|
||||
from ouroboros.tools.shell_guards import LIGHT_SHELL_WRITER_COMMANDS, interpreter_family
|
||||
|
||||
member = bool(argv) and (
|
||||
(interpreter_family(executable) or executable) in LIGHT_SHELL_WRITER_COMMANDS
|
||||
)
|
||||
if member and executable not in PURE_FILTER_WRITER_COMMANDS:
|
||||
return True
|
||||
if _shell_write_indicator_scan(raw_cmd, include_bare_open=True, exclude_prose_words=True):
|
||||
return True
|
||||
if _shell_write_indicator_scan(raw_cmd, include_bare_open=True):
|
||||
# Only the prose words fired. A SINGLE-segment pure-filter head's real
|
||||
# channels are all target/option evidence already judged above, so prose
|
||||
# inside its script/pattern argument (`sed -n '/delete/p' f`) is never a
|
||||
# channel — but the head speaks only for ITS OWN segment: a compound line
|
||||
# (`sort f && find … -delete`) goes to the carve, which judges every
|
||||
# segment and fails closed (sol-max round-2). Otherwise a provably
|
||||
# read-only line reads; anything unproven keeps the legacy fail-closed
|
||||
# classification.
|
||||
compound = any(
|
||||
str(t) in ("&&", "||", ";", "|", "&") for t in shell_argv(raw_cmd)
|
||||
)
|
||||
if executable in PURE_FILTER_WRITER_COMMANDS and not compound:
|
||||
return False
|
||||
if is_pure_read is None:
|
||||
return True
|
||||
if isinstance(raw_cmd, list):
|
||||
text_lower = " ".join(str(x) for x in raw_cmd).lower()
|
||||
else:
|
||||
text_lower = str(raw_cmd).lower()
|
||||
return not is_pure_read(text_lower)
|
||||
return False
|
||||
|
|
@ -53,6 +53,8 @@ DATA = pathlib.Path(
|
|||
if str(REPO) not in sys.path:
|
||||
sys.path.insert(0, str(REPO))
|
||||
|
||||
from ouroboros.openrouter_attribution import OPENROUTER_APP_HEADERS # noqa: E402
|
||||
|
||||
# Release diffs touch protected core paths; only pro mode may stage them for
|
||||
# review. An explicit operator env value still wins.
|
||||
os.environ.setdefault("OUROBOROS_RUNTIME_MODE", "pro")
|
||||
|
|
@ -81,6 +83,7 @@ _REVIEW_SUBSTRATE_PATHS = frozenset({
|
|||
"ouroboros/context_budget.py",
|
||||
"ouroboros/deadline_utils.py",
|
||||
"ouroboros/llm.py",
|
||||
"ouroboros/openrouter_attribution.py",
|
||||
"ouroboros/outcomes.py",
|
||||
"ouroboros/platform_layer.py",
|
||||
"ouroboros/pricing.py",
|
||||
|
|
@ -380,7 +383,7 @@ def _probe_model_for_key(token: str, model: str) -> tuple[bool, str]:
|
|||
|
||||
response = httpx.post(
|
||||
"https://openrouter.ai/api/v1/chat/completions",
|
||||
headers={"Authorization": f"Bearer {token}"},
|
||||
headers={"Authorization": f"Bearer {token}", **OPENROUTER_APP_HEADERS},
|
||||
json={
|
||||
"model": model,
|
||||
"max_tokens": 1,
|
||||
|
|
|
|||
246
tests/test_claudexor_quota_envelope.py
Normal file
246
tests/test_claudexor_quota_envelope.py
Normal file
|
|
@ -0,0 +1,246 @@
|
|||
"""One-read quota envelope regressions for route health and Accounts status."""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
from concurrent.futures import Future
|
||||
|
||||
import httpx
|
||||
|
||||
from ouroboros.gateways.claudexor import ClaudexorGateway, DaemonEndpoint
|
||||
from ouroboros.subagents import _exhausted_window
|
||||
|
||||
|
||||
def _spent(subject_id: str) -> dict:
|
||||
return {
|
||||
"subject": {"harness": "claude", "subject_id": subject_id},
|
||||
"freshness": "fresh",
|
||||
"constraints": [{"used_ratio": 1.0, "resets_at": "2099-01-01T00:00:00Z"}],
|
||||
}
|
||||
|
||||
|
||||
def test_route_health_contradictory_same_subject_absence_fails_open():
|
||||
class Gateway:
|
||||
reads = 0
|
||||
|
||||
def quota_state(self):
|
||||
self.reads += 1
|
||||
return {
|
||||
"snapshots": [_spent("proton4")],
|
||||
"absences": [{
|
||||
"subject": {"harness": "claude", "subject_id": "proton4"},
|
||||
"reason": "refresh_failed",
|
||||
}],
|
||||
"refreshed_at": "2026-08-30T00:00:00Z",
|
||||
}
|
||||
|
||||
def quota_snapshots(self): # pragma: no cover - forbidden production path
|
||||
raise AssertionError("route health must not perform a second quota read")
|
||||
|
||||
def quota_absences(self): # pragma: no cover - forbidden production path
|
||||
raise AssertionError("route health must not perform a second quota read")
|
||||
|
||||
gateway = Gateway()
|
||||
assert _exhausted_window(gateway, "claude", pinned_profile="proton4") == (False, "")
|
||||
assert gateway.reads == 1
|
||||
|
||||
|
||||
def test_route_health_gap_for_another_subject_keeps_unpinned_route_unknown():
|
||||
class Gateway:
|
||||
def quota_state(self):
|
||||
return {
|
||||
"snapshots": [_spent("proton4")],
|
||||
"absences": [{
|
||||
"subject": {"harness": "claude", "subject_id": "proton3"},
|
||||
"reason": "refresh_failed",
|
||||
}],
|
||||
}
|
||||
|
||||
assert _exhausted_window(Gateway(), "claude") == (False, "")
|
||||
|
||||
|
||||
def test_real_gateway_route_health_performs_one_physical_quota_get():
|
||||
requests: list[str] = []
|
||||
|
||||
def handler(request: httpx.Request) -> httpx.Response:
|
||||
requests.append(f"{request.method} {request.url.path}")
|
||||
return httpx.Response(200, json={
|
||||
"snapshots": [_spent("proton4")],
|
||||
"absences": [{
|
||||
"subject": {"harness": "claude", "subject_id": "proton4"},
|
||||
"reason": "refresh_failed",
|
||||
}],
|
||||
"refreshed_at": "2026-08-30T00:00:00Z",
|
||||
})
|
||||
|
||||
with ClaudexorGateway(DaemonEndpoint("127.0.0.1", 1, "test-token")) as gateway:
|
||||
gateway._client.close()
|
||||
gateway._client = httpx.Client(
|
||||
base_url="http://127.0.0.1:1",
|
||||
transport=httpx.MockTransport(handler),
|
||||
)
|
||||
assert _exhausted_window(
|
||||
gateway, "claude", pinned_profile="proton4") == (False, "")
|
||||
|
||||
assert requests == ["GET /v2/quota"]
|
||||
|
||||
|
||||
def test_same_envelope_sentinel_cannot_mix_with_a_later_quota_epoch():
|
||||
class Gateway:
|
||||
reads = 0
|
||||
|
||||
def quota_state(self):
|
||||
self.reads += 1
|
||||
if self.reads > 1:
|
||||
raise AssertionError("a second epoch was read")
|
||||
return {"snapshots": [_spent("proton4")], "absences": []}
|
||||
|
||||
gateway = Gateway()
|
||||
assert _exhausted_window(gateway, "claude", pinned_profile="proton4") == (
|
||||
True, "2099-01-01T00:00:00Z")
|
||||
assert gateway.reads == 1
|
||||
|
||||
|
||||
def test_route_health_normalizes_malformed_quota_envelope_collections():
|
||||
class Gateway:
|
||||
def __init__(self, envelope):
|
||||
self.envelope = envelope
|
||||
|
||||
def quota_state(self):
|
||||
return self.envelope
|
||||
|
||||
for absences in (None, 42, "bad", {}):
|
||||
assert _exhausted_window(
|
||||
Gateway({"snapshots": [_spent("proton4")], "absences": absences}),
|
||||
"claude",
|
||||
pinned_profile="proton4",
|
||||
) == (True, "2099-01-01T00:00:00Z")
|
||||
|
||||
assert _exhausted_window(
|
||||
Gateway({"snapshots": [_spent("proton4")]}),
|
||||
"claude",
|
||||
pinned_profile="proton4",
|
||||
) == (True, "2099-01-01T00:00:00Z")
|
||||
|
||||
for snapshots in (None, 42, "bad", {}):
|
||||
assert _exhausted_window(
|
||||
Gateway({"snapshots": snapshots, "absences": []}),
|
||||
"claude",
|
||||
pinned_profile="proton4",
|
||||
) == (False, "")
|
||||
|
||||
assert _exhausted_window(
|
||||
Gateway({"absences": []}), "claude", pinned_profile="proton4"
|
||||
) == (False, "")
|
||||
|
||||
|
||||
def test_status_projects_snapshots_and_absences_from_its_single_read(monkeypatch, tmp_path):
|
||||
from ouroboros import claudexor_daemon as owned
|
||||
from ouroboros.gateway.claudexor_accounts import _status_payload
|
||||
from ouroboros.gateways import claudexor as gateway_module
|
||||
|
||||
class Daemon:
|
||||
def status_dict(self):
|
||||
return {"state": "running"}
|
||||
|
||||
class Gateway:
|
||||
engine_version = "3.9.0"
|
||||
quota_reads = 0
|
||||
|
||||
def __init__(self, _endpoint):
|
||||
pass
|
||||
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *_args):
|
||||
return False
|
||||
|
||||
def handshake(self):
|
||||
return {}
|
||||
|
||||
def agent_capabilities(self):
|
||||
return {"harnesses": []}
|
||||
|
||||
def harnesses(self):
|
||||
return []
|
||||
|
||||
def credential_profiles(self):
|
||||
return {"profiles": [], "harnessAccounts": []}
|
||||
|
||||
def operations(self):
|
||||
return {"operations": []}
|
||||
|
||||
def quota_state(self):
|
||||
type(self).quota_reads += 1
|
||||
return {
|
||||
"snapshots": [_spent("proton4")],
|
||||
"absences": [{
|
||||
"subject": {"harness": "claude", "subject_id": "proton3"},
|
||||
"reason": "poll_paced",
|
||||
}],
|
||||
# POST-refresh response metadata is not part of the ordinary
|
||||
# GET projection and must not grow a dead status contract.
|
||||
"refreshed_at": "2026-08-30T00:00:00Z",
|
||||
"refresh_skipped": [{"vendor": "claude", "not_before": "2026-08-30T01:00:00Z"}],
|
||||
}
|
||||
|
||||
def quota_snapshots(self): # pragma: no cover - forbidden status reader
|
||||
raise AssertionError("status must not perform a second quota read")
|
||||
|
||||
monkeypatch.setattr(owned, "get_owned_daemon", lambda: Daemon())
|
||||
monkeypatch.setattr(owned, "owned_config_dir", lambda: tmp_path / "cfg")
|
||||
monkeypatch.setattr(gateway_module, "discover_daemon_at", lambda _path: object())
|
||||
monkeypatch.setattr(gateway_module, "ClaudexorGateway", Gateway)
|
||||
|
||||
payload = _status_payload(False)
|
||||
assert Gateway.quota_reads == 1
|
||||
assert payload["quota"][0]["subject"]["subject_id"] == "proton4"
|
||||
assert payload["quota_absences"][0]["reason"] == "poll_paced"
|
||||
assert "quota_refreshed_at" not in payload
|
||||
assert "quota_refresh_skipped" not in payload
|
||||
|
||||
|
||||
def test_status_treats_absent_or_malformed_additive_quota_fields_as_empty(monkeypatch, tmp_path):
|
||||
from ouroboros import claudexor_daemon as owned
|
||||
from ouroboros.gateway.claudexor_accounts import _status_payload
|
||||
from ouroboros.gateways import claudexor as gateway_module
|
||||
|
||||
class Daemon:
|
||||
def status_dict(self):
|
||||
return {"state": "running"}
|
||||
|
||||
class Gateway:
|
||||
engine_version = "3.8.3"
|
||||
|
||||
def __init__(self, _endpoint): pass
|
||||
def __enter__(self): return self
|
||||
def __exit__(self, *_args): return False
|
||||
def handshake(self): return {}
|
||||
def agent_capabilities(self): return {"harnesses": []}
|
||||
def harnesses(self): return []
|
||||
def credential_profiles(self): return {"profiles": [], "harnessAccounts": []}
|
||||
def operations(self): return {"operations": []}
|
||||
def quota_state(self):
|
||||
return {"snapshots": [], "absences": 42}
|
||||
|
||||
monkeypatch.setattr(owned, "get_owned_daemon", lambda: Daemon())
|
||||
monkeypatch.setattr(owned, "owned_config_dir", lambda: tmp_path / "cfg")
|
||||
monkeypatch.setattr(gateway_module, "discover_daemon_at", lambda _path: object())
|
||||
monkeypatch.setattr(gateway_module, "ClaudexorGateway", Gateway)
|
||||
payload = _status_payload(False)
|
||||
assert payload["reads"]["quota"] == "ok"
|
||||
assert payload["quota_absences"] == []
|
||||
|
||||
|
||||
def test_quota_facet_rejects_missing_or_non_list_snapshots():
|
||||
from ouroboros.gateway.claudexor_accounts import _facet_outcome
|
||||
|
||||
for body in ({"absences": []}, {"snapshots": "bad", "absences": []}):
|
||||
call = Future()
|
||||
call.set_result(body)
|
||||
state, value, error = _facet_outcome(
|
||||
call,
|
||||
envelope=("snapshots",),
|
||||
list_fields=("snapshots",),
|
||||
)
|
||||
assert (state, value, error) == ("failed", None, None)
|
||||
|
|
@ -51,6 +51,7 @@ def test_provider_probe_checks_exact_model_without_server_search(monkeypatch, tm
|
|||
return {"data": {"limit_remaining": 100}}
|
||||
assert method == "POST"
|
||||
captured["body"] = body
|
||||
captured["headers"] = headers
|
||||
return {
|
||||
"id": "response-1",
|
||||
"model": "deepseek/deepseek-v4-flash-0731",
|
||||
|
|
@ -77,6 +78,11 @@ def test_provider_probe_checks_exact_model_without_server_search(monkeypatch, tm
|
|||
|
||||
assert captured["body"]["messages"] == [{"role": "user", "content": "Reply with OK."}]
|
||||
assert "tools" not in captured["body"]
|
||||
from ouroboros.openrouter_attribution import OPENROUTER_APP_HEADERS
|
||||
|
||||
assert {
|
||||
key: captured["headers"][key] for key in OPENROUTER_APP_HEADERS
|
||||
} == OPENROUTER_APP_HEADERS
|
||||
assert executor.provider_observation["observed_model"] == (
|
||||
"deepseek/deepseek-v4-flash-0731"
|
||||
)
|
||||
|
|
|
|||
|
|
@ -48,6 +48,7 @@ def test_contributor_trust_boundary_covers_functional_review_dependencies():
|
|||
"ouroboros/delegate_output.py",
|
||||
"ouroboros/gateways/claudexor.py",
|
||||
"ouroboros/outcomes.py",
|
||||
"ouroboros/openrouter_attribution.py",
|
||||
"ouroboros/platform_layer.py",
|
||||
"ouroboros/pricing.py",
|
||||
# The v6.87.21 seam split moved route vocabulary, transport dispatch and
|
||||
|
|
|
|||
|
|
@ -1,5 +1,17 @@
|
|||
import pytest
|
||||
|
||||
from ouroboros.llm import LLMClient
|
||||
from ouroboros.openrouter_attribution import OPENROUTER_APP_HEADERS
|
||||
|
||||
|
||||
def test_resolve_openrouter_target_uses_canonical_app_attribution():
|
||||
target = LLMClient()._resolve_remote_target("openai/gpt-5.6-sol")
|
||||
|
||||
assert OPENROUTER_APP_HEADERS == {
|
||||
"HTTP-Referer": "https://ouroboros-agent.ai/",
|
||||
"X-OpenRouter-Title": "Ouroboros",
|
||||
}
|
||||
assert target["default_headers"] == OPENROUTER_APP_HEADERS
|
||||
|
||||
|
||||
def test_resolve_openai_target(monkeypatch):
|
||||
|
|
|
|||
|
|
@ -353,6 +353,9 @@ class TestSupportedParametersFilter:
|
|||
"type": "openrouter:web_search",
|
||||
"parameters": {"search_context_size": "medium", "max_total_results": 10},
|
||||
}]
|
||||
from ouroboros.openrouter_attribution import OPENROUTER_APP_HEADERS
|
||||
|
||||
assert captured["client"]["default_headers"] == OPENROUTER_APP_HEADERS
|
||||
|
||||
def test_parameter_rejection_learns_sampling_strip_without_version_gate(self, monkeypatch):
|
||||
from ouroboros.llm import LLMClient
|
||||
|
|
|
|||
|
|
@ -31,7 +31,8 @@ def test_model_catalog_tags_provider_values(monkeypatch):
|
|||
|
||||
async def fake_openrouter(_client, _api_key):
|
||||
return [model_catalog_api._build_model_catalog_entry(
|
||||
"openrouter", "Anthropic", "anthropic/claude-opus", "Claude Opus", source="OpenRouter"
|
||||
"openrouter", "Anthropic", "anthropic/claude-sonnet-4-6", "Claude Sonnet 4.6",
|
||||
source="OpenRouter",
|
||||
)]
|
||||
|
||||
async def fake_anthropic(_client, _api_key):
|
||||
|
|
@ -58,9 +59,10 @@ def test_model_catalog_tags_provider_values(monkeypatch):
|
|||
|
||||
response = asyncio.run(model_catalog_api.api_model_catalog(None))
|
||||
payload = json.loads(response.body.decode("utf-8"))
|
||||
values = {item["value"] for item in payload["items"]}
|
||||
items_by_value = {item["value"]: item for item in payload["items"]}
|
||||
values = set(items_by_value)
|
||||
|
||||
assert "anthropic/claude-opus" in values
|
||||
assert "anthropic/claude-sonnet-4-6" in values
|
||||
assert "openai::gpt-4.1" in values
|
||||
assert "anthropic::claude-sonnet-4-6" in values
|
||||
assert "openai-compatible::compatible-pro" in values
|
||||
|
|
@ -71,6 +73,19 @@ def test_model_catalog_tags_provider_values(monkeypatch):
|
|||
assert "minimax::MiniMax-M3" in values
|
||||
assert payload["errors"] == []
|
||||
|
||||
openrouter_item = items_by_value["anthropic/claude-sonnet-4-6"]
|
||||
direct_item = items_by_value["anthropic::claude-sonnet-4-6"]
|
||||
assert openrouter_item["source"] == "OpenRouter"
|
||||
assert openrouter_item["label"] == "OpenRouter · Claude Sonnet 4.6"
|
||||
assert direct_item["source"] == "Anthropic"
|
||||
assert direct_item["label"] == "Anthropic · Claude Sonnet 4.6"
|
||||
assert openrouter_item["label"] != direct_item["label"]
|
||||
assert all(item["label"] == f'{item["source"]} · {item["name"]}' for item in payload["items"])
|
||||
assert [
|
||||
item["value"] for item in payload["items"]
|
||||
if item["provider"] == "Anthropic" and item["name"] == "Claude Sonnet 4.6"
|
||||
] == ["anthropic/claude-sonnet-4-6", "anthropic::claude-sonnet-4-6"]
|
||||
|
||||
|
||||
def test_model_catalog_returns_errors_nonfatally(monkeypatch):
|
||||
monkeypatch.setattr(model_catalog_api, "load_settings", lambda: {
|
||||
|
|
|
|||
60
tests/test_openrouter_attribution.py
Normal file
60
tests/test_openrouter_attribution.py
Normal file
|
|
@ -0,0 +1,60 @@
|
|||
from __future__ import annotations
|
||||
|
||||
import urllib.request
|
||||
|
||||
from ouroboros.openrouter_attribution import OPENROUTER_APP_HEADERS
|
||||
from scripts.run_external_review import _probe_model_for_key
|
||||
|
||||
|
||||
def test_external_review_probe_uses_canonical_app_attribution(monkeypatch):
|
||||
import httpx
|
||||
|
||||
captured = {}
|
||||
|
||||
class Response:
|
||||
status_code = 200
|
||||
|
||||
@staticmethod
|
||||
def json():
|
||||
return {"choices": [{"message": {"content": "pong"}}]}
|
||||
|
||||
def post(_url, **kwargs):
|
||||
captured.update(kwargs)
|
||||
return Response()
|
||||
|
||||
monkeypatch.setattr(httpx, "post", post)
|
||||
|
||||
assert _probe_model_for_key("secret", "openai/gpt-5.6-sol")[0] is True
|
||||
assert {
|
||||
key: captured["headers"][key] for key in OPENROUTER_APP_HEADERS
|
||||
} == OPENROUTER_APP_HEADERS
|
||||
|
||||
|
||||
def test_gaia_judge_uses_canonical_app_attribution(monkeypatch):
|
||||
import devtools.benchmarks.gaia.audit_leakage as audit
|
||||
|
||||
captured = {}
|
||||
|
||||
class Response:
|
||||
def __enter__(self):
|
||||
return self
|
||||
|
||||
def __exit__(self, *_args):
|
||||
return False
|
||||
|
||||
def read(self):
|
||||
return b'{"choices":[{"message":{"content":"{\\"verdict\\":\\"clean\\",\\"rationale\\":\\"ok\\"}"}}]}'
|
||||
|
||||
def urlopen(req, timeout=0):
|
||||
captured["headers"] = req.headers
|
||||
return Response()
|
||||
|
||||
monkeypatch.setattr(urllib.request, "urlopen", urlopen)
|
||||
|
||||
result = audit._judge("sample", "answer", [], "openai/gpt-5.6-luna", "secret")
|
||||
assert result["verdict"] == "clean"
|
||||
actual_headers = {key.lower(): value for key, value in captured["headers"].items()}
|
||||
assert all(
|
||||
actual_headers.get(key.lower()) == value
|
||||
for key, value in OPENROUTER_APP_HEADERS.items()
|
||||
)
|
||||
140
tests/test_owner_control_read_carve.py
Normal file
140
tests/test_owner_control_read_carve.py
Normal file
|
|
@ -0,0 +1,140 @@
|
|||
"""Read-carve for the owner-control mention detectors (family-wide).
|
||||
|
||||
The scope-floor guard adjudicated the contract at v6.80.0: naming an owner
|
||||
key/endpoint blocks UNLESS the whole command line is demonstrably read-only
|
||||
inspection. The other six family members (runtime mode, context mode, safety
|
||||
mode, skill attestation, mutative toggle, evolution controls) stayed
|
||||
read-blind, deterministically blocking the product's own mandated inspection
|
||||
flows — ``grep OUROBOROS_SAFETY_MODE data/settings.json``, the reuse-first
|
||||
callers matrix (``rg ouroboros.config.save_settings``), route inspection
|
||||
(``rg /api/owner/safety-mode``) — in every runtime mode. These tests pin the
|
||||
family-wide carve: pure-read heads pass, every write shape and every
|
||||
non-inspection head (interpreter, HTTP client) still blocks, and a caller
|
||||
that cannot supply the write-shape fact stays fail-closed by default.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pathlib
|
||||
|
||||
import pytest
|
||||
|
||||
pytestmark = pytest.mark.serial
|
||||
|
||||
from ouroboros.tools.registry import (
|
||||
ToolContext,
|
||||
ToolRegistry,
|
||||
_detect_context_mode_self_lowering,
|
||||
_detect_evolution_owner_control_self_change,
|
||||
_detect_mutative_toggle_self_change,
|
||||
_detect_owner_skill_attest_self_call,
|
||||
_detect_runtime_mode_elevation,
|
||||
_detect_safety_mode_self_lowering,
|
||||
)
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _home_outside_tmp(tmp_path, monkeypatch):
|
||||
fake_home = tmp_path / "_home"
|
||||
fake_home.mkdir(exist_ok=True)
|
||||
monkeypatch.setattr(pathlib.Path, "home", lambda: fake_home)
|
||||
|
||||
|
||||
DETECTOR_READ_WRITE_CASES = [
|
||||
(
|
||||
_detect_runtime_mode_elevation,
|
||||
"grep -rn ouroboros.config.save_settings ouroboros/",
|
||||
"python -c \"from ouroboros.config import save_settings; save_settings({'OUROBOROS_RUNTIME_MODE': 'pro'})\"",
|
||||
),
|
||||
(
|
||||
_detect_context_mode_self_lowering,
|
||||
"rg /api/owner/context-mode ouroboros/gateway",
|
||||
"curl -X POST http://127.0.0.1:8765/api/owner/context-mode -d '{\"mode\":\"low\"}'",
|
||||
),
|
||||
(
|
||||
_detect_safety_mode_self_lowering,
|
||||
"grep OUROBOROS_SAFETY_MODE data/settings.json",
|
||||
"python -c \"import httpx; httpx.request('POST','http://127.0.0.1:8765/api/owner/safety-mode',json={'mode':'off'})\"",
|
||||
),
|
||||
(
|
||||
_detect_owner_skill_attest_self_call,
|
||||
"rg '/api/owner/skills/.+/attest-review' ouroboros/gateway",
|
||||
"curl -X POST http://127.0.0.1:8765/api/owner/skills/foo/attest-review",
|
||||
),
|
||||
(
|
||||
_detect_mutative_toggle_self_change,
|
||||
"grep OUROBOROS_ALLOW_MUTATIVE_SUBAGENTS data/settings.json",
|
||||
"python -c \"import json,pathlib; p=pathlib.Path('data/settings.json'); d=json.loads(p.read_text()); d['OUROBOROS_ALLOW_MUTATIVE_SUBAGENTS']='true'; p.write_text(json.dumps(d))\"",
|
||||
),
|
||||
(
|
||||
_detect_evolution_owner_control_self_change,
|
||||
"grep OUROBOROS_POST_TASK_EVOLUTION data/settings.json",
|
||||
"sh -c \"ouroboros settings set OUROBOROS_POST_TASK_EVOLUTION true\"",
|
||||
),
|
||||
]
|
||||
|
||||
|
||||
@pytest.mark.parametrize("detector, read_cmd, write_cmd", DETECTOR_READ_WRITE_CASES)
|
||||
def test_pure_read_inspection_passes_and_mutation_blocks(detector, read_cmd, write_cmd):
|
||||
# Pure read-only inspection with no write shape passes.
|
||||
assert detector(read_cmd.lower(), writeish=False) is False
|
||||
# The same read shape carrying a write-shape fact stays fail-closed.
|
||||
assert detector(read_cmd.lower(), writeish=True) is True
|
||||
# A mutation shape blocks regardless of the writeish fact: interpreters and
|
||||
# HTTP clients are not inspection heads.
|
||||
assert detector(write_cmd.lower(), writeish=False) is True
|
||||
|
||||
|
||||
@pytest.mark.parametrize("detector, read_cmd, _write_cmd", DETECTOR_READ_WRITE_CASES)
|
||||
def test_default_stays_fail_closed_without_the_fact(detector, read_cmd, _write_cmd):
|
||||
# A caller that cannot supply the write-shape fact keeps the pre-carve
|
||||
# fail-closed behavior (writeish defaults True).
|
||||
assert detector(read_cmd.lower()) is True
|
||||
|
||||
|
||||
def test_inplace_editor_on_allowlisted_head_is_not_pure_read():
|
||||
"""sol review: yq is a read head, but `yq -i` edits the named file in place.
|
||||
A settings mutation through an in-place editor must NOT be exempted as
|
||||
inspection (jq has no in-place edit and stays a stdout-only read)."""
|
||||
from ouroboros.tools.registry import _is_pure_read_inspection
|
||||
assert _is_pure_read_inspection("yq -i '.ouroboros_safety_mode = \"off\"' /x/data/settings.json".lower()) is False
|
||||
assert _is_pure_read_inspection("yq --inplace '.ouroboros_context_mode = \"low\"' /x/data/settings.json".lower()) is False
|
||||
# yq WITHOUT -i and jq (no in-place) stay reads.
|
||||
assert _is_pure_read_inspection("yq '.ouroboros_safety_mode' /x/data/settings.json".lower()) is True
|
||||
assert _is_pure_read_inspection("jq '.ouroboros_safety_mode' /x/data/settings.json".lower()) is True
|
||||
# The owner-control detectors therefore block the in-place edit.
|
||||
assert _detect_safety_mode_self_lowering(
|
||||
"yq -i '.ouroboros_safety_mode = \"off\"' /x/data/settings.json".lower(), writeish=False
|
||||
) is True
|
||||
assert _detect_context_mode_self_lowering(
|
||||
"yq -i '.ouroboros_context_mode = \"low\"' /x/data/settings.json".lower(), writeish=False
|
||||
) is True
|
||||
|
||||
|
||||
def test_registry_level_read_carve_end_to_end(tmp_path):
|
||||
system = tmp_path / "system"
|
||||
data = tmp_path / "data"
|
||||
for p in (system, data):
|
||||
p.mkdir()
|
||||
(data / "settings.json").write_text("{}", encoding="utf-8")
|
||||
reg = ToolRegistry(repo_dir=system, drive_root=data)
|
||||
reg.set_context(ToolContext(repo_dir=system, drive_root=data, task_id="carve-test"))
|
||||
|
||||
# Route inspection of an owner endpoint in the repo source: allowed.
|
||||
out = reg._run_shell_safety_check(
|
||||
{"cmd": ["rg", "/api/owner/safety-mode", str(system)], "cwd": str(system)}, "advanced"
|
||||
)
|
||||
assert out is None
|
||||
# An HTTP client naming the same endpoint: blocked, whatever verb it spells.
|
||||
out = reg._run_shell_safety_check(
|
||||
{
|
||||
"cmd": [
|
||||
"python3",
|
||||
"-c",
|
||||
"import httpx; httpx.request('POST','http://127.0.0.1:8765/api/owner/safety-mode',json={'mode':'off'})",
|
||||
],
|
||||
"cwd": str(system),
|
||||
},
|
||||
"advanced",
|
||||
)
|
||||
assert "SAFETY_MODE_SELF_LOWERING_BLOCKED" in (out or "")
|
||||
|
|
@ -1028,7 +1028,17 @@ def test_run_shell_blocks_sort_uniq_protected_output_paths(cmd, tmp_path, monkey
|
|||
assert "BIBLE.md" in result or "protected" in result.lower()
|
||||
|
||||
|
||||
@pytest.mark.parametrize("cmd", ["cat BIBLE.md", "git diff BIBLE.md", "du BIBLE.md"])
|
||||
@pytest.mark.parametrize("cmd", [
|
||||
"cat BIBLE.md",
|
||||
"git diff BIBLE.md",
|
||||
"du BIBLE.md",
|
||||
# Mode-aware write shape (write_shape.py): pure-filter reads and prose-word
|
||||
# inspection lines are reads, not "modifications" of the mentioned file.
|
||||
"sed -n '1,40p' BIBLE.md",
|
||||
"grep -n delete ouroboros/safety.py",
|
||||
"rg truncate ouroboros/runtime_mode_policy.py",
|
||||
"python3 -c \"print(open('ouroboros/safety.py').read())\"",
|
||||
])
|
||||
def test_run_shell_allows_readonly_mentions_of_protected_paths(cmd, tmp_path, monkeypatch):
|
||||
monkeypatch.setenv("OUROBOROS_RUNTIME_MODE", "advanced")
|
||||
reg = _registry(tmp_path)
|
||||
|
|
|
|||
|
|
@ -290,7 +290,9 @@ def test_open_based_pure_read_not_blocked_despite_coarse_writeish(tmp_path):
|
|||
and let the pure read through (the original GAIA class)."""
|
||||
outside = tmp_path / "data" / "logs" / "events.jsonl"
|
||||
cmd = f"python3 -c \"print(open('{outside}').read())\""
|
||||
# Registry computes writeish=True for this command (the `open(` token).
|
||||
# The registry's workspace composition now computes writeish=False for this
|
||||
# shape (interpreter_write_shape is mode-aware); this lane must keep letting
|
||||
# the pure read through even when a caller still hands it the coarse fact.
|
||||
assert _guard(tmp_path, cmd, writeish=True) == []
|
||||
|
||||
|
||||
|
|
|
|||
|
|
@ -98,6 +98,56 @@ def _chat_bubble_count(page) -> int:
|
|||
return page.locator(".chat-bubble").count()
|
||||
|
||||
|
||||
def _menu_metrics(page, menu, trigger) -> dict:
|
||||
return page.evaluate(
|
||||
"""
|
||||
([menu, trigger]) => {
|
||||
const menuRect = menu.getBoundingClientRect();
|
||||
const triggerRect = trigger.getBoundingClientRect();
|
||||
const hitItems = Array.from(menu.querySelectorAll('.task-control-item')).map((item) => {
|
||||
const rect = item.getBoundingClientRect();
|
||||
const hit = document.elementFromPoint(
|
||||
rect.left + rect.width / 2,
|
||||
rect.top + rect.height / 2,
|
||||
);
|
||||
return Boolean(hit && item.contains(hit));
|
||||
});
|
||||
return {
|
||||
parentIsBody: menu.parentElement === document.body,
|
||||
position: getComputedStyle(menu).position,
|
||||
viewport: { width: window.innerWidth, height: window.innerHeight },
|
||||
menu: {
|
||||
top: menuRect.top, right: menuRect.right,
|
||||
bottom: menuRect.bottom, left: menuRect.left,
|
||||
},
|
||||
trigger: {
|
||||
top: triggerRect.top, right: triggerRect.right,
|
||||
bottom: triggerRect.bottom, left: triggerRect.left,
|
||||
},
|
||||
hitItems,
|
||||
};
|
||||
}
|
||||
""",
|
||||
[menu.element_handle(), trigger.element_handle()],
|
||||
)
|
||||
|
||||
|
||||
def _assert_menu_geometry(metrics: dict, *, placement: str | None = None) -> None:
|
||||
assert metrics["parentIsBody"] is True
|
||||
assert metrics["position"] == "fixed"
|
||||
assert all(metrics["hitItems"]), metrics
|
||||
rect = metrics["menu"]
|
||||
viewport = metrics["viewport"]
|
||||
assert rect["left"] >= 7.5, metrics
|
||||
assert rect["right"] <= viewport["width"] - 7.5, metrics
|
||||
assert rect["top"] >= 7.5, metrics
|
||||
assert rect["bottom"] <= viewport["height"] - 7.5, metrics
|
||||
if placement == "below":
|
||||
assert rect["top"] >= metrics["trigger"]["bottom"] + 3, metrics
|
||||
elif placement == "above":
|
||||
assert rect["bottom"] <= metrics["trigger"]["top"] - 3, metrics
|
||||
|
||||
|
||||
@pytest.mark.ui_browser
|
||||
def test_s3_chat_card_dropdown_hurry_and_soft_stop(direct_server_with_data):
|
||||
"""Chat surface: frozen dropdown, no-chat hurry with idempotent request_id
|
||||
|
|
@ -124,6 +174,28 @@ def test_s3_chat_card_dropdown_hurry_and_soft_stop(direct_server_with_data):
|
|||
browser = pw.chromium.launch(headless=True)
|
||||
page = browser.new_page(viewport={"width": 1440, "height": 1000})
|
||||
try:
|
||||
# Retain each observer callback so the singleton-generation
|
||||
# guard can be tested against a late delivery from a closed
|
||||
# predecessor. Native observation still drives the page.
|
||||
page.add_init_script(
|
||||
"""
|
||||
(() => {
|
||||
const NativeIntersectionObserver = window.IntersectionObserver;
|
||||
window.__s3IntersectionObservers = [];
|
||||
window.IntersectionObserver = class {
|
||||
constructor(callback, options) {
|
||||
this.callback = callback;
|
||||
this.native = new NativeIntersectionObserver(callback, options);
|
||||
window.__s3IntersectionObservers.push(this);
|
||||
}
|
||||
observe(target) { return this.native.observe(target); }
|
||||
unobserve(target) { return this.native.unobserve(target); }
|
||||
disconnect() { return this.native.disconnect(); }
|
||||
takeRecords() { return this.native.takeRecords(); }
|
||||
};
|
||||
})();
|
||||
""",
|
||||
)
|
||||
page.goto(url, wait_until="domcontentloaded", timeout=30_000)
|
||||
live = page.locator('.chat-live-card[data-task-id="live-root"]')
|
||||
live.wait_for(state="attached", timeout=30_000)
|
||||
|
|
@ -136,14 +208,220 @@ def test_s3_chat_card_dropdown_hurry_and_soft_stop(direct_server_with_data):
|
|||
# 1) The dropdown offers EXACTLY the three frozen actions, in
|
||||
# order and verbatim; Escape dismisses = the run continues.
|
||||
trigger.click()
|
||||
menu = live.locator(".task-control-menu")
|
||||
menu = page.locator("body > .task-control-menu")
|
||||
menu.wait_for(state="visible", timeout=10_000)
|
||||
assert _menu_labels(menu) == EXPECTED_ACTIONS
|
||||
_assert_menu_geometry(_menu_metrics(page, menu, trigger), placement="below")
|
||||
page.screenshot(
|
||||
path=str(data_dir.parent / "s3-chat-control-open.png"), full_page=False,
|
||||
)
|
||||
page.keyboard.press("Escape")
|
||||
menu.wait_for(state="detached", timeout=10_000)
|
||||
assert trigger.is_enabled()
|
||||
assert trigger.get_attribute("aria-expanded") == "false"
|
||||
assert trigger.evaluate("(node) => document.activeElement === node") is True
|
||||
assert page.evaluate("() => window.__s3Calls.length") == 0
|
||||
|
||||
# Owner decision 2A survives the browser's default pointer
|
||||
# focus without cancelling the outside element's actual click.
|
||||
page.evaluate(
|
||||
"""
|
||||
() => {
|
||||
const outside = document.createElement('button');
|
||||
outside.id = 's3-outside-focus-target';
|
||||
outside.textContent = 'Outside';
|
||||
Object.assign(outside.style, {
|
||||
position: 'fixed', left: '8px', top: '8px', zIndex: '200',
|
||||
});
|
||||
window.__s3OutsideClicks = 0;
|
||||
outside.addEventListener('click', () => { window.__s3OutsideClicks += 1; });
|
||||
document.body.appendChild(outside);
|
||||
}
|
||||
""",
|
||||
)
|
||||
trigger.click()
|
||||
menu.wait_for(state="visible", timeout=10_000)
|
||||
page.locator("#s3-outside-focus-target").click()
|
||||
menu.wait_for(state="detached", timeout=10_000)
|
||||
page.wait_for_function(
|
||||
"() => document.activeElement?.matches('[data-cancel-run]')",
|
||||
timeout=10_000,
|
||||
)
|
||||
assert page.evaluate("() => window.__s3OutsideClicks") == 1
|
||||
assert page.evaluate("() => window.__s3Calls.length") == 0
|
||||
page.locator("#s3-outside-focus-target").evaluate("node => node.remove()")
|
||||
|
||||
# A queued callback from observer A must not close successor B.
|
||||
observer_count = page.evaluate("() => window.__s3IntersectionObservers.length")
|
||||
trigger.click()
|
||||
menu.wait_for(state="visible", timeout=10_000)
|
||||
page.keyboard.press("Escape")
|
||||
menu.wait_for(state="detached", timeout=10_000)
|
||||
trigger.click()
|
||||
menu.wait_for(state="visible", timeout=10_000)
|
||||
page.evaluate(
|
||||
"""
|
||||
([index, trigger]) => {
|
||||
window.__s3IntersectionObservers[index].callback([{
|
||||
target: trigger, isIntersecting: false,
|
||||
}]);
|
||||
}
|
||||
""",
|
||||
[observer_count, trigger.element_handle()],
|
||||
)
|
||||
page.wait_for_timeout(50)
|
||||
assert menu.count() == 1
|
||||
assert trigger.get_attribute("aria-expanded") == "true"
|
||||
page.keyboard.press("Escape")
|
||||
menu.wait_for(state="detached", timeout=10_000)
|
||||
|
||||
# The stable Chat DOM owns the deterministic viewport-flip proof.
|
||||
page.evaluate(
|
||||
"""
|
||||
() => {
|
||||
const live = document.querySelector(
|
||||
'.chat-live-card[data-task-id="live-root"]',
|
||||
);
|
||||
window.__s3OriginalLiveStyle = live.getAttribute('style');
|
||||
Object.assign(live.style, {
|
||||
position: 'fixed', left: '16px', right: 'auto',
|
||||
bottom: '8px', top: 'auto', margin: '0', zIndex: '6',
|
||||
});
|
||||
}
|
||||
""",
|
||||
)
|
||||
trigger.click()
|
||||
menu.wait_for(state="visible", timeout=10_000)
|
||||
_assert_menu_geometry(_menu_metrics(page, menu, trigger), placement="above")
|
||||
page.screenshot(
|
||||
path=str(data_dir.parent / "s3-chat-control-flipped.png"), full_page=False,
|
||||
)
|
||||
page.keyboard.press("Escape")
|
||||
menu.wait_for(state="detached", timeout=10_000)
|
||||
page.evaluate(
|
||||
"""
|
||||
() => {
|
||||
const live = document.querySelector(
|
||||
'.chat-live-card[data-task-id="live-root"]',
|
||||
);
|
||||
if (window.__s3OriginalLiveStyle === null) live.removeAttribute('style');
|
||||
else live.setAttribute('style', window.__s3OriginalLiveStyle);
|
||||
}
|
||||
""",
|
||||
)
|
||||
|
||||
# Owner decision 1D: any nested scroll or resize dismisses
|
||||
# without issuing a task-control action.
|
||||
trigger.click()
|
||||
menu.wait_for(state="visible", timeout=10_000)
|
||||
page.locator("#chat-messages").dispatch_event("scroll")
|
||||
menu.wait_for(state="detached", timeout=10_000)
|
||||
assert trigger.get_attribute("aria-expanded") == "false"
|
||||
assert trigger.evaluate("(node) => document.activeElement === node") is True
|
||||
assert page.evaluate("() => window.__s3Calls.length") == 0
|
||||
|
||||
trigger.click()
|
||||
menu.wait_for(state="visible", timeout=10_000)
|
||||
page.set_viewport_size({"width": 1360, "height": 920})
|
||||
menu.wait_for(state="detached", timeout=10_000)
|
||||
assert trigger.get_attribute("aria-expanded") == "false"
|
||||
assert trigger.evaluate("(node) => document.activeElement === node") is True
|
||||
assert page.evaluate("() => window.__s3Calls.length") == 0
|
||||
|
||||
# Narrow viewports must shrink the menu instead of allowing the
|
||||
# historical 220px minimum to overflow the page.
|
||||
page.set_viewport_size({"width": 320, "height": 700})
|
||||
trigger.click()
|
||||
menu.wait_for(state="visible", timeout=10_000)
|
||||
_assert_menu_geometry(_menu_metrics(page, menu, trigger))
|
||||
page.screenshot(
|
||||
path=str(data_dir.parent / "s3-chat-control-narrow.png"), full_page=False,
|
||||
)
|
||||
page.keyboard.press("Escape")
|
||||
menu.wait_for(state="detached", timeout=10_000)
|
||||
|
||||
# A genuinely short viewport height-clamps the portal. Its own
|
||||
# scroll must keep it open so every action remains reachable;
|
||||
# only scroll that can stale anchor geometry invokes 1D close.
|
||||
page.set_viewport_size({"width": 320, "height": 150})
|
||||
page.evaluate(
|
||||
"""
|
||||
() => {
|
||||
const live = document.querySelector(
|
||||
'.chat-live-card[data-task-id="live-root"]',
|
||||
);
|
||||
window.__s3ShortLiveStyle = live.getAttribute('style');
|
||||
Object.assign(live.style, {
|
||||
position: 'fixed', left: '8px', right: 'auto',
|
||||
top: '0px', bottom: 'auto', width: '304px',
|
||||
margin: '0', zIndex: '6',
|
||||
});
|
||||
const trigger = live.querySelector('[data-cancel-run]');
|
||||
live.style.top = `${70 - trigger.getBoundingClientRect().top}px`;
|
||||
}
|
||||
""",
|
||||
)
|
||||
trigger.click()
|
||||
menu.wait_for(state="visible", timeout=10_000)
|
||||
short_metrics = menu.evaluate(
|
||||
"""
|
||||
node => ({
|
||||
clientHeight: node.clientHeight,
|
||||
scrollHeight: node.scrollHeight,
|
||||
top: node.getBoundingClientRect().top,
|
||||
bottom: node.getBoundingClientRect().bottom,
|
||||
})
|
||||
""",
|
||||
)
|
||||
assert short_metrics["scrollHeight"] > short_metrics["clientHeight"], short_metrics
|
||||
assert short_metrics["top"] >= 7.5, short_metrics
|
||||
assert short_metrics["bottom"] <= 142.5, short_metrics
|
||||
items = menu.locator(".task-control-item")
|
||||
for index in range(items.count()):
|
||||
menu.evaluate(
|
||||
"""
|
||||
(node, itemIndex) => {
|
||||
node.scrollTop = node.querySelectorAll('.task-control-item')[itemIndex]
|
||||
.offsetTop;
|
||||
}
|
||||
""",
|
||||
index,
|
||||
)
|
||||
page.wait_for_timeout(50)
|
||||
assert menu.count() == 1
|
||||
assert trigger.get_attribute("aria-expanded") == "true"
|
||||
assert items.nth(index).evaluate(
|
||||
"""
|
||||
item => {
|
||||
const rect = item.getBoundingClientRect();
|
||||
const hit = document.elementFromPoint(
|
||||
rect.left + rect.width / 2,
|
||||
rect.top + rect.height / 2,
|
||||
);
|
||||
return Boolean(hit && item.contains(hit));
|
||||
}
|
||||
""",
|
||||
) is True
|
||||
assert menu.evaluate("node => node.scrollTop") > 0
|
||||
page.screenshot(
|
||||
path=str(data_dir.parent / "s3-chat-control-short-scroll.png"),
|
||||
full_page=False,
|
||||
)
|
||||
page.keyboard.press("Escape")
|
||||
menu.wait_for(state="detached", timeout=10_000)
|
||||
page.evaluate(
|
||||
"""
|
||||
() => {
|
||||
const live = document.querySelector(
|
||||
'.chat-live-card[data-task-id="live-root"]',
|
||||
);
|
||||
if (window.__s3ShortLiveStyle === null) live.removeAttribute('style');
|
||||
else live.setAttribute('style', window.__s3ShortLiveStyle);
|
||||
}
|
||||
""",
|
||||
)
|
||||
page.set_viewport_size({"width": 1440, "height": 1000})
|
||||
|
||||
# 2) "Hurry up": typed control + LOCAL toast; the chat DOM
|
||||
# gains ZERO bubbles (HQ1's no-chat contract, live DOM).
|
||||
trigger.click()
|
||||
|
|
@ -272,14 +550,34 @@ def test_s3_activity_tab_dropdown_parity_and_no_chat_hurry(direct_server_with_da
|
|||
|
||||
# Parity: the same shared menu with the exact frozen actions.
|
||||
row_btn.click()
|
||||
menu = page.locator("#dashboard-panel-activity .task-control-menu")
|
||||
menu = page.locator("body > .task-control-menu")
|
||||
menu.wait_for(state="visible", timeout=10_000)
|
||||
assert _menu_labels(menu) == EXPECTED_ACTIONS
|
||||
assert menu.evaluate("(node) => node.parentElement === document.body") is True
|
||||
assert menu.evaluate("(node) => getComputedStyle(node).position") == "fixed"
|
||||
page.screenshot(
|
||||
path=str(data_dir.parent / "s3-activity-control-open.png"), full_page=True,
|
||||
)
|
||||
|
||||
# Activity refresh replaces its subtree before awaiting network
|
||||
# reads. The trigger observer must dispose the body portal.
|
||||
page.evaluate(
|
||||
"""
|
||||
() => window.dispatchEvent(new CustomEvent(
|
||||
'ouro:dashboard-subtab-shown', { detail: { tab: 'activity' } },
|
||||
))
|
||||
""",
|
||||
)
|
||||
menu.wait_for(state="detached", timeout=10_000)
|
||||
assert page.evaluate("() => window.__s3Calls.length") == 0
|
||||
row_btn.wait_for(state="visible", timeout=30_000)
|
||||
|
||||
# "Hurry up" from Activity: local toast only, no chat bubble.
|
||||
# Baseline captured HERE (post-navigation) so async startup
|
||||
# bubbles (awaken banner, history replay) are already settled.
|
||||
bubbles_before = _chat_bubble_count(page)
|
||||
row_btn.click()
|
||||
menu.wait_for(state="visible", timeout=10_000)
|
||||
menu.locator('[data-task-control="hurry"]').click()
|
||||
page.locator(".toast", has_text="Hurry up: accepted").wait_for(
|
||||
state="visible", timeout=10_000,
|
||||
|
|
|
|||
|
|
@ -67,6 +67,10 @@ def test_runtime_policy_blocks_are_semantic_tool_failures():
|
|||
("run_command", "⚠️ SHELL_CWD_BLOCKED: cwd escapes allowed roots.", "cwd_blocked"),
|
||||
("run_script", "⚠️ RUN_SCRIPT_BLOCKED: interpreter must be one of ['python3'].", "run_script_blocked"),
|
||||
("run_command", "⚠️ WORKSPACE_SHELL_BLOCKED: write-like shell command mentions Ouroboros system/data paths.", "workspace_blocked"),
|
||||
# The path/route-naming message shape production emits since the mode-aware
|
||||
# write-shape fix (guard B names the resolved offending path and the route).
|
||||
("run_command", "⚠️ WORKSPACE_SHELL_BLOCKED: write-like shell command mentions Ouroboros system/data paths. Blocked path: /x/data/y. Use the gated read_file/write_file tools for runtime data.", "workspace_blocked"),
|
||||
("run_command", "⚠️ WORKSPACE_SHELL_BLOCKED: write-like shell commands may not target paths outside the selected process root. Blocked path: /outside/z. Selected process root: /app.", "workspace_blocked"),
|
||||
("run_command", "⚠️ ELEVATION_BLOCKED: shell command pattern looks like an elevation attempt.", "elevation_blocked"),
|
||||
("run_command", "⚠️ SKILL_STATE_WRITE_BLOCKED: skill trust state is owner controlled.", "skill_state_blocked"),
|
||||
("run_command", "⚠️ ARTIFACT_OUTPUT_ERROR: command succeeded but declared output registration failed.", "artifact_output_error"),
|
||||
|
|
|
|||
|
|
@ -3680,7 +3680,7 @@ def test_ui_smoke_cancel_run_button_eligibility_and_cancelled_state(direct_serve
|
|||
assert "cancelled" in (gone_phase.get_attribute("class") or "")
|
||||
# Dropdown wiring (S3 Q2): open, then dismiss = keep running.
|
||||
cancel_btn.click()
|
||||
menu = live.locator('.task-control-menu')
|
||||
menu = page.locator('body > .task-control-menu')
|
||||
menu.wait_for(state="visible", timeout=10_000)
|
||||
assert "Wrap up" in menu.inner_text()
|
||||
page.keyboard.press("Escape")
|
||||
|
|
|
|||
436
tests/test_workspace_write_shape.py
Normal file
436
tests/test_workspace_write_shape.py
Normal file
|
|
@ -0,0 +1,436 @@
|
|||
"""Mode-aware write-shape classification for interpreter shell commands.
|
||||
|
||||
The coarse ``open(`` token marks a read-only ``open(p, 'rb')`` as writeish.
|
||||
The light-mode runtime_data lane already re-judges that class mode-aware
|
||||
("the original GAIA class", tests/test_runtime_reliability_v655.py); the
|
||||
workspace write guard's ``writeish`` composition did not, so a pure-read
|
||||
hash/compare one-liner in an external workspace was refused as a
|
||||
"write-like shell command" — a false reason with no route. These tests pin
|
||||
the mode-aware composition: pure interpreter reads are not write-shaped,
|
||||
every real write shape still is, and the runtime/secret READ policy for
|
||||
external workspaces stays intact via its own honest guard.
|
||||
"""
|
||||
|
||||
from __future__ import annotations
|
||||
|
||||
import pathlib
|
||||
|
||||
import pytest
|
||||
|
||||
# Shares the real-subprocess-adjacent registry harness with
|
||||
# test_external_workspace_access.py; keep it in the serial lane.
|
||||
pytestmark = pytest.mark.serial
|
||||
|
||||
from ouroboros.tools.registry import ToolContext, ToolRegistry
|
||||
from ouroboros.tools.shell_guards import interpreter_write_shape, shell_has_write_indicator
|
||||
|
||||
|
||||
@pytest.fixture(autouse=True)
|
||||
def _home_outside_tmp(tmp_path, monkeypatch):
|
||||
# Same premise as test_external_workspace_access.py: keep tmp scratch
|
||||
# outside $HOME on every platform so host-scratch reads stay non-runtime.
|
||||
fake_home = tmp_path / "_home"
|
||||
fake_home.mkdir(exist_ok=True)
|
||||
monkeypatch.setattr(pathlib.Path, "home", lambda: fake_home)
|
||||
|
||||
|
||||
def _registry(tmp_path: pathlib.Path, *, mode: str = "external") -> ToolRegistry:
|
||||
system = tmp_path / "system"
|
||||
workspace = tmp_path / "workspace"
|
||||
data = tmp_path / "data"
|
||||
for p in (system, workspace, data):
|
||||
p.mkdir(exist_ok=True)
|
||||
reg = ToolRegistry(repo_dir=system, drive_root=data)
|
||||
reg.set_context(
|
||||
ToolContext(
|
||||
repo_dir=system,
|
||||
drive_root=data,
|
||||
workspace_root=workspace,
|
||||
workspace_mode=mode,
|
||||
task_id="task-write-shape",
|
||||
)
|
||||
)
|
||||
return reg
|
||||
|
||||
|
||||
READ_ONLY_HASH_SCRIPT = (
|
||||
"import hashlib\n"
|
||||
"def h(p):\n"
|
||||
" with open(p, 'rb') as f:\n"
|
||||
" return hashlib.sha256(f.read()).hexdigest()\n"
|
||||
"print(h({target!r}))\n"
|
||||
)
|
||||
|
||||
|
||||
# --- unit layer: the classifier itself -------------------------------------
|
||||
|
||||
|
||||
def test_read_only_open_is_not_interpreter_write_shape():
|
||||
cmd = ["python3", "-c", "with open('f.bin', 'rb') as f:\n print(len(f.read()))"]
|
||||
assert interpreter_write_shape(cmd) is False
|
||||
# The legacy coarse classifier keeps its pinned behavior for its other
|
||||
# consumers (_protected_shell_block, ws5 carryover).
|
||||
assert shell_has_write_indicator(cmd) is True
|
||||
|
||||
|
||||
def test_write_mode_open_and_pathlib_open_stay_write_shaped():
|
||||
assert interpreter_write_shape(["python3", "-c", "open('/d/x', 'w').write('hi')"]) is True
|
||||
assert (
|
||||
interpreter_write_shape(
|
||||
["python3", "-c", "from pathlib import Path; Path('/d/x').open('w')"]
|
||||
)
|
||||
is True
|
||||
)
|
||||
|
||||
|
||||
def test_opaque_subprocess_and_library_saves_stay_write_shaped():
|
||||
assert (
|
||||
interpreter_write_shape(
|
||||
["python3", "-c", "import subprocess; subprocess.run(['rm', '-rf', '/d/x'])"]
|
||||
)
|
||||
is True
|
||||
)
|
||||
assert interpreter_write_shape(["python3", "-c", "df.to_csv('out.csv')"]) is True
|
||||
assert interpreter_write_shape(["python3", "-c", "fh.writelines(rows)"]) is True
|
||||
assert (
|
||||
interpreter_write_shape(
|
||||
["node", "-e", "const {writeFileSync} = require('fs'); writeFileSync('x', 'y')"]
|
||||
)
|
||||
is True
|
||||
)
|
||||
|
||||
|
||||
def test_shell_level_signals_still_write_shaped_for_interpreters():
|
||||
assert interpreter_write_shape(["sh", "-c", "python3 -c 'print(1)' && rm -rf /tmp/x"]) is True
|
||||
assert interpreter_write_shape("python3 gen.py > out.txt") is True
|
||||
assert interpreter_write_shape(["sh", "-c", "python3 gen.py && cp out.txt /tmp/y"]) is True
|
||||
|
||||
|
||||
def test_ruby_perl_pure_reads_are_not_write_shaped():
|
||||
"""LIGHT_SHELL_WRITER_COMMANDS membership (a coarse shell-writer role) must not
|
||||
re-add the write shape the mode-aware classifier just re-judged."""
|
||||
assert interpreter_write_shape(["ruby", "-e", "puts File.read('/tmp/f.txt')"]) is False
|
||||
# perl 3-arg READ open: the filename's own letters (the 'x' in f.txt) must not
|
||||
# classify the mode.
|
||||
assert (
|
||||
interpreter_write_shape(["perl", "-e", "open(my $fh, '<', '/tmp/f.txt'); print <$fh>"])
|
||||
is False
|
||||
)
|
||||
assert interpreter_write_shape(["ruby", "-e", "File.write('/tmp/x', 'y')"]) is True
|
||||
|
||||
|
||||
def test_perl_ruby_native_write_idioms_stay_write_shaped():
|
||||
"""fable-5 review: dropping the membership floor demands the vocabulary
|
||||
actually SEE perl/ruby write spellings — '>'-mode opens, File.delete,
|
||||
FileUtils with a variable argument, IO.binwrite."""
|
||||
assert interpreter_write_shape(["perl", "-e", "open(FH,'>','/outside/x'); print FH 'data'"]) is True
|
||||
assert interpreter_write_shape(["perl", "-e", "open(FH, '>>', $log); print FH $line"]) is True
|
||||
assert interpreter_write_shape(["ruby", "-e", "File.delete('/outside/x')"]) is True
|
||||
assert interpreter_write_shape(["ruby", "-e", "f='/x'; FileUtils.rm_rf(f)"]) is True
|
||||
assert interpreter_write_shape(["ruby", "-e", "IO.binwrite('a.bin', d)"]) is True
|
||||
|
||||
|
||||
def test_ruby_file_open_is_mode_aware(tmp_path):
|
||||
"""sol review: File.open must be a write target only with a write-mode 2nd arg —
|
||||
File.open('/x','r') is a READ and must not be reported as a write, while
|
||||
File.open('/x','w') and File.new('/x','w') stay writes."""
|
||||
from ouroboros.tools.shell_guards import writer_target_tokens
|
||||
read = ["ruby", "-e", "File.open('/tmp/r','r') { |f| puts f.read }"]
|
||||
# The literal path is NOT emitted as a write target for a read-mode open.
|
||||
assert "/tmp/r" not in writer_target_tokens(read)
|
||||
assert interpreter_write_shape(read) is False
|
||||
assert interpreter_write_shape(["ruby", "-e", "File.open('/tmp/o','w') { |f| f.puts 'x' }"]) is True
|
||||
assert "/tmp/o" in writer_target_tokens(["ruby", "-e", "File.open('/tmp/o','w') { |f| f.puts 'x' }"])
|
||||
assert interpreter_write_shape(["ruby", "-e", "File.new('/tmp/o','w')"]) is True
|
||||
assert interpreter_write_shape(["ruby", "-e", "File.new('/tmp/o','r')"]) is False
|
||||
|
||||
|
||||
def test_keyword_mode_open_is_write_shaped():
|
||||
"""sol review: open('/x', mode='w') with a statically-known keyword mode is a
|
||||
real write (distinct from the disclosed variable-mode residual)."""
|
||||
assert interpreter_write_shape(["python3", "-c", "open('/tmp/o', mode='w')"]) is True
|
||||
assert interpreter_write_shape(["python3", "-c", "open('/tmp/o', mode='rb')"]) is False
|
||||
|
||||
|
||||
def test_unspaced_posix_redirect_is_write_shaped():
|
||||
"""fable-5 review: `python3 gen.py>out.txt` is ONE shlex token; the redirect
|
||||
shape must be recognized mid-token, while located inline-code bodies keep
|
||||
their '>' comparisons/filehandles as reads."""
|
||||
assert interpreter_write_shape("python3 gen.py>out.txt") is True
|
||||
assert interpreter_write_shape("python3 gen.py>>log.txt") is True
|
||||
assert interpreter_write_shape(["python3", "-c", "print(1 if a > b else 2)"]) is False
|
||||
assert interpreter_write_shape(["python3", "-c", "x = {'k': 'v => w'}; print(x)"]) is False
|
||||
|
||||
|
||||
def test_prose_words_are_not_write_shapes_for_interpreters():
|
||||
"""Natural-language words in code text ('scp done', 'count deleted rows') are
|
||||
not write evidence; structural spellings (os.remove, rm in a compound) are."""
|
||||
assert interpreter_write_shape(["python3", "-c", "print(open('/tmp/f').read()); print('scp done')"]) is False
|
||||
assert interpreter_write_shape(["python3", "-c", "print('count deleted rows'); print(open('/tmp/f').read())"]) is False
|
||||
assert interpreter_write_shape(["python3", "-c", "print('results truncated')"]) is False
|
||||
assert interpreter_write_shape(["python3", "-c", "import os; os.remove('/tmp/x')"]) is True
|
||||
assert interpreter_write_shape(["python3", "-c", "f.truncate(0)"]) is True
|
||||
|
||||
|
||||
# --- guard layer: workspace lanes ------------------------------------------
|
||||
|
||||
|
||||
def test_external_pure_read_outside_runtime_is_allowed(tmp_path):
|
||||
"""The census class (rows 1-5): a read-only interpreter hash/inspect over
|
||||
host scratch was refused as 'write-like'; it is a plain allowed read."""
|
||||
reg = _registry(tmp_path, mode="external")
|
||||
scratch = tmp_path / "scratch"
|
||||
scratch.mkdir()
|
||||
target = scratch / "artifact.bin"
|
||||
target.write_bytes(b"payload")
|
||||
cmd = ["python3", "-c", READ_ONLY_HASH_SCRIPT.format(target=str(target))]
|
||||
assert reg._run_shell_safety_check({"cmd": cmd, "cwd": str(tmp_path / "workspace")}, "advanced") is None
|
||||
|
||||
|
||||
def test_workspace_mode_pure_read_outside_root_is_allowed(tmp_path):
|
||||
reg = _registry(tmp_path, mode="workspace")
|
||||
scratch = tmp_path / "scratch"
|
||||
scratch.mkdir()
|
||||
target = scratch / "report.txt"
|
||||
target.write_text("data", encoding="utf-8")
|
||||
cmd = ["python3", "-c", f"print(open({str(target)!r}, 'r').read())"]
|
||||
assert reg._run_shell_safety_check({"cmd": cmd, "cwd": str(tmp_path / "workspace")}, "advanced") is None
|
||||
|
||||
|
||||
def test_external_sh_wrapped_pure_read_is_allowed(tmp_path):
|
||||
reg = _registry(tmp_path, mode="external")
|
||||
scratch = tmp_path / "scratch"
|
||||
scratch.mkdir()
|
||||
target = scratch / "blob.bin"
|
||||
target.write_bytes(b"x" * 16)
|
||||
inner = f"python3 -c \"print(open({str(target)!r}, 'rb').read(8))\""
|
||||
assert reg._run_shell_safety_check({"cmd": ["sh", "-c", inner], "cwd": str(tmp_path / "workspace")}, "advanced") is None
|
||||
|
||||
|
||||
def test_external_pure_read_of_runtime_still_blocked_via_read_guard(tmp_path):
|
||||
"""The owner contract stands: external shell may not READ the runtime/data
|
||||
drive — but the block now comes from the honest read guard, which names
|
||||
the gated read_file route instead of calling a read 'write-like'."""
|
||||
reg = _registry(tmp_path, mode="external")
|
||||
data = tmp_path / "data"
|
||||
(data / "settings.json").write_text("{}", encoding="utf-8")
|
||||
cmd = [
|
||||
"python3",
|
||||
"-c",
|
||||
READ_ONLY_HASH_SCRIPT.format(target=str(data / "settings.json")),
|
||||
]
|
||||
out = reg._run_shell_safety_check({"cmd": cmd, "cwd": str(tmp_path / "workspace")}, "advanced") or ""
|
||||
assert "WORKSPACE_SHELL_BLOCKED" in out
|
||||
assert "read_file" in out
|
||||
assert "write-like" not in out
|
||||
|
||||
|
||||
def test_external_write_mode_open_to_runtime_still_blocked(tmp_path):
|
||||
reg = _registry(tmp_path, mode="external")
|
||||
data = tmp_path / "data"
|
||||
cmd = ["python3", "-c", f"open({str(data / 'x')!r}, 'w').write('hi')"]
|
||||
out = reg._run_shell_safety_check({"cmd": cmd, "cwd": str(tmp_path / "workspace")}, "advanced") or ""
|
||||
assert "WORKSPACE_SHELL_BLOCKED" in out
|
||||
# A bare write-mode open with NO .write( chain (truncation alone) as well.
|
||||
bare = ["python3", "-c", f"open({str(data / 'x')!r}, 'w')"]
|
||||
out2 = reg._run_shell_safety_check({"cmd": bare, "cwd": str(tmp_path / "workspace")}, "advanced") or ""
|
||||
assert "WORKSPACE_SHELL_BLOCKED" in out2
|
||||
|
||||
|
||||
def test_external_ruby_pure_read_allowed_and_ruby_write_blocked(tmp_path):
|
||||
reg = _registry(tmp_path, mode="external")
|
||||
scratch = tmp_path / "scratch"
|
||||
scratch.mkdir()
|
||||
target = scratch / "f.txt"
|
||||
target.write_text("data", encoding="utf-8")
|
||||
read_cmd = ["ruby", "-e", f"puts File.read({str(target)!r})"]
|
||||
assert reg._run_shell_safety_check({"cmd": read_cmd, "cwd": str(tmp_path / "workspace")}, "advanced") is None
|
||||
data = tmp_path / "data"
|
||||
write_cmd = ["ruby", "-e", f"File.write({str(data / 'x')!r}, 'y')"]
|
||||
out = reg._run_shell_safety_check({"cmd": write_cmd, "cwd": str(tmp_path / "workspace")}, "advanced") or ""
|
||||
assert "WORKSPACE_SHELL_BLOCKED" in out
|
||||
|
||||
|
||||
def test_external_pathlib_write_open_to_runtime_still_blocked(tmp_path):
|
||||
reg = _registry(tmp_path, mode="external")
|
||||
data = tmp_path / "data"
|
||||
cmd = [
|
||||
"python3",
|
||||
"-c",
|
||||
f"from pathlib import Path; Path({str(data / 'x')!r}).open('w')",
|
||||
]
|
||||
out = reg._run_shell_safety_check({"cmd": cmd, "cwd": str(tmp_path / "workspace")}, "advanced") or ""
|
||||
assert "WORKSPACE_SHELL_BLOCKED" in out
|
||||
|
||||
|
||||
def test_external_opaque_subprocess_naming_runtime_still_blocked(tmp_path):
|
||||
reg = _registry(tmp_path, mode="external")
|
||||
data = tmp_path / "data"
|
||||
cmd = [
|
||||
"python3",
|
||||
"-c",
|
||||
f"import subprocess; subprocess.run(['rm', '-rf', {str(data / 'x')!r}])",
|
||||
]
|
||||
out = reg._run_shell_safety_check({"cmd": cmd, "cwd": str(tmp_path / "workspace")}, "advanced") or ""
|
||||
assert "WORKSPACE_SHELL_BLOCKED" in out
|
||||
|
||||
|
||||
def test_pure_filter_reads_outside_root_are_allowed(tmp_path):
|
||||
"""Scope-C: sort/uniq/sed -n/tar -tf/gzip -l READ invocations must not be
|
||||
'write-like' — membership alone is not a write channel."""
|
||||
reg = _registry(tmp_path, mode="external")
|
||||
scratch = tmp_path / "scratch"
|
||||
scratch.mkdir()
|
||||
data_file = scratch / "data.csv"
|
||||
data_file.write_text("b\na\n", encoding="utf-8")
|
||||
archive = scratch / "a.tar"
|
||||
archive.write_bytes(b"x" * 16)
|
||||
for cmd in (
|
||||
["sort", str(data_file)],
|
||||
["uniq", str(data_file)],
|
||||
["sed", "-n", "1p", str(data_file)],
|
||||
["tar", "-tf", str(archive)],
|
||||
["gzip", "-l", str(archive)],
|
||||
):
|
||||
out = reg._run_shell_safety_check({"cmd": cmd, "cwd": str(tmp_path / "workspace")}, "advanced")
|
||||
assert out is None, (cmd, out)
|
||||
|
||||
|
||||
def test_sed_script_write_channels_stay_write_shaped(tmp_path):
|
||||
"""fable-5 round-2: sed writes WITHOUT -i too — the POSIX in-script `w FILE`
|
||||
command, the `s///w` flag, GNU `s///e` execute, a -f script file (unprovable),
|
||||
and the GNU attached `-ibak` suffix. All must keep writer targets; plain
|
||||
filters and patterns containing prose words stay reads."""
|
||||
from ouroboros.tools.shell_guards import writer_target_tokens
|
||||
for cmd in (
|
||||
["sed", "w out.py", "f"],
|
||||
["sed", "-n", "s/a/b/w out.txt", "f"],
|
||||
["sed", "s/x/y/e", "f"],
|
||||
["sed", "-f", "script.sed", "f"],
|
||||
["sed", "-e", "w dump.txt", "f"],
|
||||
["sed", "-ibak", "s/x/y/", "f"],
|
||||
):
|
||||
assert writer_target_tokens(cmd), cmd
|
||||
for cmd in (
|
||||
["sed", "-n", "1,40p", "f"],
|
||||
["sed", "-n", "/delete/p", "f"],
|
||||
["sed", "s/hello/world/g", "f"],
|
||||
):
|
||||
assert writer_target_tokens(cmd) == [], cmd
|
||||
# e2e: the in-script write to a runtime path is refused; the same shape
|
||||
# reading host scratch passes.
|
||||
reg = _registry(tmp_path, mode="external")
|
||||
data = tmp_path / "data"
|
||||
out = reg._run_shell_safety_check(
|
||||
{"cmd": ["sed", f"w {data / 'x'}", "/etc/hostname"], "cwd": str(tmp_path / "workspace")},
|
||||
"advanced",
|
||||
) or ""
|
||||
assert "WORKSPACE_SHELL_BLOCKED" in out
|
||||
scratch = tmp_path / "scratch"
|
||||
scratch.mkdir()
|
||||
target = scratch / "f.txt"
|
||||
target.write_text("x", encoding="utf-8")
|
||||
assert reg._run_shell_safety_check(
|
||||
{"cmd": ["sed", "-n", "/delete/p", str(target)], "cwd": str(tmp_path / "workspace")},
|
||||
"advanced",
|
||||
) is None
|
||||
|
||||
|
||||
def test_sol_r2_channel_grammar_writes_stay_write_shaped():
|
||||
"""sol-max round-2: option/operand grammar gaps — every one of these is a
|
||||
real write and must report a target (or write shape)."""
|
||||
from ouroboros.tools.shell_guards import writer_target_tokens
|
||||
for cmd in (
|
||||
["sort", "--output", "/o", "/etc/hosts"],
|
||||
["uniq", "-", "/o"],
|
||||
["sed", "-nibak", "s/a/b/", "/f"],
|
||||
["sed", "-n", "1w /o", "/etc/hosts"],
|
||||
["tar", "-cf/o.tar", "/etc/hosts"],
|
||||
["tar", "--extract", "--file=/i.tar", "--directory=/od"],
|
||||
["tar", "xf", "/a.tar"],
|
||||
["gzip", "-S.tgz", "/f"],
|
||||
):
|
||||
assert writer_target_tokens(cmd), cmd
|
||||
# Old-style/list/read spellings stay reads.
|
||||
for cmd in (
|
||||
["tar", "tf", "/a.tar"],
|
||||
["tar", "-tf", "/a.tar"],
|
||||
["gzip", "-l", "/a.gz"],
|
||||
["sed", "s/e/x/", "/f"],
|
||||
["sed", "-n", "/e/p", "/f"],
|
||||
):
|
||||
assert writer_target_tokens(cmd) == [], cmd
|
||||
|
||||
|
||||
def test_sol_r2_compound_and_carve_holes_closed(tmp_path):
|
||||
"""sol-max round-2: a pure-filter HEAD speaks only for its own segment, and
|
||||
uniq's '-' stdin operand cannot hide its output operand from the carve."""
|
||||
from ouroboros.tools.registry import _is_pure_read_inspection
|
||||
from ouroboros.tools.write_shape import non_interpreter_write_shape
|
||||
assert (
|
||||
non_interpreter_write_shape(
|
||||
"sort /etc/hosts && find /d -name x -delete",
|
||||
["sort"], "sort", is_pure_read=_is_pure_read_inspection,
|
||||
)
|
||||
is True
|
||||
)
|
||||
assert _is_pure_read_inspection("printf x | uniq - data/settings.json") is False
|
||||
assert _is_pure_read_inspection("uniq /var/log/a.txt") is True
|
||||
|
||||
|
||||
def test_inline_body_isolation_for_joined_and_wrapped_forms():
|
||||
"""sol-max round-2: a '>' comparison inside a located body is not a redirect
|
||||
even when the body rides a joined flag or an sh -c wrap; real redirects
|
||||
outside the body still classify."""
|
||||
assert interpreter_write_shape(["python3", "-cprint(2 > 1)"]) is False
|
||||
assert (
|
||||
interpreter_write_shape(
|
||||
["sh", "-c", "python3 -c 'print(2 > 1); print(open(\"/etc/hosts\").read())'"]
|
||||
)
|
||||
is False
|
||||
)
|
||||
assert interpreter_write_shape(["sh", "-c", "python3 gen.py > out.txt"]) is True
|
||||
|
||||
|
||||
def test_pure_filter_write_channels_still_blocked(tmp_path):
|
||||
"""The real channels stay writes: sort -o, sed -i, uniq's second operand,
|
||||
tar extract, gzip default all still take guard B."""
|
||||
reg = _registry(tmp_path, mode="external")
|
||||
scratch = tmp_path / "scratch"
|
||||
scratch.mkdir()
|
||||
src = scratch / "in.txt"
|
||||
src.write_text("x\n", encoding="utf-8")
|
||||
for cmd in (
|
||||
["sort", "-o", str(scratch / "out.txt"), str(src)],
|
||||
["sed", "-i", "s/a/b/", str(src)],
|
||||
["uniq", str(src), str(scratch / "out.txt")],
|
||||
["tar", "-xf", str(scratch / "a.tar"), "-C", str(scratch)],
|
||||
["gzip", str(src)],
|
||||
):
|
||||
out = reg._run_shell_safety_check({"cmd": cmd, "cwd": str(tmp_path / "workspace")}, "advanced") or ""
|
||||
assert "WORKSPACE_SHELL_BLOCKED" in out, (cmd, out)
|
||||
|
||||
|
||||
def test_workspace_write_block_message_names_path_and_route(tmp_path):
|
||||
"""The five formerly byte-identical guard-B messages carry the resolved
|
||||
offending path (the light-lane message is the exemplar)."""
|
||||
reg = _registry(tmp_path, mode="external")
|
||||
data = tmp_path / "data"
|
||||
cmd = ["python3", "-c", f"open({str(data / 'x')!r}, 'w').write('hi')"]
|
||||
out = reg._run_shell_safety_check({"cmd": cmd, "cwd": str(tmp_path / "workspace")}, "advanced") or ""
|
||||
assert "Blocked path:" in out
|
||||
assert "read_file" in out
|
||||
|
||||
|
||||
def test_outside_root_write_block_message_names_path_and_root(tmp_path):
|
||||
"""The outside-process-root variant names the blocked path and the selected
|
||||
process root, so the agent can self-correct instead of guessing."""
|
||||
reg = _registry(tmp_path, mode="external")
|
||||
scratch = tmp_path / "scratch"
|
||||
scratch.mkdir()
|
||||
cmd = ["python3", "-c", f"open({str(scratch / 'out.txt')!r}, 'w').write('hi')"]
|
||||
out = reg._run_shell_safety_check({"cmd": cmd, "cwd": str(tmp_path / "workspace")}, "advanced") or ""
|
||||
assert "WORKSPACE_SHELL_BLOCKED" in out
|
||||
assert "outside the selected process root" in out
|
||||
assert "Blocked path:" in out
|
||||
assert "Selected process root:" in out
|
||||
|
|
@ -150,7 +150,9 @@ def test_scope_review_floor_self_lowering_detector():
|
|||
from ouroboros.tools.shell_guards import shell_has_write_indicator
|
||||
|
||||
def verdict(cmd: str) -> bool:
|
||||
"""Judge exactly as the shell guard does: the same `writeish` fact it computes."""
|
||||
"""Judge with the coarse write-shape FACT arm of the guard's composition
|
||||
(interpreter argv takes the mode-aware classifier in the registry; the
|
||||
detector's own contract is identical for both facts)."""
|
||||
return det(cmd.lower(), writeish=shell_has_write_indicator(cmd))
|
||||
|
||||
# Mutation shapes stay blocked — including the ones the write-shape fact alone does
|
||||
|
|
|
|||
|
|
@ -856,6 +856,7 @@
|
|||
* @property {Array<Object>=} harnesses
|
||||
* @property {Object=} profiles
|
||||
* @property {Array<Object>=} quota
|
||||
* @property {Array<Object>=} quota_absences
|
||||
* @property {ClaudexorStatusReads=} reads
|
||||
* @property {boolean=} unified_accounts
|
||||
* @property {SubagentLastDelegation=} subagent_last_delegation
|
||||
|
|
|
|||
|
|
@ -135,7 +135,7 @@ export function humanizeResetAt(resetsAt, nowMs = Date.now()) {
|
|||
|
||||
export function quotaSummary(snapshots, harnessId, subjectId = '',
|
||||
{ quotaRead = READ_OK, nowMs = Date.now(),
|
||||
fallbackSubjectIds = [] } = {}) {
|
||||
fallbackSubjectIds = [], absences = [] } = {}) {
|
||||
// The exhausted window is SHOWN with its reset time, never hidden (Q2-б):
|
||||
// hiding it would make the D28 fallback to API money unexplainable. What
|
||||
// CHANGED is only the wording — the owner asked for the limit text to be
|
||||
|
|
@ -189,7 +189,8 @@ export function quotaSummary(snapshots, harnessId, subjectId = '',
|
|||
for (const snap of rows) {
|
||||
for (const constraint of snap.constraints || []) {
|
||||
const used = Number(constraint.used_ratio);
|
||||
const spent = Boolean(constraint.cooldown_until) || (Number.isFinite(used) && used >= 1.0);
|
||||
const spent = Boolean(constraint.cooldown_until)
|
||||
|| (Number.isFinite(used) && used >= 1.0);
|
||||
const models = Array.isArray(constraint.applies_to_models)
|
||||
? constraint.applies_to_models.filter(Boolean) : [];
|
||||
if (models.length) {
|
||||
|
|
@ -220,14 +221,47 @@ export function quotaSummary(snapshots, harnessId, subjectId = '',
|
|||
else if (worst) {
|
||||
base = `${Math.min(100, Math.round(worst.used * 100))}% used${resets ? ` · resets ${resets}` : ''}`;
|
||||
}
|
||||
// Read, and nothing to report about THIS account: say the usage is
|
||||
// unavailable rather than implying an empty window.
|
||||
if (!base && !note) return { label: 'Usage unavailable', exhausted: false, resetsAt: '', tone: 'muted' };
|
||||
// Match typed gaps by the exact current subject only: the legacy default
|
||||
// alias belongs to retained snapshots, never to another subject's
|
||||
// credential verdict. Claudexor suppresses snapshot-covered absences at
|
||||
// its response boundary; a contradictory future body stays visible here
|
||||
// rather than authorizing a stronger consumer-side interpretation.
|
||||
const absence = (Array.isArray(absences) ? absences : []).find((row) => {
|
||||
const subject = row?.subject;
|
||||
if (!subject || typeof subject !== 'object') return false;
|
||||
if (typeof subject.harness !== 'string') return false;
|
||||
if (subject.subject_id !== null && typeof subject.subject_id !== 'string') return false;
|
||||
return subject.harness === String(harnessId)
|
||||
&& String(subject.subject_id || '') === String(subjectId);
|
||||
}) || null;
|
||||
const reason = typeof absence?.reason === 'string' ? absence.reason : '';
|
||||
const labels = {
|
||||
refresh_failed: 'Usage refresh failed',
|
||||
rate_limited: 'Usage check rate-limited',
|
||||
probe_skipped_rate_limited: 'Usage check paused after a rate limit',
|
||||
poll_paced: 'Usage check paced',
|
||||
not_logged_in: 'Usage unavailable · not signed in',
|
||||
auth_revoked: 'Usage unavailable · sign-in revoked',
|
||||
};
|
||||
const absenceLabel = reason ? (labels[reason] || 'Usage unavailable') : '';
|
||||
// Claudexor's absence detail is already redacted at the producer. Display
|
||||
// it as text only; it never chooses reason, tone, login action, or routing.
|
||||
const absenceDetail = reason && typeof absence?.detail === 'string'
|
||||
? absence.detail.trim() : '';
|
||||
const gap = [absenceLabel, absenceDetail].filter(Boolean).join(' · ');
|
||||
if (!base && !note) {
|
||||
return {
|
||||
label: gap || 'Usage unavailable',
|
||||
exhausted: false,
|
||||
resetsAt: '',
|
||||
tone: 'muted',
|
||||
};
|
||||
}
|
||||
return {
|
||||
exhausted,
|
||||
resetsAt,
|
||||
tone: exhausted ? 'warn' : 'muted',
|
||||
label: [base, note].filter(Boolean).join(' · '),
|
||||
label: [base, note, gap].filter(Boolean).join(' · '),
|
||||
};
|
||||
}
|
||||
|
||||
|
|
@ -598,7 +632,8 @@ export function accountMetaLine(row, payload, { quotaRead = READ_OK, nowMs = Dat
|
|||
parts.push('disabled — excluded from rotation');
|
||||
}
|
||||
parts.push(quotaSummary(payload?.quota || [], row.harness, row.profile_id,
|
||||
{ quotaRead, nowMs, fallbackSubjectIds: quotaSubjectAliases(row, payload) }).label);
|
||||
{ quotaRead, nowMs, fallbackSubjectIds: quotaSubjectAliases(row, payload),
|
||||
absences: payload?.quota_absences || [] }).label);
|
||||
const identity = row.identity || {};
|
||||
if (identity.plan) parts.push(String(identity.plan));
|
||||
// The email is metadata only while it is not already the row's name.
|
||||
|
|
@ -947,7 +982,8 @@ export function accountRowFacts(row, payload,
|
|||
return {
|
||||
badge: verificationBadge(row, { known: accountsRead === READ_OK }),
|
||||
quota: quotaSummary(payload?.quota || [], row.harness, row.profile_id,
|
||||
{ quotaRead, nowMs, fallbackSubjectIds: quotaSubjectAliases(row, payload) }),
|
||||
{ quotaRead, nowMs, fallbackSubjectIds: quotaSubjectAliases(row, payload),
|
||||
absences: payload?.quota_absences || [] }),
|
||||
name: accountName(row),
|
||||
meta: accountMetaLine(row, payload, { quotaRead, nowMs }),
|
||||
};
|
||||
|
|
|
|||
|
|
@ -158,37 +158,144 @@ export async function requestStop(taskId, action) {
|
|||
|
||||
let openMenu = null;
|
||||
let openTrigger = null;
|
||||
let stopMenuWatcher = null;
|
||||
let focusReturnTimer = null;
|
||||
|
||||
const MENU_MARGIN = 8;
|
||||
const MENU_GAP = 4;
|
||||
|
||||
function clamp(value, minimum, maximum) {
|
||||
return Math.min(Math.max(value, minimum), Math.max(minimum, maximum));
|
||||
}
|
||||
|
||||
function positionTaskControlMenu(anchor, menu) {
|
||||
if (!anchor?.isConnected || !menu?.isConnected) {
|
||||
closeTaskControlMenu();
|
||||
return false;
|
||||
}
|
||||
|
||||
const anchorRect = anchor.getBoundingClientRect();
|
||||
if (anchorRect.width <= 0 || anchorRect.height <= 0) {
|
||||
closeTaskControlMenu();
|
||||
return false;
|
||||
}
|
||||
|
||||
const menuRect = menu.getBoundingClientRect();
|
||||
// scrollHeight preserves the intrinsic item height if the viewport-level
|
||||
// fallback max-height has already constrained the rendered border box.
|
||||
const borderHeight = Math.max(0, menuRect.height - menu.clientHeight);
|
||||
const naturalHeight = Math.max(menuRect.height, menu.scrollHeight + borderHeight);
|
||||
const spaceAbove = Math.max(0, anchorRect.top - MENU_GAP - MENU_MARGIN);
|
||||
const spaceBelow = Math.max(
|
||||
0,
|
||||
window.innerHeight - MENU_MARGIN - anchorRect.bottom - MENU_GAP,
|
||||
);
|
||||
const openUpwards = spaceBelow < naturalHeight && spaceAbove > spaceBelow;
|
||||
const availableHeight = openUpwards ? spaceAbove : spaceBelow;
|
||||
const effectiveHeight = Math.min(naturalHeight, availableHeight);
|
||||
const top = openUpwards
|
||||
? Math.max(MENU_MARGIN, anchorRect.top - MENU_GAP - effectiveHeight)
|
||||
: anchorRect.bottom + MENU_GAP;
|
||||
const left = clamp(
|
||||
anchorRect.right - menuRect.width,
|
||||
MENU_MARGIN,
|
||||
window.innerWidth - menuRect.width - MENU_MARGIN,
|
||||
);
|
||||
|
||||
menu.style.setProperty('--tcm-top', `${Math.round(top)}px`);
|
||||
menu.style.setProperty('--tcm-left', `${Math.round(left)}px`);
|
||||
menu.style.setProperty('--tcm-max-height', `${Math.floor(availableHeight)}px`);
|
||||
return true;
|
||||
}
|
||||
|
||||
function watchTaskControlMenu(anchor, menu) {
|
||||
const isCurrentMenu = () => openMenu === menu && openTrigger === anchor;
|
||||
const closeForViewportChange = (event) => {
|
||||
// A height-clamped portal is its own scrollport. Scrolling it does not
|
||||
// move the trigger, so keep it open; ancestor/page scroll still closes.
|
||||
if (event.type === 'scroll' && event.target === menu) return;
|
||||
if (isCurrentMenu()) closeTaskControlMenu();
|
||||
};
|
||||
document.addEventListener('scroll', closeForViewportChange, true);
|
||||
window.addEventListener('resize', closeForViewportChange);
|
||||
|
||||
const observer = new IntersectionObserver((entries) => {
|
||||
if (!isCurrentMenu()) return;
|
||||
const entry = entries.find((candidate) => candidate.target === anchor);
|
||||
if (!anchor.isConnected || (entry && !entry.isIntersecting)) {
|
||||
closeTaskControlMenu();
|
||||
}
|
||||
}, { threshold: 0 });
|
||||
observer.observe(anchor);
|
||||
|
||||
return () => {
|
||||
observer.takeRecords();
|
||||
observer.disconnect();
|
||||
document.removeEventListener('scroll', closeForViewportChange, true);
|
||||
window.removeEventListener('resize', closeForViewportChange);
|
||||
};
|
||||
}
|
||||
|
||||
function clearPendingFocusReturn() {
|
||||
if (focusReturnTimer !== null) window.clearTimeout(focusReturnTimer);
|
||||
focusReturnTimer = null;
|
||||
}
|
||||
|
||||
function finishOutsidePointerFocusReturn(trigger) {
|
||||
clearPendingFocusReturn();
|
||||
focusReturnTimer = window.setTimeout(() => {
|
||||
focusReturnTimer = null;
|
||||
if (!openMenu && trigger?.isConnected) trigger.focus();
|
||||
}, 0);
|
||||
}
|
||||
|
||||
export function closeTaskControlMenu() {
|
||||
if (!openMenu) return;
|
||||
openMenu.remove();
|
||||
clearPendingFocusReturn();
|
||||
const menu = openMenu;
|
||||
const trigger = openTrigger;
|
||||
const restoreFocus = Boolean(menu?.contains(document.activeElement));
|
||||
const stopWatching = stopMenuWatcher;
|
||||
stopMenuWatcher = null;
|
||||
stopWatching?.();
|
||||
menu?.remove();
|
||||
openMenu = null;
|
||||
openTrigger?.setAttribute?.('aria-expanded', 'false');
|
||||
trigger?.setAttribute?.('aria-expanded', 'false');
|
||||
openTrigger = null;
|
||||
document.removeEventListener('pointerdown', onOutsidePointer, true);
|
||||
document.removeEventListener('keydown', onMenuKeydown, true);
|
||||
if (restoreFocus && trigger?.isConnected) trigger.focus();
|
||||
}
|
||||
|
||||
function onOutsidePointer(event) {
|
||||
if (openMenu && !openMenu.contains(event.target)) closeTaskControlMenu();
|
||||
if (!openMenu || openMenu.contains(event.target)) return;
|
||||
const trigger = openTrigger;
|
||||
const restoreFocus = openMenu.contains(document.activeElement);
|
||||
closeTaskControlMenu();
|
||||
// The pointerdown default focuses the clicked element after capture-phase
|
||||
// listeners run. Restore once more in the next task without cancelling the
|
||||
// outside element's click; a successor task menu cancels this stale timer.
|
||||
if (restoreFocus && trigger?.isConnected) finishOutsidePointerFocusReturn(trigger);
|
||||
}
|
||||
|
||||
function onMenuKeydown(event) {
|
||||
// Dismiss = continue the run (Q2: the dismiss affordance replaced the old
|
||||
// explicit "keep running" item).
|
||||
if (event.key === 'Escape') closeTaskControlMenu();
|
||||
if (event.key === 'Escape') {
|
||||
event.preventDefault();
|
||||
closeTaskControlMenu();
|
||||
}
|
||||
}
|
||||
|
||||
/**
|
||||
* Open the three-action dropdown anchored inside `anchor`'s positioned parent.
|
||||
* Open the three-action dropdown in a viewport-level portal anchored to `anchor`.
|
||||
* Dismissing (outside click / Escape) continues the run. Selecting an item
|
||||
* closes the menu and invokes `onAction(actionId)`.
|
||||
* @param {HTMLElement} anchor trigger element; the menu mounts next to it
|
||||
* @param {HTMLElement} anchor trigger element
|
||||
* @param {{cancelPending?: boolean, busy?: boolean, onAction: (action: string) => void}} opts
|
||||
*/
|
||||
export function openTaskControlMenu(anchor, { cancelPending = false, busy = false, onAction } = {}) {
|
||||
closeTaskControlMenu();
|
||||
if (!anchor || !anchor.parentElement) return null;
|
||||
if (!anchor?.isConnected || !document.body) return null;
|
||||
// A11y: the trigger owns a popup menu; expanded tracks the open state.
|
||||
anchor.setAttribute('aria-haspopup', 'menu');
|
||||
anchor.setAttribute('aria-expanded', 'true');
|
||||
|
|
@ -210,9 +317,11 @@ export function openTaskControlMenu(anchor, { cancelPending = false, busy = fals
|
|||
});
|
||||
menu.appendChild(item);
|
||||
}
|
||||
anchor.parentElement.appendChild(menu);
|
||||
document.body.appendChild(menu);
|
||||
openMenu = menu;
|
||||
openTrigger = anchor;
|
||||
stopMenuWatcher = watchTaskControlMenu(anchor, menu);
|
||||
if (!positionTaskControlMenu(anchor, menu)) return null;
|
||||
// A11y: keyboard users land on the first actionable item on open.
|
||||
menu.querySelector('button:not(:disabled)')?.focus?.();
|
||||
document.addEventListener('pointerdown', onOutsidePointer, true);
|
||||
|
|
|
|||
|
|
@ -373,17 +373,14 @@ body.resizing-panels { user-select: none; }
|
|||
}
|
||||
|
||||
/* S3 (Q2/HQ1): the three-action task stop/hurry dropdown (Chat + Activity). */
|
||||
.chat-live-actions,
|
||||
.activity-row-actions {
|
||||
position: relative;
|
||||
}
|
||||
|
||||
.task-control-menu {
|
||||
position: absolute;
|
||||
top: calc(100% + var(--space-1));
|
||||
right: 0;
|
||||
z-index: 90;
|
||||
min-width: 220px;
|
||||
position: fixed;
|
||||
top: var(--tcm-top, 0);
|
||||
left: var(--tcm-left, 0);
|
||||
z-index: 120;
|
||||
width: min(220px, calc(100vw - 16px));
|
||||
max-height: var(--tcm-max-height, calc(100vh - 16px));
|
||||
overflow-y: auto;
|
||||
padding: var(--space-2);
|
||||
border: 1px solid var(--surface-border);
|
||||
border-radius: var(--radius-md);
|
||||
|
|
@ -6425,7 +6422,7 @@ textarea.chat-input {
|
|||
/* Narrow viewports: the actions column would squeeze the account name to a
|
||||
couple of characters, so the row becomes one column and the buttons drop
|
||||
under the metadata they act on. */
|
||||
@media (max-width: 640px) {
|
||||
@media (max-width: 980px) {
|
||||
.harness-account-row {
|
||||
grid-template-columns: minmax(0, 1fr);
|
||||
grid-template-areas: "main" "meta" "actions";
|
||||
|
|
|
|||
|
|
@ -54,6 +54,15 @@ import {
|
|||
submitLoginInput,
|
||||
} from '../modules/harness_login_cards.js';
|
||||
|
||||
test('account actions stack at the app shell compact breakpoint', () => {
|
||||
const css = readFileSync(new URL('../style.css', import.meta.url), 'utf8');
|
||||
assert.ok(css.includes(`@media (max-width: 980px) {
|
||||
.harness-account-row {
|
||||
grid-template-columns: minmax(0, 1fr);
|
||||
grid-template-areas: "main" "meta" "actions";`),
|
||||
'account actions must drop below status before the persistent sidebar squeezes the row');
|
||||
});
|
||||
|
||||
test('managed runtime keeps one contextual Connect intent across install, repair, and update', () => {
|
||||
const payload = (runtime, daemon = {}) => ({ daemon: { state: 'not_provisioned', runtime, ...daemon } });
|
||||
|
||||
|
|
@ -272,6 +281,105 @@ test('a named profile\'s exhausted window is never reported as the default accou
|
|||
assert.equal(namedRow.resetsAt, '2026-08-04T00:00:00Z');
|
||||
});
|
||||
|
||||
test('typed quota gaps are exact-subject, neutral, and distinct in words', () => {
|
||||
const absences = [
|
||||
{ subject: { harness: 'claude', subject_id: 'proton4' }, reason: 'refresh_failed', detail: 'secret-like prose is not parsed' },
|
||||
{ subject: { harness: 'claude', subject_id: 'proton3' }, reason: 'rate_limited' },
|
||||
];
|
||||
const waiting = quotaSummary([], 'claude', 'proton4', { absences });
|
||||
assert.equal(waiting.label, 'Usage refresh failed · secret-like prose is not parsed');
|
||||
assert.equal(waiting.tone, 'muted');
|
||||
assert.equal(quotaSummary([], 'claude', 'proton3', { absences }).label,
|
||||
'Usage check rate-limited');
|
||||
for (const [reason, label] of [
|
||||
['probe_skipped_rate_limited', 'Usage check paused after a rate limit'],
|
||||
['poll_paced', 'Usage check paced'],
|
||||
['not_logged_in', 'Usage unavailable · not signed in'],
|
||||
]) {
|
||||
assert.equal(quotaSummary([], 'claude', 'proton4', { absences: [{
|
||||
subject: { harness: 'claude', subject_id: 'proton4' }, reason,
|
||||
}] }).label, label, reason);
|
||||
}
|
||||
assert.equal(quotaSummary([], 'claude', 'proton2', { absences }).label,
|
||||
'Usage unavailable', 'a sibling absence must not color this subject');
|
||||
const future = quotaSummary([], 'claude', 'proton4', { absences: [
|
||||
{ subject: { harness: 'claude', subject_id: 'proton4' }, reason: 'future_reason', detail: 'auth_revoked' },
|
||||
] });
|
||||
assert.equal(future.tone, 'muted', 'unknown reason and detail prose stay neutral');
|
||||
assert.equal(future.label, 'Usage unavailable · auth_revoked',
|
||||
'detail is displayed as text but cannot select the warning tone');
|
||||
assert.equal(quotaSummary([], 'claude', 'proton4', { absences: [
|
||||
{ subject: { harness: 'claude', subject_id: 'proton4' }, reason: 'auth_revoked' },
|
||||
] }).tone, 'muted');
|
||||
|
||||
assert.deepEqual(quotaSummary([], 'claude', 'claude-default', {
|
||||
fallbackSubjectIds: [''],
|
||||
absences: [{
|
||||
subject: { harness: 'claude', subject_id: '' }, reason: 'auth_revoked',
|
||||
}],
|
||||
}), { label: 'Usage unavailable', exhausted: false, resetsAt: '', tone: 'muted' },
|
||||
'a legacy snapshot alias must never borrow another subject\'s auth absence');
|
||||
|
||||
const exactGapOverAlias = quotaSummary([{
|
||||
subject: { harness: 'claude', subject_id: '' }, freshness: 'fresh',
|
||||
constraints: [{ used_ratio: 0.5 }],
|
||||
}], 'claude', 'claude-default', {
|
||||
fallbackSubjectIds: [''],
|
||||
absences: [{
|
||||
subject: { harness: 'claude', subject_id: 'claude-default' },
|
||||
reason: 'auth_revoked',
|
||||
}],
|
||||
});
|
||||
assert.equal(exactGapOverAlias.label, '50% used · Usage unavailable · sign-in revoked');
|
||||
assert.equal(exactGapOverAlias.tone, 'muted',
|
||||
'the exact auth verdict remains explicit in words without a second warning state');
|
||||
|
||||
assert.equal(quotaSummary([], 'claude', 'proton4', { absences: [{
|
||||
subject: { harness: 'claude', subject_id: 'proton4' },
|
||||
reason: 'refresh_failed', detail: { nested: 'bad' },
|
||||
}] }).label, 'Usage refresh failed', 'malformed detail is ignored');
|
||||
assert.equal(quotaSummary([], 'claude', 'proton4', { absences: [{
|
||||
subject: { harness: 'claude', subject_id: 'proton4' },
|
||||
reason: { nested: 'bad' }, detail: 'must not survive a malformed reason',
|
||||
}] }).label, 'Usage unavailable', 'malformed reason and its detail are ignored');
|
||||
});
|
||||
|
||||
test('a contradictory same-subject absence stays visible and fail-open', () => {
|
||||
const summary = quotaSummary([{
|
||||
subject: { harness: 'claude', subject_id: 'proton4' }, freshness: 'fresh',
|
||||
constraints: [{ used_ratio: 0.25 }],
|
||||
}], 'claude', 'proton4', { absences: [{
|
||||
subject: { harness: 'claude', subject_id: 'proton4' }, reason: 'auth_revoked',
|
||||
}] });
|
||||
assert.equal(summary.label, '25% used · Usage unavailable · sign-in revoked');
|
||||
assert.equal(summary.tone, 'muted');
|
||||
|
||||
const emptyButFresh = quotaSummary([{
|
||||
subject: { harness: 'claude', subject_id: 'proton4' }, freshness: 'fresh',
|
||||
constraints: [],
|
||||
}], 'claude', 'proton4', { absences: [{
|
||||
subject: { harness: 'claude', subject_id: 'proton4' }, reason: 'auth_revoked',
|
||||
}] });
|
||||
assert.deepEqual(emptyButFresh,
|
||||
{ label: 'Usage unavailable · sign-in revoked', exhausted: false, resetsAt: '', tone: 'muted' });
|
||||
});
|
||||
|
||||
test('stale quota does not become a current percentage beside its typed gap', () => {
|
||||
const summary = quotaSummary([{
|
||||
subject: { harness: 'claude', subject_id: 'proton4' }, freshness: 'stale',
|
||||
constraints: [{ used_ratio: 0.65, resets_at: '2026-08-31T00:00:00Z' }],
|
||||
}], 'claude', 'proton4', { absences: [{
|
||||
subject: { harness: 'claude', subject_id: 'proton4' }, reason: 'poll_paced',
|
||||
detail: 'next poll is paced',
|
||||
}] });
|
||||
assert.deepEqual(summary, {
|
||||
label: 'Usage check paced · next poll is paced',
|
||||
exhausted: false,
|
||||
resetsAt: '',
|
||||
tone: 'muted',
|
||||
});
|
||||
});
|
||||
|
||||
test('a model-scoped window never paints the whole account exhausted — it is a compact note', () => {
|
||||
// The daemon schema's own words (@claudexor/schema quota.ts): a non-null
|
||||
// applies_to_models is a per-model cap, and "a model-specific cap never
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue