mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-03 12:18:39 +00:00
Complete Nano dispatch context propagation
Co-authored-by: Claudexor <noreply@claudexor.dev>
This commit is contained in:
parent
f46c4985c3
commit
42151c8a2b
6 changed files with 52 additions and 10 deletions
|
|
@ -199,6 +199,7 @@ class LLMClient(
|
|||
caller_execution_deadline: Optional[float] = None,
|
||||
wait_for_resources: bool = True,
|
||||
processing_preference: str | None = None,
|
||||
context_mode: str | None = None,
|
||||
) -> Tuple[Dict[str, Any], Dict[str, Any]]:
|
||||
"""Single LLM call returning (message, usage); no_proxy avoids macOS fork proxy crashes.
|
||||
|
||||
|
|
@ -225,6 +226,8 @@ class LLMClient(
|
|||
local_kwargs = {"timeout": timeout}
|
||||
if processing_preference:
|
||||
local_kwargs["processing_preference"] = processing_preference
|
||||
if context_mode:
|
||||
local_kwargs["context_mode"] = context_mode
|
||||
message, usage = self._chat_local(
|
||||
messages, tools, max_tokens, tool_choice, **local_kwargs,
|
||||
)
|
||||
|
|
@ -233,7 +236,8 @@ class LLMClient(
|
|||
# system proxy lookup without every caller remembering a flag.
|
||||
no_proxy = no_proxy or in_worker_process()
|
||||
target = {**self._resolve_remote_target(model),
|
||||
"processing_preference": processing_preference}
|
||||
"processing_preference": processing_preference,
|
||||
"context_mode": context_mode}
|
||||
if temperature is None and target.get("provider") != "claudexor":
|
||||
temperature = default_temperature
|
||||
message, usage = self._chat_remote(
|
||||
|
|
@ -282,6 +286,7 @@ class LLMClient(
|
|||
caller_execution_deadline: Optional[float] = None,
|
||||
wait_for_resources: bool = True,
|
||||
processing_preference: str | None = None,
|
||||
context_mode: str | None = None,
|
||||
) -> Tuple[Dict[str, Any], Dict[str, Any]]:
|
||||
"""Async remote chat; no_proxy keeps forked macOS workers off OS proxy APIs.
|
||||
|
||||
|
|
@ -300,6 +305,8 @@ class LLMClient(
|
|||
local_kwargs = {"timeout": timeout}
|
||||
if processing_preference:
|
||||
local_kwargs["processing_preference"] = processing_preference
|
||||
if context_mode:
|
||||
local_kwargs["context_mode"] = context_mode
|
||||
result = self._chat_local(messages, tools, max_tokens, tool_choice, **local_kwargs)
|
||||
return result, last_physical_attempt_capture()
|
||||
|
||||
|
|
@ -313,7 +320,8 @@ class LLMClient(
|
|||
result[1]["ledger_attempt_ids"] = list(attempt_ids)
|
||||
return result
|
||||
target = {**self._resolve_remote_target(model),
|
||||
"processing_preference": processing_preference}
|
||||
"processing_preference": processing_preference,
|
||||
"context_mode": context_mode}
|
||||
if temperature is None and target.get("provider") != "claudexor":
|
||||
temperature = default_temperature
|
||||
carried_turn_state = turn_state_for_route(model_turn_state, target.get("provider"))
|
||||
|
|
|
|||
|
|
@ -198,6 +198,7 @@ class _LocalLaneMixin:
|
|||
self, messages: List[Dict[str, Any]], tools: Optional[List[Dict[str, Any]]],
|
||||
max_tokens: int, tool_choice: str, timeout: Optional[float] = None,
|
||||
processing_preference: Optional[str] = None,
|
||||
context_mode: Optional[str] = None,
|
||||
) -> Tuple[Dict[str, Any], Dict[str, Any]]:
|
||||
"""Prepare the complete local payload for sizing and actual dispatch."""
|
||||
messages = self._normalize_system_message_placement(messages)
|
||||
|
|
@ -241,7 +242,8 @@ class _LocalLaneMixin:
|
|||
preference = resolve_processing_preference(override=processing_preference)
|
||||
target = {"provider": "local", "resolved_model": "local-model", "usage_model": "local-model",
|
||||
"processing_preference": preference, "context_window_tokens": evidence.get("context_window"),
|
||||
"context_window_confirmed": evidence.get("confirmed") is True}
|
||||
"context_window_confirmed": evidence.get("confirmed") is True,
|
||||
"context_mode": context_mode}
|
||||
return target, kwargs
|
||||
|
||||
def _finalize_local_candidate(self, target: Dict[str, Any], payload: Dict[str, Any]) -> Dict[str, Any]:
|
||||
|
|
@ -270,11 +272,12 @@ class _LocalLaneMixin:
|
|||
self, messages: List[Dict[str, Any]], tools: Optional[List[Dict[str, Any]]],
|
||||
max_tokens: int, tool_choice: str, timeout: Optional[float] = None,
|
||||
processing_preference: Optional[str] = None,
|
||||
context_mode: Optional[str] = None,
|
||||
) -> Tuple[Dict[str, Any], Dict[str, Any]]:
|
||||
"""Send exactly the previously prepared complete local candidate."""
|
||||
client = self._get_local_client()
|
||||
local_target, candidate = self._build_local_candidate(
|
||||
messages, tools, max_tokens, tool_choice, timeout, processing_preference)
|
||||
messages, tools, max_tokens, tool_choice, timeout, processing_preference, context_mode)
|
||||
candidate = self._finalize_local_candidate(local_target, candidate)
|
||||
clean_tools = candidate.get("tools")
|
||||
preference = local_target["processing_preference"]
|
||||
|
|
|
|||
|
|
@ -119,7 +119,7 @@ from ouroboros.nanny_pacing import (
|
|||
)
|
||||
|
||||
|
||||
def _setup_dynamic_tools(tools_registry, tool_schemas, messages):
|
||||
def _setup_dynamic_tools(tools_registry, tool_schemas, messages, context_mode="max"):
|
||||
"""Attach list/enable tool handlers and mutate the active schema list."""
|
||||
enabled_extra: set = set()
|
||||
active_tool_names = {
|
||||
|
|
@ -134,7 +134,7 @@ def _setup_dynamic_tools(tools_registry, tool_schemas, messages):
|
|||
else []
|
||||
)
|
||||
non_core = [
|
||||
t for t in list_non_core_tools(tools_registry)
|
||||
t for t in list_non_core_tools(tools_registry, context_mode=context_mode)
|
||||
if t["name"] not in active_tool_names
|
||||
]
|
||||
if not non_core:
|
||||
|
|
@ -193,7 +193,7 @@ def _setup_dynamic_tools(tools_registry, tool_schemas, messages):
|
|||
tools_registry.override_handler("list_available_tools", _handle_list_tools)
|
||||
tools_registry.override_handler("enable_tools", _handle_enable_tools)
|
||||
|
||||
non_core_count = len(list_non_core_tools(tools_registry))
|
||||
non_core_count = len(list_non_core_tools(tools_registry, context_mode=context_mode))
|
||||
if non_core_count > 0:
|
||||
_append_or_merge_user_message(
|
||||
messages,
|
||||
|
|
@ -401,8 +401,10 @@ def run_llm_loop(
|
|||
from ouroboros.tools import tool_discovery as _td
|
||||
_td.set_registry(tools)
|
||||
|
||||
tool_schemas = saved["tool_schemas"] if saved else initial_tool_schemas(tools)
|
||||
tool_schemas, _enabled_extra_tools = _setup_dynamic_tools(tools, tool_schemas, messages)
|
||||
tool_schemas = saved["tool_schemas"] if saved else initial_tool_schemas(tools, context_mode=active_context_mode)
|
||||
tool_schemas, _enabled_extra_tools = _setup_dynamic_tools(
|
||||
tools, tool_schemas, messages, context_mode=active_context_mode
|
||||
)
|
||||
ctx.event_queue, ctx.task_id, ctx.messages = event_queue, task_id, messages
|
||||
stateful_executor = StatefulToolExecutor()
|
||||
exit_ctx = _LoopExitContext(
|
||||
|
|
|
|||
|
|
@ -1377,6 +1377,7 @@ def call_llm_with_retry(
|
|||
"model_role": model_role, "model_turn_state": model_turn_state,
|
||||
"model_account_override": model_account_override,
|
||||
"processing_preference": processing_preference,
|
||||
"context_mode": getattr(physical_context, "rendered_mode", None),
|
||||
"reasoning_effort": effort,
|
||||
"max_tokens": MAIN_LOOP_MAX_TOKENS,
|
||||
"stream": True, "caller_deadline_ts": (None if deadline_ts is None
|
||||
|
|
|
|||
|
|
@ -20,3 +20,19 @@ def test_usage_ledger_accepts_nano_physical_context():
|
|||
}
|
||||
}
|
||||
_validate_candidate_facts(row, 1)
|
||||
|
||||
|
||||
def test_chat_carries_nano_mode_to_physical_target(monkeypatch):
|
||||
from ouroboros.llm import LLMClient
|
||||
|
||||
client = LLMClient()
|
||||
captured = {}
|
||||
monkeypatch.setattr(client, "_resolve_remote_target", lambda _model: {"provider": "openai"})
|
||||
|
||||
def remote(target, *_args, **_kwargs):
|
||||
captured.update(target)
|
||||
return {"content": "ok"}, {}
|
||||
|
||||
monkeypatch.setattr(client, "_chat_remote", remote)
|
||||
client.chat([{"role": "user", "content": "hello"}], "openai::test", context_mode="nano")
|
||||
assert captured["context_mode"] == "nano"
|
||||
|
|
|
|||
|
|
@ -41,7 +41,7 @@ def test_non_core_listing_excludes_core_media_tools():
|
|||
|
||||
def test_loop_bootstraps_from_tool_policy():
|
||||
source = inspect.getsource(loop_mod)
|
||||
assert "initial_tool_schemas(tools)" in source
|
||||
assert "initial_tool_schemas(tools, context_mode=active_context_mode)" in source
|
||||
assert "schemas(core_only=True)" not in source
|
||||
|
||||
|
||||
|
|
@ -103,6 +103,18 @@ def test_enable_tools_does_not_duplicate_active_tool_schemas():
|
|||
assert "already active" in extra_again_result
|
||||
|
||||
|
||||
def test_nano_initial_view_uses_compact_schema_selection():
|
||||
registry = _build_registry()
|
||||
max_names = {schema["function"]["name"] for schema in initial_tool_schemas(registry)}
|
||||
nano_names = {
|
||||
schema["function"]["name"]
|
||||
for schema in initial_tool_schemas(registry, context_mode="nano")
|
||||
}
|
||||
assert nano_names
|
||||
assert nano_names <= max_names
|
||||
assert len(nano_names) < len(max_names)
|
||||
|
||||
|
||||
def test_list_available_tools_hides_enabled_extra_tools():
|
||||
registry = _build_registry()
|
||||
tool_schemas = initial_tool_schemas(registry)
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue