mirror of
https://github.com/razzant/ouroboros.git
synced 2026-10-03 04:07:04 +00:00
Say only what the code knows: neutral cause sentences, exact-name dedupe
Roast round 1: four cause sentences claimed more than the record proves — "too few reviewers could be read" is false when the reviewers were read and the cycle cap refused the re-authored answer; "failed before any reviewer answered", "after it was reviewed" and "the review was skipped" likewise. They now state the cause neutrally. The X › X dedupe compares the two names exactly (a prefix rule folded "Art" into "Arthur"). The JS test that parsed the Python table with a regex was a second SSOT parser and is gone; the shared parity fixture is the pin. The long-work sentence says "one message saying what I will check and why"; the promote description names the queue slot, admission and reviews an independent task gets; send_user_message says "card". Outcome honesty is shorter than the original (prompts/SYSTEM.md 24391 -> 24289 bytes). The configuration chapter no longer says direct turns are not named. Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com>
This commit is contained in:
parent
148551ff2a
commit
ac17b70cca
10 changed files with 41 additions and 56 deletions
|
|
@ -85,7 +85,7 @@ A registry of `config.SETTINGS_DEFAULTS` (exact defaults stay canonical in `conf
|
|||
| OUROBOROS_MODEL_MAX_CONCURRENCY | 3 | Per-(model,route) concurrent provider-call cap (`model_concurrency.py`) |
|
||||
| OUROBOROS_MODEL_SLOT_MAX_WAIT_SEC | 180 | Concurrency-slot wait bound |
|
||||
| OUROBOROS_PROJECT_NAMING_TIMEOUT_SEC | 60 | Project-naming call ceiling |
|
||||
| OUROBOROS_PROJECT_NAMING_ASYNC_TIMEOUT_SEC | 8 | Bound of the inline naming call when a card is turned into a project (`gateway/projects.py`); direct turns are not named in the background |
|
||||
| OUROBOROS_PROJECT_NAMING_ASYNC_TIMEOUT_SEC | 8 | Bound of the inline naming call when a card is turned into a project (`gateway/projects.py`); a direct Main turn is named in the background once it starts working (`spawn_turn_namer`, bounded by `OUROBOROS_PROJECT_NAMING_TIMEOUT_SEC` + 30 s) |
|
||||
| OUROBOROS_UPDATE_LETTER_TIMEOUT_SEC | 120 | Update-letter LIGHT one-shot ceiling, slot wait and provider call together (`update_letter.py`) |
|
||||
| OUROBOROS_FALLBACK_COOLDOWN_ENABLED | true | 429-aware per-process model cooldown |
|
||||
| OUROBOROS_FALLBACK_COOLDOWN_SEC | 120 | Cooldown window |
|
||||
|
|
|
|||
|
|
@ -543,8 +543,8 @@ TASK_CAUSE_PHRASES = {
|
|||
# Acceptance-decision reasons. An accepted decision renders no clause at
|
||||
# all, so clean_pass and clean_pass_obligations_closed carry no sentence.
|
||||
"author_finish": "The answer was delivered on Main's own judgement; the reviewers had not signed it off.",
|
||||
"review_degraded": "Not enough reviewer verdicts could be read to settle the answer.",
|
||||
"infra_failure": "The review could not run: it failed before any reviewer answered.",
|
||||
"review_degraded": "No reviewer verdict was established for this answer.",
|
||||
"infra_failure": "A review infrastructure failure prevented a settled verdict.",
|
||||
"dialogue_terminal": "The reviewers and Main could not agree, and both positions were kept.",
|
||||
"improvement_capsule": "The reviewers asked for one more pass and Main was given their notes.",
|
||||
"fence_reopen_failed": "The requested extra pass could not be started, so the answer stands as it was.",
|
||||
|
|
@ -556,11 +556,11 @@ TASK_CAUSE_PHRASES = {
|
|||
"no_actionable_changes": "The re-review was not clean and suggested nothing to change.",
|
||||
"identical_acceptance_refused": "Nothing had changed since the last review, so the recorded verdict stands.",
|
||||
"review_skipped_deadline_reserve": "There was not enough time left to review the answer.",
|
||||
"delivery_binding_superseded": "The answer changed after it was reviewed, so the review no longer covered it.",
|
||||
"delivery_binding_superseded": "The answer or its evidence changed, so the earlier review no longer covered it.",
|
||||
"owner_followup": "A new message from you arrived, so the review was set aside for it.",
|
||||
"evidence_refresh": "The work changed after the review was frozen, so it no longer covered the answer.",
|
||||
"revision_unavailable_on_forced_rail": "The task had to stop, so the requested rework never happened.",
|
||||
"owner_hurry": "You asked me to hurry, so the review was skipped.",
|
||||
"owner_hurry": "You asked me to hurry, so no further review was started.",
|
||||
"unspecified": "The answer was not signed off, and no cause was recorded.",
|
||||
# The rail that ended the task before an owed acceptance panel could run.
|
||||
"acceptance_bypassed_budget_exhausted": "The task ran out of budget before the answer could be reviewed.",
|
||||
|
|
|
|||
|
|
@ -831,11 +831,11 @@ def task_presentation_snapshot(drive_root: Any, task_id: str, *, task: Any = Non
|
|||
break
|
||||
task_name = task_name or "Task"
|
||||
label = f"{pname} › {task_name}" if pname else task_name
|
||||
# One name, said once: a task whose own name IS the project name (or its
|
||||
# opening words) renders "Launch › Launch" / "Launch › Launch release",
|
||||
# which reads as two different things. Keep the longer half alone.
|
||||
if pname and (task_name.startswith(pname) or pname.startswith(task_name)):
|
||||
label = task_name if len(task_name) >= len(pname) else pname
|
||||
# One name, said once: a task whose own name IS the project name renders
|
||||
# "Launch › Launch", which reads as two different things. Exact equality
|
||||
# only — a prefix rule would fold "Art" into "Arthur".
|
||||
if pname and task_name == pname:
|
||||
label = task_name
|
||||
return {"project_id": pid, "project_name": pname, "task_id": tid,
|
||||
"project_routable": registered, "task_name": task_name,
|
||||
"target_label": label}
|
||||
|
|
|
|||
|
|
@ -102,10 +102,10 @@ log = logging.getLogger(__name__)
|
|||
# 300-line function gate; v6.70.0 added the ground-truth-probe contract).
|
||||
_PROMOTE_CHAT_DESCRIPTION = (
|
||||
"Promote real work out of this conversation into a supervised pooled task "
|
||||
"while the conversation remains available. This conversation keeps its own "
|
||||
"tools, files and multi-step work; promote when the work is better as an "
|
||||
"independent task — its own card, queue slot, admission and reviews, steerable "
|
||||
"from chat — or when the owner asked for a task or a project. "
|
||||
"while the conversation remains available. Tools, files and several steps can "
|
||||
"stay in the conversation; promote when independent work is useful — its own "
|
||||
"queue slot, admission and reviews, steerable from chat — or when the owner "
|
||||
"explicitly asks for a separate task. "
|
||||
"Before framing the objective around an EXISTING artifact "
|
||||
"('check/fix/extend the X skill/file'), ground-truth its existence with one cheap probe "
|
||||
"first (skills: list_skills; files: list_files) — memory of past work is not evidence "
|
||||
|
|
@ -341,7 +341,7 @@ def get_tools() -> List[ToolEntry]:
|
|||
"line of longer work (what I am about to do and why), or a mid-work "
|
||||
"insight, a question, or an invitation to collaborate. It appears in "
|
||||
"the conversation as a normal reply and leaves the work running; later "
|
||||
"progress stays in the activity block and the final answer is delivered "
|
||||
"progress stays in the card and the final answer is delivered "
|
||||
"automatically.",
|
||||
"parameters": {"type": "object", "properties": {
|
||||
"text": {"type": "string", "description": "Message text"},
|
||||
|
|
|
|||
|
|
@ -225,11 +225,11 @@ need it.
|
|||
broad fallbacks, silent catches, or shims lacking a concrete reachable
|
||||
failure mode. Mid-task I ask: am I solving the class or patching symptoms, am
|
||||
I adding surface area, am I still within my human's stated scope?
|
||||
- For long work, my first line to my human is what I am about to do and why,
|
||||
while the work continues; later progress is concise — what I learned and
|
||||
the next step — explaining the thought, not narrating tool calls. After a
|
||||
repeatable workflow I capture the recipe: trigger, authoritative files and
|
||||
logs, commands, validation, known false leads.
|
||||
- Before long work I send my human one message saying what I will check and
|
||||
why; progress after that is concise — what I learned and the next step —
|
||||
explaining the thought, not narrating tool calls. After a repeatable
|
||||
workflow I capture the recipe: trigger, authoritative files and logs,
|
||||
commands, validation, known false leads.
|
||||
- `task_acceptance_review` can nominate my complete task result for review.
|
||||
After the whole tool-result block, the host advances the same operation as
|
||||
final delivery; checking an intermediate artifact is not whole-task acceptance.
|
||||
|
|
@ -239,16 +239,14 @@ need it.
|
|||
|
||||
### Outcome honesty
|
||||
|
||||
Every task ends in one of three honest states, and I say which one plainly.
|
||||
Either I solved it and verified that against the task's own surface; or I got
|
||||
part of the way and hand over the real partial result with the unverified and
|
||||
incomplete parts marked; or something blocked me, and I say what it was, show
|
||||
the exact evidence, and name the next action someone could take. When a
|
||||
deadline, a budget or a round limit forces me to finish, I extract the best
|
||||
verified result I have and mark the gaps. Handing over an honest partial result
|
||||
is an expected ending, not a failure; returning nothing is the only real failure
|
||||
mode. I never claim more than I verified: calling something solved without
|
||||
checking it is worse than admitting it is partial.
|
||||
Every task ends in one of three honest states, and I say which plainly:
|
||||
solved and verified against the task's own surface; partly done, with the
|
||||
real partial result handed over and its unverified or missing parts marked;
|
||||
or blocked, with what blocked me, the exact evidence and the next action
|
||||
someone could take. When a deadline, budget or round limit forces me to
|
||||
finish, I extract the best verified result I have and mark the gaps. An
|
||||
honest partial result is an expected ending; returning nothing is the only
|
||||
real failure mode. I never claim more than I verified.
|
||||
|
||||
## Capability Acquisition
|
||||
|
||||
|
|
|
|||
|
|
@ -504,7 +504,7 @@ def test_host_verdict_leads_both_lifecycle_rows(tmp_path, monkeypatch):
|
|||
) is True
|
||||
assert queued[0]["text"] == (
|
||||
"Launch 🚀 › Ship release · Done with warnings\n"
|
||||
"Not enough reviewer verdicts could be read to settle the answer. "
|
||||
"No reviewer verdict was established for this answer. "
|
||||
"Open the Project for details."
|
||||
)
|
||||
assert "final_message" not in queued[0]["text"]
|
||||
|
|
@ -533,7 +533,7 @@ def test_host_verdict_leads_both_lifecycle_rows(tmp_path, monkeypatch):
|
|||
projection = next(row for row in rows if row.get("summary_kind") == "terminal_root_projection")
|
||||
assert projection["text"] == (
|
||||
"Done with warnings. Root task root-project. "
|
||||
"Not enough reviewer verdicts could be read to settle the answer."
|
||||
"No reviewer verdict was established for this answer."
|
||||
)
|
||||
assert "final_message" not in projection["text"]
|
||||
# The room is the project and result_ref is the reader, so neither the id
|
||||
|
|
|
|||
|
|
@ -409,8 +409,8 @@ export function taskStoppedWithSummary(evt) {
|
|||
// web/tests/fixtures/outcome_phase_parity.json pins both.
|
||||
const TASK_CAUSE_PHRASES = {
|
||||
author_finish: "The answer was delivered on Main's own judgement; the reviewers had not signed it off.",
|
||||
review_degraded: "Not enough reviewer verdicts could be read to settle the answer.",
|
||||
infra_failure: "The review could not run: it failed before any reviewer answered.",
|
||||
review_degraded: "No reviewer verdict was established for this answer.",
|
||||
infra_failure: "A review infrastructure failure prevented a settled verdict.",
|
||||
dialogue_terminal: "The reviewers and Main could not agree, and both positions were kept.",
|
||||
improvement_capsule: "The reviewers asked for one more pass and Main was given their notes.",
|
||||
fence_reopen_failed: "The requested extra pass could not be started, so the answer stands as it was.",
|
||||
|
|
@ -422,11 +422,11 @@ const TASK_CAUSE_PHRASES = {
|
|||
no_actionable_changes: "The re-review was not clean and suggested nothing to change.",
|
||||
identical_acceptance_refused: "Nothing had changed since the last review, so the recorded verdict stands.",
|
||||
review_skipped_deadline_reserve: "There was not enough time left to review the answer.",
|
||||
delivery_binding_superseded: "The answer changed after it was reviewed, so the review no longer covered it.",
|
||||
delivery_binding_superseded: "The answer or its evidence changed, so the earlier review no longer covered it.",
|
||||
owner_followup: "A new message from you arrived, so the review was set aside for it.",
|
||||
evidence_refresh: "The work changed after the review was frozen, so it no longer covered the answer.",
|
||||
revision_unavailable_on_forced_rail: "The task had to stop, so the requested rework never happened.",
|
||||
owner_hurry: "You asked me to hurry, so the review was skipped.",
|
||||
owner_hurry: "You asked me to hurry, so no further review was started.",
|
||||
unspecified: "The answer was not signed off, and no cause was recorded.",
|
||||
acceptance_bypassed_budget_exhausted: "The task ran out of budget before the answer could be reviewed.",
|
||||
acceptance_bypassed_round_limit: "The task hit its round limit before the answer could be reviewed.",
|
||||
|
|
|
|||
10
web/tests/fixtures/outcome_phase_parity.json
vendored
10
web/tests/fixtures/outcome_phase_parity.json
vendored
|
|
@ -111,7 +111,7 @@
|
|||
},
|
||||
"phase": "warn",
|
||||
"headline": "Done with warnings",
|
||||
"acceptance_clause": "Not enough reviewer verdicts could be read to settle the answer."
|
||||
"acceptance_clause": "No reviewer verdict was established for this answer."
|
||||
},
|
||||
{
|
||||
"name": "an accepted decision leaves the execution reason standing",
|
||||
|
|
@ -211,14 +211,14 @@
|
|||
"record": {"status": "completed", "outcome_axes": {"execution": {"status": "ok"}, "review": {"status": "degraded", "acceptance_decision": {"status": "finalized_unaccepted", "reason": "review_degraded"}}}},
|
||||
"phase": "warn",
|
||||
"headline": "Done with warnings",
|
||||
"acceptance_clause": "Not enough reviewer verdicts could be read to settle the answer."
|
||||
"acceptance_clause": "No reviewer verdict was established for this answer."
|
||||
},
|
||||
{
|
||||
"name": "acceptance reason infra_failure says its own sentence",
|
||||
"record": {"status": "completed", "outcome_axes": {"execution": {"status": "ok"}, "review": {"status": "degraded", "acceptance_decision": {"status": "finalized_unaccepted", "reason": "infra_failure"}}}},
|
||||
"phase": "warn",
|
||||
"headline": "Done with warnings",
|
||||
"acceptance_clause": "The review could not run: it failed before any reviewer answered."
|
||||
"acceptance_clause": "A review infrastructure failure prevented a settled verdict."
|
||||
},
|
||||
{
|
||||
"name": "acceptance reason dialogue_terminal says its own sentence",
|
||||
|
|
@ -302,7 +302,7 @@
|
|||
"record": {"status": "completed", "outcome_axes": {"execution": {"status": "ok"}, "review": {"status": "degraded", "acceptance_decision": {"status": "finalized_unaccepted", "reason": "delivery_binding_superseded"}}}},
|
||||
"phase": "warn",
|
||||
"headline": "Done with warnings",
|
||||
"acceptance_clause": "The answer changed after it was reviewed, so the review no longer covered it."
|
||||
"acceptance_clause": "The answer or its evidence changed, so the earlier review no longer covered it."
|
||||
},
|
||||
{
|
||||
"name": "acceptance reason owner_followup says its own sentence",
|
||||
|
|
@ -330,7 +330,7 @@
|
|||
"record": {"status": "completed", "outcome_axes": {"execution": {"status": "ok"}, "review": {"status": "degraded", "acceptance_decision": {"status": "finalized_unaccepted", "reason": "owner_hurry"}}}},
|
||||
"phase": "warn",
|
||||
"headline": "Done with warnings",
|
||||
"acceptance_clause": "You asked me to hurry, so the review was skipped."
|
||||
"acceptance_clause": "You asked me to hurry, so no further review was started."
|
||||
},
|
||||
{
|
||||
"name": "acceptance reason unspecified says its own sentence",
|
||||
|
|
|
|||
|
|
@ -93,19 +93,6 @@ test('every acceptance reason the host can record has a sentence', () => {
|
|||
}
|
||||
});
|
||||
|
||||
test('the host and the card carry the same cause table', () => {
|
||||
// ONE table, two languages. The fixture pins the sentences a record can
|
||||
// reach; this pins the table itself, so an entry added on one side only
|
||||
// fails here instead of silently drifting.
|
||||
const py = readFileSync(new URL('../../ouroboros/project_dialogue.py', import.meta.url), 'utf8');
|
||||
const block = py.split('TASK_CAUSE_PHRASES = {')[1].split('\n}')[0];
|
||||
const entries = [...block.matchAll(/^ {4}"([a-z_]+)": "((?:[^"\\]|\\.)*)",$/gm)];
|
||||
assert.ok(entries.length >= 30, `expected the host table, saw ${entries.length}`);
|
||||
for (const [, code, sentence] of entries) {
|
||||
assert.equal(taskReasonPhrase(code), sentence, code);
|
||||
}
|
||||
});
|
||||
|
||||
test('one status-word family: the card phase matches the host over the shared fixture', () => {
|
||||
// The same fixture is read by tests/test_project_plain_rows.py, so a
|
||||
// divergence between this severity fold and the host's durable label word
|
||||
|
|
@ -149,7 +136,7 @@ const A4 = {
|
|||
test('an unaccepted decision explains the warning in its own words', () => {
|
||||
assert.equal(
|
||||
taskReasonDetail(A4),
|
||||
'Not enough reviewer verdicts could be read to settle the answer.',
|
||||
'No reviewer verdict was established for this answer.',
|
||||
);
|
||||
assert.doesNotMatch(taskReasonDetail(A4), /final_message/);
|
||||
// The stored reviewer rationale belongs to the card body, the task result
|
||||
|
|
|
|||
|
|
@ -418,7 +418,7 @@ test('a review-caused warning names the acceptance decision on the card and in L
|
|||
const live = summarizeChatLiveEvent(evt);
|
||||
const replay = summarizeLogEvent(evt);
|
||||
assert.deepEqual({ phase: live.phase, headline: live.headline }, { phase: 'warn', headline: 'Done with warnings' });
|
||||
assert.match(live.body, /Not enough reviewer verdicts could be read to settle the answer\./);
|
||||
assert.match(live.body, /No reviewer verdict was established for this answer\./);
|
||||
assert.doesNotMatch(live.body, /final_message/);
|
||||
// The raw code lives on in the record half, never in the card body.
|
||||
assert.doesNotMatch(live.body, /finalized_unaccepted/);
|
||||
|
|
|
|||
Loading…
Add table
Add a link
Reference in a new issue