fix(review): keep checkpoint lifecycle truth ordered

Preserve lifecycle-only review attempts as neutral until a semantic verdict is proven, keep delayed history from overriding newer attempts, and restore explicit collapsed-checkpoint UI coverage.

Co-authored-by: Ouroboros <311266734+ouroboros-agent@users.noreply.github.com>
This commit is contained in:
Ouroboros 2026-08-26 22:11:29 +03:00
parent 7665357293
commit b9ef3a5e69
7 changed files with 360 additions and 171 deletions

View file

@ -195,6 +195,7 @@ BAND_PATHS = {
"web/modules/settings.js": None,
"web/modules/widgets.js": None,
"web/tests/harness_login_cards.test.js": "Login-card suite grew past 1000 lines with the name-the-account face cases (agy pickup, issue #232); split when the next face lands.",
"web/tests/review_presentation.test.js": "Review Checkpoint lifecycle and verdict reconciliation remain covered by one focused presentation suite.",
}
BYTE_BASELINE_DEBT = {

View file

@ -4,7 +4,8 @@ Trimmed in v5.16.0-rc.2: only the fixture-backed parametrized contract
remains here. The previous CSS/JS source-string assertions (chat composer
geometry, live-card timeline, plan-mode dropdown, mobile keyboard layout,
etc.) were retired — those surfaces are exercised by the Playwright
``ui_browser`` smoke suite where it applies (`tests/test_ui_smoke_playwright.py`)
``ui_browser`` smoke suite where it applies (`tests/test_ui_smoke_playwright.py`
and `tests/test_ui_smoke_review_checkpoint.py`)
and by manual UI review for the rest. Static pin-by-string assertions
were retired with the understanding that a regression in those visual
surfaces will surface in the ui_browser CI lane rather than in unit

View file

@ -9,7 +9,8 @@ import pytest
def test_gateway_frontend_uses_api_client_boundary():
"""Minimal UI-browser lane sentinel for the Gateway Boundary refactor.
Full browser launch coverage remains in ``test_ui_smoke_playwright.py``.
Full browser launch coverage remains in the ``test_ui_smoke_playwright.py``
and ``test_ui_smoke_review_checkpoint.py`` smoke modules.
This focused check keeps the lane aware of the new frontend boundary even
when Playwright is unavailable locally.
"""

View file

@ -20,13 +20,18 @@ REPO_ROOT = os.path.dirname(os.path.dirname(__file__))
def _open_review_checkpoint(card, *, open_card=True):
if open_card:
card.locator(":scope > [data-live-summary-button]").click()
section = card.locator("[data-review-section]")
assert card.is_visible()
section = card.locator(":scope > [data-live-reviews-host] [data-review-section]")
section.wait_for(state="visible", timeout=5_000)
assert section.get_attribute("data-expanded") == "0"
section.locator("[data-review-section-toggle]").click()
assert section.get_attribute("data-expanded") == "1"
group = section.locator("[data-review-group]").first
assert group.locator("[data-review-group-toggle]").get_attribute("aria-expanded") == "false"
group.locator("[data-review-group-toggle]").click()
attempt = group.locator("[data-review-attempt-toggle]").first
attempt.click()
assert attempt.get_attribute("aria-expanded") == "true"
def _free_port() -> int:
@ -954,165 +959,6 @@ def test_ui_smoke_direct_mode_loads_chat_and_dashboard(direct_server):
_run_core_ui_assertions(direct_server)
@pytest.mark.ui_browser
def test_ui_smoke_review_truth_is_visible_in_chat_and_logs(direct_server_with_data):
pytest.importorskip("playwright.sync_api", reason="Playwright is not installed")
from playwright.sync_api import Error as PlaywrightError
from playwright.sync_api import sync_playwright
url = direct_server_with_data["url"]
data_dir = direct_server_with_data["data_dir"]
logs_dir = data_dir / "logs"
logs_dir.mkdir(parents=True, exist_ok=True)
projection = {
"panels": [{
"panel_id": "panel_visual_truth",
"surface": "task_acceptance",
"authority": "host_root",
"aggregate_signal": "DEGRADED",
"transport_status": "partial",
"parse_status": "malformed",
"quorum": {"required": 2, "contributed": 1, "configured": 3},
"enforcement_impact": "degrades_completion",
"reason": "One reviewer timed out, so the panel did not reach quorum.",
"candidate_hash": "candidate-visual",
"evidence_revision": "evidence-visual",
"fence_hash": "fence-visual-hash",
"actors": [
{
"slot_id": "fable",
"actor_role": "task acceptance",
"provider": "anthropic",
"model": "anthropic/claude-fable-5",
"transport_status": "success",
"parse_status": "valid",
"semantic_verdict": "DEGRADED",
"quorum_contribution": True,
"enforcement_impact": "supports_pass",
"reason": "The browser evidence is incomplete.",
},
{
"slot_id": "sol",
"actor_role": "task acceptance",
"provider": "openai",
"model": "openai/gpt-5.6-sol",
"transport_status": "timeout",
"parse_status": "malformed",
"semantic_verdict": "",
"quorum_contribution": False,
"enforcement_impact": "abstains",
"reason": "Provider request timed out.",
},
],
}],
}
axes = {
"lifecycle": {"status": "completed"},
"execution": {"status": "ok"},
"objective": {"status": "best_effort"},
"review": {"status": "degraded"},
"artifacts": {"status": "ready"},
}
summary = {
"ts": "2026-07-15T10:00:00+00:00",
"direction": "system",
"type": "task_summary",
"task_id": "review-ui",
"chat_id": 1,
"text": "Task finished with review evidence.",
"tool_calls": 0,
"rounds": 1,
"outcome_axes": axes,
"review_projection": projection,
}
event = {
"ts": "2026-07-15T10:00:01+00:00",
"type": "task_done",
"task_id": "review-ui",
"task_type": "task",
"status": "completed",
"outcome_axes": axes,
"review_projection": projection,
}
ordinary_final = {
"ts": "2026-07-15T10:00:00.500000+00:00",
"direction": "out",
"chat_id": 1,
"task_id": "review-no-summary",
"text": "Normal final answer after the terminal progress anchor.",
"format": "markdown",
}
(logs_dir / "chat.jsonl").write_text(
json.dumps(summary) + "\n" + json.dumps(ordinary_final) + "\n",
encoding="utf-8",
)
(logs_dir / "events.jsonl").write_text(json.dumps(event) + "\n", encoding="utf-8")
(logs_dir / "progress.jsonl").write_text(json.dumps({
"ts": "2026-07-15T09:59:59+00:00",
"chat_id": 1,
"task_id": "review-no-summary",
"content": "Terminal review must survive without a task summary.",
}) + "\n", encoding="utf-8")
task_results = data_dir / "task_results"
task_results.mkdir(parents=True, exist_ok=True)
(task_results / "review-no-summary.json").write_text(json.dumps({
"task_id": "review-no-summary",
"status": "completed",
"reason_code": "acceptance_degraded",
"outcome_axes": axes,
"review_projection": projection,
}) + "\n", encoding="utf-8")
try:
with sync_playwright() as pw:
browser = pw.chromium.launch(headless=True)
page = browser.new_page(viewport={"width": 1440, "height": 1000})
try:
page.goto(url, wait_until="domcontentloaded", timeout=30_000)
card = page.locator('.chat-live-card[data-task-id="review-ui"]')
card.wait_for(state="attached", timeout=30_000)
assert card.get_attribute("data-expanded") == "0"
assert "Review panel panel_visual_truth" not in card.inner_text()
_open_review_checkpoint(card)
chat_text = card.inner_text()
assert "Done with warnings" in chat_text
assert "Notice" not in chat_text
assert "Review panel panel_visual_truth" in chat_text
assert "Reviewer fable" in chat_text
assert "Reviewer sol" in chat_text
no_summary = page.locator('.chat-live-card[data-task-id="review-no-summary"]')
no_summary.wait_for(state="attached", timeout=30_000)
assert no_summary.get_attribute("data-expanded") == "0"
_open_review_checkpoint(no_summary)
assert no_summary.locator('[data-live-phase]').first.get_attribute("data-phase") == "warn"
assert "Review panel panel_visual_truth" in no_summary.inner_text()
page.wait_for_timeout(900) # cover the routine background history sync
assert no_summary.locator('.chat-live-line-repeat:not([hidden])').count() == 0
assert card.locator('.chat-live-line-repeat:not([hidden])').count() == 0
page.screenshot(path=str(data_dir.parent / "review-truth-chat.png"), full_page=True)
page.click('[data-nav-page="dashboard"]')
page.click('[data-dashboard-tab="logs"]')
log_card = page.locator('.log-task-card[data-task-group="review-ui"]')
log_card.wait_for(state="attached", timeout=30_000)
assert log_card.is_visible()
review = log_card.locator('[data-task-review]')
assert review.is_visible()
log_text = review.inner_text()
assert "Review panel panel_visual_truth" in log_text
assert "Reviewer fable" in log_text
assert "Reviewer sol" in log_text
assert log_card.locator('[data-task-phase]').inner_text() == "warn"
review.scroll_into_view_if_needed()
review.screenshot(path=str(data_dir.parent / "review-truth-logs.png"))
finally:
browser.close()
except PlaywrightError as exc:
if "Executable doesn't exist" in str(exc) or "playwright install" in str(exc).lower():
pytest.skip(str(exc))
raise
@pytest.mark.ui_browser
@pytest.mark.parametrize("browser_engine", ["chromium", "webkit"])
def test_ui_smoke_collapsed_activity_line_named_vs_unnamed(
@ -1450,6 +1296,7 @@ def test_ui_smoke_live_card_mutations_preserve_viewport(
assert abs(card_top(page, anchor_id) - anchor_before) <= 6
review_child = page.locator('.chat-live-card[data-task-id="vp-late-child"]')
review_before = review_child.evaluate("card => card.getBoundingClientRect().height")
emit(page, {
"type": "chat", "role": "assistant", "is_progress": True,
"chat_id": 1, "task_id": "vp-late-child",
@ -1469,7 +1316,9 @@ def test_ui_smoke_live_card_mutations_preserve_viewport(
})
assert review_child.get_attribute("data-expanded") == "0"
review_child.locator(":scope > [data-live-summary-button]").evaluate("el => el.click()")
page.evaluate(settle)
assert review_child.get_attribute("data-expanded") == "1"
assert review_child.evaluate("card => card.getBoundingClientRect().height") > review_before + 20
assert abs(card_top(page, anchor_id) - anchor_before) <= 6
page.screenshot(
path=str(data_dir.parent / f"live-card-viewport-{browser_engine}.png"),
@ -2066,6 +1915,7 @@ def test_ui_smoke_direct_mode_nests_subagent_child_cards(direct_server_with_data
assert "Scheduled subagent child1" not in expanded_text
assert child_summary.get_attribute("aria-expanded") == "true"
assert child.locator("[data-live-timeline]").first.get_attribute("id")
assert result_toggle.get_attribute("aria-controls")
page.reload(wait_until="domcontentloaded", timeout=30_000)
page.wait_for_function("() => document.querySelectorAll('.chat-live-card').length === 4", timeout=30_000)
@ -2090,6 +1940,7 @@ def test_ui_smoke_direct_mode_nests_subagent_child_cards(direct_server_with_data
assert replay_child.locator(":scope > [data-live-summary-button] [data-live-phase]").first.get_attribute("data-phase") == "warn"
assert replay_grandchild.get_attribute("data-finished") == "1"
assert replay_child.get_attribute("data-expanded") == "0"
assert replay_grandchild.get_attribute("data-expanded") == "0"
assert "researcher (child1)" in replay_child.inner_text()
assert "child=child1" not in replay_child.inner_text()
assert "role=researcher" not in replay_child.inner_text()

View file

@ -0,0 +1,168 @@
from __future__ import annotations
import json
import pytest
pytest_plugins = ("tests.test_ui_smoke_playwright",)
from tests.test_ui_smoke_playwright import _open_review_checkpoint
@pytest.mark.ui_browser
def test_ui_smoke_review_truth_is_visible_in_chat_and_logs(direct_server_with_data):
pytest.importorskip("playwright.sync_api", reason="Playwright is not installed")
from playwright.sync_api import Error as PlaywrightError
from playwright.sync_api import sync_playwright
url = direct_server_with_data["url"]
data_dir = direct_server_with_data["data_dir"]
logs_dir = data_dir / "logs"
logs_dir.mkdir(parents=True, exist_ok=True)
projection = {
"panels": [{
"panel_id": "panel_visual_truth",
"surface": "task_acceptance",
"authority": "host_root",
"aggregate_signal": "DEGRADED",
"transport_status": "partial",
"parse_status": "malformed",
"quorum": {"required": 2, "contributed": 1, "configured": 3},
"enforcement_impact": "degrades_completion",
"reason": "One reviewer timed out, so the panel did not reach quorum.",
"candidate_hash": "candidate-visual",
"evidence_revision": "evidence-visual",
"fence_hash": "fence-visual-hash",
"actors": [
{
"slot_id": "fable",
"actor_role": "task acceptance",
"provider": "anthropic",
"model": "anthropic/claude-fable-5",
"transport_status": "success",
"parse_status": "valid",
"semantic_verdict": "DEGRADED",
"quorum_contribution": True,
"enforcement_impact": "supports_pass",
"reason": "The browser evidence is incomplete.",
},
{
"slot_id": "sol",
"actor_role": "task acceptance",
"provider": "openai",
"model": "openai/gpt-5.6-sol",
"transport_status": "timeout",
"parse_status": "malformed",
"semantic_verdict": "",
"quorum_contribution": False,
"enforcement_impact": "abstains",
"reason": "Provider request timed out.",
},
],
}],
}
axes = {
"lifecycle": {"status": "completed"},
"execution": {"status": "ok"},
"objective": {"status": "best_effort"},
"review": {"status": "degraded"},
"artifacts": {"status": "ready"},
}
summary = {
"ts": "2026-07-15T10:00:00+00:00",
"direction": "system",
"type": "task_summary",
"task_id": "review-ui",
"chat_id": 1,
"text": "Task finished with review evidence.",
"tool_calls": 0,
"rounds": 1,
"outcome_axes": axes,
"review_projection": projection,
}
event = {
"ts": "2026-07-15T10:00:01+00:00",
"type": "task_done",
"task_id": "review-ui",
"task_type": "task",
"status": "completed",
"outcome_axes": axes,
"review_projection": projection,
}
ordinary_final = {
"ts": "2026-07-15T10:00:00.500000+00:00",
"direction": "out",
"chat_id": 1,
"task_id": "review-no-summary",
"text": "Normal final answer after the terminal progress anchor.",
"format": "markdown",
}
(logs_dir / "chat.jsonl").write_text(
json.dumps(summary) + "\n" + json.dumps(ordinary_final) + "\n",
encoding="utf-8",
)
(logs_dir / "events.jsonl").write_text(json.dumps(event) + "\n", encoding="utf-8")
(logs_dir / "progress.jsonl").write_text(json.dumps({
"ts": "2026-07-15T09:59:59+00:00",
"chat_id": 1,
"task_id": "review-no-summary",
"content": "Terminal review must survive without a task summary.",
}) + "\n", encoding="utf-8")
task_results = data_dir / "task_results"
task_results.mkdir(parents=True, exist_ok=True)
(task_results / "review-no-summary.json").write_text(json.dumps({
"task_id": "review-no-summary",
"status": "completed",
"reason_code": "acceptance_degraded",
"outcome_axes": axes,
"review_projection": projection,
}) + "\n", encoding="utf-8")
try:
with sync_playwright() as pw:
browser = pw.chromium.launch(headless=True)
page = browser.new_page(viewport={"width": 1440, "height": 1000})
try:
page.goto(url, wait_until="domcontentloaded", timeout=30_000)
card = page.locator('.chat-live-card[data-task-id="review-ui"]')
card.wait_for(state="attached", timeout=30_000)
assert card.get_attribute("data-expanded") == "0"
assert "Review panel panel_visual_truth" not in card.inner_text()
_open_review_checkpoint(card)
chat_text = card.inner_text()
assert "Done with warnings" in chat_text
assert "Notice" not in chat_text
assert "Review panel panel_visual_truth" in chat_text
assert "Reviewer fable" in chat_text
assert "Reviewer sol" in chat_text
no_summary = page.locator('.chat-live-card[data-task-id="review-no-summary"]')
no_summary.wait_for(state="attached", timeout=30_000)
assert no_summary.get_attribute("data-expanded") == "0"
_open_review_checkpoint(no_summary)
assert no_summary.locator('[data-live-phase]').first.get_attribute("data-phase") == "warn"
assert "Review panel panel_visual_truth" in no_summary.inner_text()
page.wait_for_timeout(900) # cover the routine background history sync
assert no_summary.locator('.chat-live-line-repeat:not([hidden])').count() == 0
assert card.locator('.chat-live-line-repeat:not([hidden])').count() == 0
page.screenshot(path=str(data_dir.parent / "review-truth-chat.png"), full_page=True)
page.click('[data-nav-page="dashboard"]')
page.click('[data-dashboard-tab="logs"]')
log_card = page.locator('.log-task-card[data-task-group="review-ui"]')
log_card.wait_for(state="attached", timeout=30_000)
assert log_card.is_visible()
review = log_card.locator('[data-task-review]')
assert review.is_visible()
log_text = review.inner_text()
assert "Review panel panel_visual_truth" in log_text
assert "Reviewer fable" in log_text
assert "Reviewer sol" in log_text
assert log_card.locator('[data-task-phase]').inner_text() == "warn"
review.scroll_into_view_if_needed()
review.screenshot(path=str(data_dir.parent / "review-truth-logs.png"))
finally:
browser.close()
except PlaywrightError as exc:
if "Executable doesn't exist" in str(exc) or "playwright install" in str(exc).lower():
pytest.skip(str(exc))
raise

View file

@ -135,7 +135,10 @@ function normalizeSkillAttempt(attempt, defaults = {}, ordinal = 0) {
: statusTone(state, rawStatus),
verdict: rawStatus,
lifecycleStatus,
timestamp: text(attempt.ts || attempt.timestamp),
timestamp: text(
attempt.ts || attempt.timestamp || attempt.finished_at
|| attempt.started_at || attempt.queued_at,
),
ordinal,
label: [
attempt.review_round != null ? `round ${attempt.review_round}` : '',
@ -251,6 +254,10 @@ export function classifyReviewLifecycle(row) {
const status = text(lifecycle.status || lifecycle.phase || 'queued');
const reviewStatus = text(lifecycle.review_status || lifecycle.review_verdict);
const lifecycleOnly = !reviewStatus;
const timestamp = text(
lifecycle.finished_at || lifecycle.started_at || lifecycle.queued_at
|| lifecycle.ts || row?.ts || row?.timestamp,
);
const group = normalizeSkillGroup({
surface: 'skill',
id: groupId,
@ -268,6 +275,7 @@ export function classifyReviewLifecycle(row) {
count_is_authoritative: lifecycle.count_is_authoritative === true,
attempts: lifecycle.job_id || lifecycle.id ? [{
...lifecycle,
ts: timestamp,
job_id: lifecycle.job_id || lifecycle.id,
status: reviewStatus,
review_status: reviewStatus,
@ -306,6 +314,10 @@ export function classifyReviewLifecyclePointer(row) {
const reviewStatus = text(pointer.review_status || pointer.review_verdict);
const lifecycleOnly = !reviewStatus;
const initiatorTaskId = text(pointer.initiator_task_id || pointer.origin_task_id);
const timestamp = text(
pointer.finished_at || pointer.started_at || pointer.queued_at
|| pointer.ts || row?.ts || row?.timestamp,
);
const group = normalizeSkillGroup({
surface: 'skill',
id: groupId,
@ -323,6 +335,7 @@ export function classifyReviewLifecyclePointer(row) {
count_is_authoritative: pointer.count_is_authoritative === true,
attempts: pointer.job_id ? [{
...pointer,
ts: timestamp,
initiator_task_id: initiatorTaskId,
status: reviewStatus,
review_status: reviewStatus,
@ -609,6 +622,13 @@ export function mergeReviewGroup(store, incoming) {
lifecycleOnly: false,
};
}
// A lifecycle frame for an already-semantic attempt carries a transport
// timestamp, not a new attempt timestamp. Keep the domain row's time as
// the ordering key so a late frame cannot move that attempt past newer
// semantic history.
if (attempt.lifecycleOnly && previous && !previous.lifecycleOnly) {
mergedAttempt.timestamp = previous.timestamp || mergedAttempt.timestamp;
}
mergedById.set(attempt.id, mergedAttempt);
if (incomingActive && previousTerminal) hasStaleActiveAttempt = true;
if (incomingActive && !previousTerminal && !previous) introducedActiveAttempt = true;
@ -632,13 +652,26 @@ export function mergeReviewGroup(store, incoming) {
const priorOnly = priorIds.filter((id) => !incomingIds.has(id));
// A projection that contains every known attempt owns their order (for
// example, terminal Skill history arriving after one live row). A bounded
// projection that omits a known attempt cannot move it to the front: keep
// the established order and append only genuinely new identities.
const order = priorOnly.length === 0
? incoming.attempts.map((attempt) => attempt.id)
: [...priorIds, ...incoming.attempts
.filter((attempt) => !priorById.has(attempt.id))
.map((attempt) => attempt.id)];
// projection that omits a known attempt keeps established order; a new
// timestamped identity is inserted before a later known timestamp.
let order;
if (priorOnly.length === 0) {
order = incoming.attempts.map((attempt) => attempt.id);
} else {
order = [...priorIds];
for (const attempt of incoming.attempts) {
if (priorById.has(attempt.id) || order.includes(attempt.id)) continue;
const timestamp = text(attempt.timestamp);
const insertBefore = timestamp
? order.findIndex((id) => {
const existing = text(mergedById.get(id)?.timestamp);
return existing && existing > timestamp;
})
: -1;
if (insertBefore >= 0) order.splice(insertBefore, 0, attempt.id);
else order.push(attempt.id);
}
}
const staleActiveRegression = (
(prior.state === 'terminal' || prior.state === 'superseded')
&& (incoming.state === 'queued' || incoming.state === 'running')
@ -688,11 +721,19 @@ export function mergeReviewGroup(store, incoming) {
}
const priorLatestAttempt = prior.attempts.at(-1);
const mergedLatestAttempt = merged.attempts.at(-1);
const newAttemptHasProvenablyNewerTimestamp = (
priorLatestAttempt?.timestamp
&& mergedLatestAttempt
&& !priorById.has(mergedLatestAttempt.id)
&& !mergedLatestAttempt.lifecycleOnly
&& hasSemanticVerdict(mergedLatestAttempt.verdict)
&& text(mergedLatestAttempt.timestamp) > text(priorLatestAttempt.timestamp)
);
if (
!activeAttempts.length
&& priorLatestAttempt?.lifecycleOnly
&& mergedLatestAttempt?.id === priorLatestAttempt.id
&& !incomingIds.has(priorLatestAttempt.id)
&& !newAttemptHasProvenablyNewerTimestamp
) {
// A stale history refresh may omit a newer terminal lifecycle row. It
// must not resurrect the older typed verdict at group level while the
@ -702,6 +743,35 @@ export function mergeReviewGroup(store, incoming) {
merged.verdict = priorLatestAttempt.verdict || '';
merged.summary = priorLatestAttempt.summary || merged.summary;
merged.lifecycleOnly = true;
merged.lifecycleStatus = priorLatestAttempt.lifecycleStatus || merged.lifecycleStatus;
}
const delayedSemanticProjection = (
incoming.surface === 'skill'
&& priorOnly.length > 0
&& !activeAttempts.length
&& priorLatestAttempt
&& !priorLatestAttempt.lifecycleOnly
&& hasSemanticVerdict(priorLatestAttempt.verdict)
&& !incomingIds.has(priorLatestAttempt.id)
&& mergedLatestAttempt
&& !mergedLatestAttempt.lifecycleOnly
&& hasSemanticVerdict(mergedLatestAttempt.verdict)
&& (
!text(priorLatestAttempt.timestamp)
|| !text(mergedLatestAttempt.timestamp)
|| text(mergedLatestAttempt.timestamp) <= text(priorLatestAttempt.timestamp)
)
);
if (delayedSemanticProjection) {
// A bounded history response may contain an older semantic row that was
// not known when the newer row arrived live. Keep the group header tied
// to the newer proved attempt; the older row remains inspectable below.
merged.state = priorLatestAttempt.state;
merged.tone = priorLatestAttempt.tone;
merged.verdict = priorLatestAttempt.verdict;
merged.summary = priorLatestAttempt.summary || merged.summary;
merged.lifecycleOnly = false;
merged.lifecycleStatus = priorLatestAttempt.lifecycleStatus || merged.lifecycleStatus;
}
merged.initiatorTaskId = uniformAttemptInitiator(
merged.attempts,

View file

@ -506,6 +506,103 @@ test('lifecycle completion stays neutral until a semantic review verdict arrives
assert.equal(staleMerged.attempts.at(-1).id, 'job-3');
});
test('lifecycle timestamps fence delayed older history without inventing a verdict', () => {
const lifecycle = reviewGroupFromLifecycle({
ts: '2026-08-26T10:03:00Z',
lifecycle: {
kind: 'review', status: 'succeeded', target: 'alpha', job_id: 'job-3',
group_id: 'task:root:alpha', presentation_owner_task_id: 'root',
},
});
assert.equal(lifecycle.attempts[0].timestamp, '2026-08-26T10:03:00Z');
const store = new Map();
mergeReviewGroup(store, reviewGroupFromHistoryRow(groupedSkillRow({
attempts: [{ job_id: 'job-1', skill: 'alpha', status: 'blockers', ts: '2026-08-26T10:00:00Z' }],
})));
mergeReviewGroup(store, lifecycle);
const delayed = reviewGroupFromHistoryRow(groupedSkillRow({
attempts: [
{ job_id: 'job-1', skill: 'alpha', status: 'blockers', ts: '2026-08-26T10:00:00Z' },
{ job_id: 'job-2', skill: 'alpha', status: 'clean', ts: '2026-08-26T10:02:00Z' },
],
}));
const merged = mergeReviewGroup(store, delayed);
assert.deepEqual(merged.attempts.map((attempt) => attempt.id), ['job-1', 'job-2', 'job-3']);
assert.equal(merged.lifecycleOnly, true);
assert.equal(merged.verdict, '');
assert.equal(merged.tone, 'neutral');
const newer = reviewGroupFromHistoryRow(groupedSkillRow({
attempts: [{ job_id: 'job-4', skill: 'alpha', status: 'clean', ts: '2026-08-26T10:04:00Z' }],
}));
const upgraded = mergeReviewGroup(store, newer);
assert.equal(upgraded.attempts.at(-1).id, 'job-4');
assert.equal(upgraded.verdict, 'clean');
assert.equal(upgraded.tone, 'done');
const mixed = reviewGroupFromHistoryRow(groupedSkillRow({
attempts: [
{ job_id: 'job-2', skill: 'alpha', status: 'clean', ts: '2026-08-26T10:02:00Z' },
{ job_id: 'job-4', skill: 'alpha', status: 'clean', ts: '2026-08-26T10:04:00Z' },
],
}));
const mixedStore = new Map();
mergeReviewGroup(mixedStore, reviewGroupFromHistoryRow(groupedSkillRow({
attempts: [{ job_id: 'job-1', skill: 'alpha', status: 'blockers', ts: '2026-08-26T10:00:00Z' }],
})));
mergeReviewGroup(mixedStore, lifecycle);
const mixedMerged = mergeReviewGroup(mixedStore, mixed);
assert.deepEqual(mixedMerged.attempts.map((attempt) => attempt.id), ['job-1', 'job-2', 'job-3', 'job-4']);
assert.equal(mixedMerged.verdict, 'clean');
assert.equal(mixedMerged.tone, 'done');
const semanticHistory = reviewGroupFromHistoryRow(groupedSkillRow({
attempts: [
{ job_id: 'job-1', skill: 'alpha', status: 'clean', ts: '2026-08-26T10:00:00Z' },
{ job_id: 'job-2', skill: 'alpha', status: 'blockers', ts: '2026-08-26T10:02:00Z' },
],
}));
const lateLifecycle = reviewGroupFromLifecycle({
lifecycle: {
kind: 'review', status: 'succeeded', target: 'alpha', job_id: 'job-2',
finished_at: '2026-08-26T10:05:00Z',
group_id: 'task:root:alpha', presentation_owner_task_id: 'root',
},
});
const sameAttemptStore = new Map();
mergeReviewGroup(sameAttemptStore, semanticHistory);
mergeReviewGroup(sameAttemptStore, lateLifecycle);
const afterLateLifecycle = sameAttemptStore.get('task:root:alpha');
assert.equal(afterLateLifecycle.attempts.find((attempt) => attempt.id === 'job-2').timestamp, '2026-08-26T10:02:00Z');
const newerHistory = reviewGroupFromHistoryRow(groupedSkillRow({
attempts: [{ job_id: 'job-3', skill: 'alpha', status: 'clean', ts: '2026-08-26T10:04:00Z' }],
}));
const afterNewerHistory = mergeReviewGroup(sameAttemptStore, newerHistory);
assert.deepEqual(afterNewerHistory.attempts.map((attempt) => attempt.id), ['job-1', 'job-2', 'job-3']);
const semanticStaleStore = new Map();
mergeReviewGroup(semanticStaleStore, reviewGroupFromHistoryRow(groupedSkillRow({
attempts: [{ job_id: 'job-3', skill: 'alpha', status: 'blockers', ts: '2026-08-26T10:03:00Z' }],
})));
const semanticStale = mergeReviewGroup(semanticStaleStore, reviewGroupFromHistoryRow(groupedSkillRow({
status: 'clean',
attempts: [{ job_id: 'job-2', skill: 'alpha', status: 'clean', ts: '2026-08-26T10:02:00Z' }],
})));
assert.deepEqual(semanticStale.attempts.map((attempt) => attempt.id), ['job-2', 'job-3']);
assert.equal(semanticStale.verdict, 'blockers');
assert.equal(semanticStale.tone, 'error');
const pointer = classifyReviewLifecyclePointer({
ts: '2026-08-26T11:00:00Z',
lifecycle_pointer: {
kind: 'review', status: 'running', target: 'alpha', job_id: 'pointer-1',
group_id: 'task:root:alpha', presentation_owner_task_id: 'root',
},
});
assert.equal(pointer.group.attempts[0].timestamp, '2026-08-26T11:00:00Z');
});
test('task acceptance adapts only task_acceptance panels; advisory and commit stay omitted', () => {
const detail = {
task_id: 'root',