chore(qa): qualify execution identity across live boundaries (#156033)

* test(qa): repair forced restart and message inspection harnesses

* test(e2e): qualify installed execution identity persistence

Extend the packed npm onboarding owner with audit opt-in, one deterministic local turn, installed CLI inspection before and after Gateway restart, and isolated-state/privacy assertions.

Qualification from 267bb9d2: new shell-flow proof fails before the harness change at the missing opt-in. All 53 focused support tests pass; final changed shell/SQLite selection took 26.94s with one worker. Changed checks and P0-P2 autoreview pass; the full export scan was reused after brace-only lint fixes. Packed Docker proof remains a remote handoff.

* test(qa): qualify live execution and channel identity

* test(audit): qualify execution identity lifecycle gaps

* test(qa): align suppression audit with required replies

* test(qa): qualify Telegram participant identity

* test(qa): provision Telegram identity fixtures

* test(qa): qualify private-production Telegram identity

* test(e2e): preserve installed execution identity proof

* test(qa): await post-delivery memory maintenance

* fix(qa): repair qualification CI boundaries
This commit is contained in:
Josh Avant 2026-09-22 20:15:02 -05:00 • committed by GitHub
parent 59a734b735
commit fdae00fe82
No known key found for this signature in database
GPG key ID: B5690EEEBB952194
49 changed files with 5840 additions and 402 deletions

View file

@ -76,6 +76,33 @@ Keep one TDLib client per restored state directory. Run custom TDLib inspection
before the recorder starts or after it exits, under the same live lease. Bot API
inspection can use the scenario `command` action while recording.
## QA Lab participant identity fixtures
The QA Lab Telegram adapter accepts optional fields on one Convex-leased Test
Server credential: `forumGroupId`, positive numeric `forumTopicId`, and
`participants`. Each additional participant supplies a unique lowercase `alias`,
`testerUserId`, `tdlibArchiveBase64`, `tdlibArchiveSha256`, and `tdlibVersion`.
The pool owner provisions these independently authorized users under the same
lease. The SUT bot and every participant must already belong to the selected
group and forum, and the topic must exist. No credential is acquired by merely
listing scenarios or running deterministic support tests.
For mixed-user flows, use `senderId: primary` and the additional aliases. A
single-user fixture binds its first scenario sender label to its leased user;
changing that label cannot impersonate a second person. `conversation.kind:
direct` sends to the SUT DM. A group/channel conversation uses `groupId`; adding
a positive numeric `threadId` selects that forum topic in `forumGroupId` (or the
existing group when it is itself a forum). Each native chat/topic belongs to one
logical conversation until transport reset. Replies require a receipt observed
by the sending participant in that chat; TDLib message IDs cannot cross accounts.
Flow preparation exposes in-memory `telegramIdentityFixture.participantAliases`
and `forumTopicId`. Set `execution.config.requireParticipantIdentityFixture:
true` to require the complete mixed-user/forum fixture before sending. It also
retains the existing `readTelegramMessages()` observer for native topic evidence.
These inputs enable real Telegram identity proof; deterministic adapter tests do
not claim that live transport or audit inspection has run.
## Backends
| Backend | Use |

View file

@ -50,7 +50,7 @@ export function parseTelegramTestCredential(value) {
}
const tdlibArchiveBase64 = requireString(payload, "tdlibArchiveBase64");
decodeBase64(tdlibArchiveBase64);
return {
const credential = {
schemaVersion: 1,
environment: "test",
groupId: requireIntegerString(payload, "groupId", /^-?\d+$/u),
@ -62,6 +62,49 @@ export function parseTelegramTestCredential(value) {
tdlibArchiveSha256,
tdlibVersion: requireString(payload, "tdlibVersion"),
};
if (payload.forumGroupId !== undefined) {
credential.forumGroupId = requireIntegerString(payload, "forumGroupId", /^-\d+$/u);
}
if (payload.forumTopicId !== undefined) {
if (!Number.isSafeInteger(payload.forumTopicId) || payload.forumTopicId <= 0) {
throw new Error("Telegram QA forumTopicId must be a positive integer.");
}
credential.forumTopicId = payload.forumTopicId;
}
if (payload.participants !== undefined) {
if (!Array.isArray(payload.participants)) {
throw new Error("Telegram QA participants must be an array.");
}
const identities = new Set([credential.testerUserId]);
const aliases = new Set(["primary"]);
credential.participants = payload.participants.map((value) => {
const participant = requireObject(value, "Telegram QA participant");
const alias = requireString(participant, "alias");
if (!/^[a-z][a-z0-9-]*$/u.test(alias) || aliases.has(alias)) {
throw new Error("Telegram QA participants require distinct lowercase aliases.");
}
const parsed = parseTelegramTestCredential({
...credential,
testerUserId: participant.testerUserId,
tdlibArchiveBase64: participant.tdlibArchiveBase64,
tdlibArchiveSha256: participant.tdlibArchiveSha256,
tdlibVersion: participant.tdlibVersion,
});
if (identities.has(parsed.testerUserId)) {
throw new Error("Telegram QA mixed participants require distinct leased user identities.");
}
aliases.add(alias);
identities.add(parsed.testerUserId);
return {
alias,
testerUserId: parsed.testerUserId,
tdlibArchiveBase64: parsed.tdlibArchiveBase64,
tdlibArchiveSha256: parsed.tdlibArchiveSha256,
tdlibVersion: parsed.tdlibVersion,
};
});
}
return credential;
}
function normalizeArchiveEntry(entry) {

View file

@ -218,3 +218,52 @@ test("failed broker release retains only the private handle for same-owner recov
fs.rmSync(fixture, { recursive: true, force: true });
}
});
test("validates optional forum and distinct participant fixtures without echoing private fields", () => {
const base = {
schemaVersion: 1,
environment: "test",
groupId: "-1001",
sutToken: "synthetic-token",
sutUsername: "sut_bot",
sutBotId: "200",
testerUserId: "100",
tdlibArchiveBase64: "YQ==",
tdlibArchiveSha256: "a".repeat(64),
tdlibVersion: "1.8.67",
};
const second = {
alias: "second",
testerUserId: "101",
tdlibArchiveBase64: "Yg==",
tdlibArchiveSha256: "b".repeat(64),
tdlibVersion: "1.8.67",
};
assert.deepEqual(
parseTelegramTestCredential({
...base,
forumGroupId: "-1002",
forumTopicId: 42,
participants: [second],
}).participants,
[second],
);
for (const patch of [
{ participants: [second, second] },
{ participants: [{ ...second, testerUserId: "100" }] },
{ participants: [{ ...second, alias: "primary" }] },
{ participants: [{ ...second, tdlibArchiveBase64: "private-invalid-value" }] },
{ participants: [{ ...second, tdlibArchiveSha256: "private-invalid-value" }] },
{ forumGroupId: "1002" },
{ forumTopicId: 0 },
]) {
assert.throws(
() => parseTelegramTestCredential({ ...base, ...patch }),
(error) => {
assert.equal(error.message.includes("private-invalid-value"), false);
assert.equal(error.message.includes(base.sutToken), false);
return true;
},
);
}
});

View file

@ -523,6 +523,19 @@ class UserDriver:
elif state == "authorizationStateWaitPassword":
password = getattr(args, "password", "") or prompt_secret("Telegram 2FA password: ")
self.client.send({"@type": "checkAuthenticationPassword", "password": password})
elif state == "authorizationStateWaitRegistration":
first_name = getattr(args, "first_name", "").strip()
if not first_name:
raise DriverError(
"New Telegram Test Server users require login --first-name."
)
self.client.send(
{
"@type": "registerUser",
"first_name": first_name,
"last_name": getattr(args, "last_name", "").strip(),
}
)
elif state == "authorizationStateReady":
return True
elif state in {"authorizationStateClosing", "authorizationStateClosed", "authorizationStateLoggingOut"}:
@ -1112,6 +1125,200 @@ def cleanup_owned_group(driver, manifest_path):
return {"ok": True, **record}
def private_forum_identity(driver):
if driver.config.get("testDc") is True:
raise DriverError("Private production forum setup requires Telegram production.")
me = driver.client.request({"@type": "getMe"})
if str(me["id"]) != str(driver.config.get("testerUserId")):
raise DriverError("Private production forum setup requires the leased QA user.")
sut = resolve_sut(driver.config, driver.bot_config)
if not sut["id"] or not sut["username"]:
raise DriverError("Private production forum setup requires the SUT bot identity.")
return {
"testerUserId": str(me["id"]),
"sutBotId": str(sut["id"]),
"sutUsername": sut["username"],
}
def public_private_forum_record(record):
return {
key: record[key]
for key in ("ok", "status", "groupId", "forumTopicId", "title", "topicTitle")
if key in record
}
def prepare_private_forum(driver, manifest_path):
identity = private_forum_identity(driver)
if manifest_path.exists():
raise DriverError(
"This lease already has a private forum record; clean it before another creation."
)
record = {
**identity,
"status": "creating",
"title": f"OpenClaw private QA {secrets.token_hex(6)}",
"topicTitle": f"Identity proof {secrets.token_hex(4)}",
}
write_json_private(manifest_path, record)
try:
basic = driver.client.request(
{
"@type": "createNewBasicGroupChat",
"user_ids": [],
"title": record["title"],
"message_auto_delete_time": 0,
}
)
record.update(status="basic-group-created", basicGroupId=str(basic["chat_id"]))
write_json_private(manifest_path, record)
upgraded = driver.client.request(
{
"@type": "upgradeBasicGroupChatToSupergroupChat",
"chat_id": int(record["basicGroupId"]),
}
)
if (
upgraded["type"].get("@type") != "chatTypeSupergroup"
or upgraded["type"].get("is_channel") is True
):
raise DriverError("Telegram did not upgrade the private fixture to a supergroup.")
record.update(
status="supergroup-created",
groupId=str(upgraded["id"]),
supergroupId=str(upgraded["type"]["supergroup_id"]),
)
write_json_private(manifest_path, record)
driver.client.request(
{
"@type": "setChatMemberStatus",
"chat_id": int(record["groupId"]),
"member_id": {
"@type": "messageSenderUser",
"user_id": int(identity["sutBotId"]),
},
"status": {
"@type": "chatMemberStatusMember",
"member_until_date": 0,
},
}
)
record["status"] = "bot-added"
write_json_private(manifest_path, record)
driver.client.request(
{
"@type": "toggleSupergroupIsForum",
"supergroup_id": int(record["supergroupId"]),
"is_forum": True,
"has_forum_tabs": True,
}
)
record["status"] = "forum-enabled"
write_json_private(manifest_path, record)
topic = driver.client.request(
{
"@type": "createForumTopic",
"chat_id": int(record["groupId"]),
"name": record["topicTitle"],
"is_name_implicit": False,
"icon": {
"@type": "forumTopicIcon",
"color": 0x6FB9F0,
"custom_emoji_id": 0,
},
}
)
record.update(status="topic-created", forumTopicId=int(topic["forum_topic_id"]))
write_json_private(manifest_path, record)
invite = driver.client.request(
{
"@type": "createChatInviteLink",
"chat_id": int(record["groupId"]),
"name": "OpenClaw private qualification",
"expiration_date": 0,
"member_limit": 1,
"creates_join_request": False,
}
)
record.update(status="ready", inviteLink=invite["invite_link"], ok=True)
write_json_private(manifest_path, record)
return public_private_forum_record(record)
except (DriverError, KeyError, TypeError, ValueError):
record["status"] = "setup-failed"
write_json_private(manifest_path, record)
raise
def cleanup_private_forum(driver, manifest_path):
record = read_json(manifest_path)
if not record:
return {"ok": True, "status": "not-created"}
identity = private_forum_identity(driver)
if any(record.get(key) != identity[key] for key in ("testerUserId", "sutBotId")):
raise DriverError("Private forum cleanup record belongs to a different identity.")
if record.get("status") == "deleted":
return public_private_forum_record(record)
group_id = record.get("groupId") or record.get("basicGroupId")
if not group_id:
record.pop("inviteLink", None)
record.update(status="not-created", ok=True)
write_json_private(manifest_path, record)
return public_private_forum_record(record)
chat = driver.client.request({"@type": "getChat", "chat_id": int(group_id)})
membership = driver.client.request(
{
"@type": "getChatMember",
"chat_id": int(group_id),
"member_id": {
"@type": "messageSenderUser",
"user_id": int(identity["testerUserId"]),
},
}
)
if (
chat["type"].get("@type") not in {"chatTypeBasicGroup", "chatTypeSupergroup"}
or chat["type"].get("is_channel") is True
or membership["status"].get("@type") != "chatMemberStatusCreator"
):
raise DriverError("The leased QA user cannot delete the private forum for all members.")
# The can_be_deleted projection can lag immediately after creation. The
# creator-owned delete is authoritative; retry only its observed propagation error.
for attempt in range(5):
try:
deletion = driver.client.request(
{"@type": "deleteChat", "chat_id": int(group_id)}
)
break
except TdRequestError as error:
if "The chat can't be deleted" not in str(error) or attempt == 4:
raise
time.sleep(1)
record.pop("inviteLink", None)
record.update(status="deleted", deletion=deletion, ok=True)
write_json_private(manifest_path, record)
return public_private_forum_record(record)
def command_private_forum(args):
if not os.environ.get("TELEGRAM_USER_DRIVER_STATE_DIR"):
raise DriverError("Private forum commands require runner-owned leased credential state.")
manifest = STATE_DIR / "owned-private-forum.json"
if args.command == "cleanup-private-forum" and not manifest.exists():
print_result({"ok": True, "status": "not-created"}, args.json, args.output)
return
config, bot_config = load_config()
driver = UserDriver(config, bot_config)
if not driver.authorize(args, need_ready=False):
raise DriverError("The leased QA user is not authorized.")
result = (
prepare_private_forum(driver, manifest)
if args.command == "prepare-private-forum"
else cleanup_private_forum(driver, manifest)
)
print_result(result, args.json, args.output)
def command_test_group(args):
if not os.environ.get("TELEGRAM_USER_DRIVER_STATE_DIR"):
raise DriverError("Test group commands require runner-owned leased credential state.")
@ -1315,6 +1522,8 @@ def serve_message(message, users):
"senderId": normalized.get("senderId"),
"senderUsername": normalized.get("senderUsername"),
"replyToMessageId": normalized.get("replyToMessageId"),
"threadId": normalized.get("threadId"),
"forumTopicId": (message.get("topic_id") or {}).get("forum_topic_id"),
"timestamp": int(message.get("date") or 0) * 1000,
"contentType": normalized["contentType"],
"text": normalized["text"],
@ -1425,6 +1634,11 @@ def command_serve(args):
chat_id = driver.resolve_chat(args.chat)
tester = driver.client.request({"@type": "getMe"})
driver.check_group_write_access(chat_id, tester["id"])
selected_chats = {chat_id}
for target in getattr(args, "observe_chat", []):
selected = driver.resolve_chat(target)
driver.check_group_write_access(selected, tester["id"])
selected_chats.add(selected)
write_ndjson(
{
"type": "ready",
@ -1435,7 +1649,7 @@ def command_serve(args):
known_messages = {}
while True:
update = driver.client.next_update(timeout=0.1)
if update and update_chat_id(update) == chat_id:
if update and update_chat_id(update) in selected_chats:
event = serve_update(update, driver.client, known_messages)
if event:
write_ndjson({"type": "update", "update": event})
@ -1451,13 +1665,27 @@ def command_serve(args):
request_id = request.get("id")
if not isinstance(request_id, str) or not request_id:
raise DriverError("serve command requires a string id")
if request.get("method") != "send":
method = request.get("method")
if method == "cleanup-private-forum":
result = cleanup_private_forum(
driver, STATE_DIR / "owned-private-forum.json"
)
write_ndjson({"type": "response", "id": request_id, "result": result})
continue
if method != "send":
raise DriverError("serve command method must be send")
text = request.get("text")
if not isinstance(text, str) or not text:
raise DriverError("serve send command requires text")
reply_to = request.get("replyToMessageId")
sent = driver.send_text(chat_id, text, reply_to=reply_to)
target = driver.resolve_chat(request["chatId"]) if "chatId" in request else chat_id
if target not in selected_chats:
driver.check_group_write_access(target, tester["id"])
selected_chats.add(target)
topic = request.get("forumTopicId")
if topic is not None and (type(topic) is not int or topic <= 0):
raise DriverError("serve forumTopicId must be a positive integer")
sent = driver.send_text(target, text, reply_to=reply_to, forum_topic_id=topic)
normalized = serve_message(sent, driver.client.users)
known_messages[(normalized["chatId"], normalized["messageId"])] = normalized
write_ndjson({"type": "response", "id": request_id, "result": normalized})
@ -1511,6 +1739,8 @@ def main():
login.add_argument("--phone", default="")
login.add_argument("--code", default="")
login.add_argument("--password", default="")
login.add_argument("--first-name", default="")
login.add_argument("--last-name", default="")
login.set_defaults(func=command_login)
status = sub.add_parser("status")
@ -1530,6 +1760,11 @@ def main():
group.add_argument("--chat", default="")
group.set_defaults(func=command_test_group)
for name in ("prepare-private-forum", "cleanup-private-forum"):
private_forum = sub.add_parser(name)
add_common(private_forum)
private_forum.set_defaults(func=command_private_forum)
confirm_qr = sub.add_parser("confirm-qr")
add_common(confirm_qr)
confirm_qr.add_argument("--link", required=True)
@ -1583,6 +1818,7 @@ def main():
serve = sub.add_parser("serve")
serve.add_argument("--chat", default="")
serve.add_argument("--observe-chat", action="append", default=[])
serve.add_argument("--timeout-ms", type=int, default=120000)
serve.set_defaults(func=command_serve)

View file

@ -42,6 +42,59 @@ def native_message(content, chat_id=-1001, message_id=42 << 20, sender_id=101):
}
class AuthorizationTest(unittest.TestCase):
def test_registers_a_new_test_user_with_explicit_names(self):
class Client:
def __init__(self):
self.requests = []
self.updates = [
{
"@type": "updateAuthorizationState",
"authorization_state": {
"@type": "authorizationStateWaitRegistration",
},
},
{
"@type": "updateAuthorizationState",
"authorization_state": {"@type": "authorizationStateReady"},
},
]
def execute(self, payload):
self.requests.append(payload)
def send(self, payload):
self.requests.append(payload)
def receive(self, _timeout):
return self.updates.pop(0)
instance = driver.UserDriver.__new__(driver.UserDriver)
instance.client = Client()
instance.printed_qr_link = None
instance.config = {}
instance.bot_config = {}
instance.authorize(
SimpleNamespace(
timeout_ms=1000,
phone="",
qr=False,
code="",
password="",
first_name="OpenClaw",
last_name="Guest",
),
)
self.assertIn(
{
"@type": "registerUser",
"first_name": "OpenClaw",
"last_name": "Guest",
},
instance.client.requests,
)
class OwnedGroupTest(unittest.TestCase):
def fixture(self):
class Client:
@ -101,6 +154,110 @@ class OwnedGroupTest(unittest.TestCase):
driver.prepare_owned_group(instance, Path(root) / "group.json")
self.assertEqual(instance.client.requests, [])
def test_private_production_forum_is_owned_and_deleted(self):
instance = self.fixture()
instance.config["testDc"] = False
def request(payload, timeout=20):
instance.client.requests.append(payload)
kind = payload["@type"]
if kind == "getMe":
return {"id": 123, "username": "leased_primary"}
if kind == "createNewBasicGroupChat":
self.assertEqual(payload["user_ids"], [])
return {"chat_id": -2042, "failed_to_add_members": {"failed_to_add_members": []}}
if kind == "upgradeBasicGroupChatToSupergroupChat":
self.assertEqual(payload["chat_id"], -2042)
return {"id": -1002042, "type": {"@type": "chatTypeSupergroup", "supergroup_id": 2042, "is_channel": False}}
if kind == "setChatMemberStatus":
self.assertEqual(payload["chat_id"], -1002042)
self.assertEqual(payload["member_id"]["user_id"], 42)
self.assertEqual(payload["status"]["@type"], "chatMemberStatusMember")
self.assertEqual(payload["status"]["member_until_date"], 0)
return {"@type": "ok"}
if kind == "toggleSupergroupIsForum":
self.assertEqual(payload, {"@type": kind, "supergroup_id": 2042, "is_forum": True, "has_forum_tabs": True})
return {"@type": "ok"}
if kind == "createForumTopic":
self.assertEqual(payload["chat_id"], -1002042)
return {"forum_topic_id": 777, "name": payload["name"]}
if kind == "createChatInviteLink":
return {"invite_link": "https://example.invalid/private-invite"}
if kind == "getChat":
return {
"id": -1002042,
"type": {"@type": "chatTypeSupergroup", "is_channel": False},
# The server can lag this projection immediately after creation.
"can_be_deleted_for_all_users": False,
}
if kind == "getChatMember":
return {"status": {"@type": "chatMemberStatusCreator"}}
if kind == "deleteChat":
return {"@type": "ok"}
raise AssertionError(kind)
instance.client.request = request
with tempfile.TemporaryDirectory() as root:
manifest = Path(root) / "private-forum.json"
prepared = driver.prepare_private_forum(instance, manifest)
self.assertEqual(prepared["groupId"], "-1002042")
self.assertEqual(prepared["forumTopicId"], 777)
self.assertNotIn("inviteLink", prepared)
self.assertIn("inviteLink", driver.read_json(manifest))
cleaned = driver.cleanup_private_forum(instance, manifest)
self.assertEqual(cleaned["status"], "deleted")
self.assertNotIn("inviteLink", driver.read_json(manifest))
self.assertEqual(
[request["@type"] for request in instance.client.requests],
[
"getMe",
"createNewBasicGroupChat",
"upgradeBasicGroupChatToSupergroupChat",
"setChatMemberStatus",
"toggleSupergroupIsForum",
"createForumTopic",
"createChatInviteLink",
"getMe",
"getChat",
"getChatMember",
"deleteChat",
],
)
def test_private_forum_cleanup_deletes_an_interrupted_basic_group(self):
instance = self.fixture()
instance.config["testDc"] = False
instance.client.members = {123}
request = instance.client.request
def request_as_creator(payload, timeout=20):
if payload["@type"] == "getChatMember":
instance.client.requests.append(payload)
return {"status": {"@type": "chatMemberStatusCreator"}}
return request(payload, timeout)
instance.client.request = request_as_creator
with tempfile.TemporaryDirectory() as root:
manifest = Path(root) / "private-forum.json"
driver.write_json_private(
manifest,
{
"basicGroupId": "-2042",
"status": "basic-group-created",
"sutBotId": "42",
"testerUserId": "123",
},
)
cleaned = driver.cleanup_private_forum(instance, manifest)
self.assertEqual(cleaned["status"], "deleted")
self.assertEqual(
[request["@type"] for request in instance.client.requests],
["getMe", "getChat", "getChatMember", "deleteChat"],
)
def test_existing_group_membership_is_repaired_and_only_added_bot_leaves(self):
instance = self.fixture()
instance.client.members = {123, 777}
@ -762,6 +919,45 @@ class RichObservationTest(unittest.TestCase):
self.assertEqual(known[(-2002, 42 << 20)]["senderId"], 202)
class ServeTargetTest(unittest.TestCase):
def test_serves_dm_group_and_forum_without_losing_observed_topic(self):
client = observation_client()
client.request.side_effect = None
client.request.return_value = {"id": 303}
client.next_update = lambda timeout: None
def send(chat_id, text, reply_to=None, forum_topic_id=None):
message = native_message({"@type": "messageText", "text": {"text": text, "entities": []}}, chat_id=chat_id, sender_id=303)
if forum_topic_id is not None:
message["topic_id"] = {"@type": "messageTopicForum", "forum_topic_id": forum_topic_id}
return message
instance = SimpleNamespace(client=client, authorize=lambda *_: None,
resolve_chat=lambda value: int(value), check_group_write_access=Mock(return_value=True),
send_text=Mock(side_effect=send))
commands = [
{"id": "1", "method": "send", "text": "dm", "chatId": "200"},
{"id": "2", "method": "send", "text": "group"},
{"id": "3", "method": "send", "text": "forum", "chatId": "-2002", "forumTopicId": 42},
{"id": "4", "method": "cleanup-private-forum"},
{"id": "5", "method": "send", "text": "invalid", "forumTopicId": True},
]
events = []
with patch.object(driver, "load_config", return_value=({}, {})), \
patch.object(driver, "UserDriver", return_value=instance), \
patch.object(driver, "cleanup_private_forum", return_value={"ok": True, "status": "deleted"}) as cleanup, \
patch.object(driver, "write_ndjson", side_effect=events.append), \
patch.object(driver.sys, "stdin", io.StringIO("\n".join(driver.json.dumps(value) for value in commands))), \
patch.object(driver.select, "select", side_effect=lambda *_: ([driver.sys.stdin], [], [])):
driver.command_serve(SimpleNamespace(chat="-1001", observe_chat=["-2002"], timeout_ms=1000))
sent = [event["result"] for event in events if "result" in event and "chatId" in event["result"]]
self.assertEqual([value["chatId"] for value in sent], [200, -1001, -2002])
self.assertEqual([value["senderId"] for value in sent], [303, 303, 303])
self.assertEqual(sent[-1]["forumTopicId"], 42)
self.assertIn("positive integer", events[-1]["error"])
self.assertEqual(instance.send_text.call_count, 3)
cleanup.assert_called_once_with(instance, driver.STATE_DIR / "owned-private-forum.json")
self.assertEqual([call.args[0] for call in instance.check_group_write_access.call_args_list], [-1001, -2002, 200])
class GroupWriteAccessTest(unittest.TestCase):
def make_driver(self, status, default=True, boosts=None, kind="chatTypeSupergroup", active=True):
instance = driver.UserDriver.__new__(driver.UserDriver)

View file

@ -953,7 +953,6 @@ extensions/qa-lab/src/live-transports/slack/slack-live.config.ts 2
extensions/qa-lab/src/live-transports/slack/slack-live.message-observations.ts 1
extensions/qa-lab/src/live-transports/slack/slack-live.observations.ts 7
extensions/qa-lab/src/live-transports/slack/slack-live.scenario-implementations.ts 2
extensions/qa-lab/src/live-transports/telegram/adapter.runtime.ts 1
extensions/qa-lab/src/live-transports/telegram/telegram-api.runtime.ts 1
extensions/qa-lab/src/live-transports/whatsapp/adapter.runtime.ts 2
extensions/qa-lab/src/live-transports/whatsapp/scenario-environment.ts 3

View file

@ -14,11 +14,15 @@ const mocks = vi.hoisted(() => ({
loadTelegramUserbotSkillRuntime: vi.fn(),
proxyClose: vi.fn(),
proxyDrainUpdates: vi.fn(),
readTelegramPrivateProductionDescriptor: vi.fn(),
requestTelegramPrivateAppTurn: vi.fn(),
resolveTelegramPrivateProductionBot: vi.fn(),
restoreCredential: vi.fn(),
shouldRetainQaGatewayCredentialLease: vi.fn(),
startApiProxy: vi.fn(),
userbotAssertHealthy: vi.fn(),
userbotClose: vi.fn(),
userbotCleanupPrivateForum: vi.fn(),
userbotSend: vi.fn(),
userbotStart: vi.fn(),
}));
@ -41,10 +45,17 @@ vi.mock("./userbot-driver.runtime.js", () => ({
TelegramUserbotDriver: { start: mocks.userbotStart },
}));
vi.mock("./private-production.runtime.js", () => ({
readTelegramPrivateProductionDescriptor: mocks.readTelegramPrivateProductionDescriptor,
requestTelegramPrivateAppTurn: mocks.requestTelegramPrivateAppTurn,
resolveTelegramPrivateProductionBot: mocks.resolveTelegramPrivateProductionBot,
}));
vi.mock("./userbot-skill.runtime.js", () => ({
loadTelegramUserbotSkillRuntime: mocks.loadTelegramUserbotSkillRuntime,
}));
import { readQaScenarioById } from "../../scenario-catalog.js";
import { createTelegramQaTransportAdapter } from "./adapter.runtime.js";
const credential = {
@ -60,12 +71,28 @@ const credential = {
tdlibVersion: "1.8.67",
} as const;
async function prepareMessageReader(
const identityCredential = {
...credential,
forumGroupId: "-100456",
forumTopicId: 42,
participants: [
{
alias: "second",
testerUserId: "101",
tdlibArchiveBase64: "Yg==",
tdlibArchiveSha256: "b".repeat(64),
tdlibVersion: "1.8.67",
},
],
};
async function prepareFlow(
adapter: Awaited<ReturnType<typeof createTelegramQaTransportAdapter>>,
config: Record<string, unknown> = {},
) {
const stateRoot = mocks.createStateRoot.mock.results.at(-1)?.value;
const prepared = await adapter.prepareFlow?.({
config: {},
return await adapter.prepareFlow?.({
config,
scenarioId: "telegram-entities",
scenarioTitle: "Telegram native entities",
gateway: {
@ -79,6 +106,12 @@ async function prepareMessageReader(
outputDir: stateRoot,
timeoutMs: 30_000,
});
}
async function prepareMessageReader(
adapter: Awaited<ReturnType<typeof createTelegramQaTransportAdapter>>,
) {
const prepared = await prepareFlow(adapter);
const read = prepared?.readTelegramMessages;
if (typeof read !== "function") {
throw new Error("Telegram flow did not expose native message observations");
@ -91,6 +124,7 @@ describe("Telegram QA transport adapter", () => {
vi.clearAllMocks();
const stateRoot = fs.mkdtempSync(path.join(os.tmpdir(), "telegram-adapter-test-"));
mocks.createStateRoot.mockReturnValue(stateRoot);
mocks.readTelegramPrivateProductionDescriptor.mockReturnValue(undefined);
mocks.acquireQaCredentialLease.mockResolvedValue({
payload: credential,
source: "convex",
@ -119,6 +153,7 @@ describe("Telegram QA transport adapter", () => {
assertHealthy: mocks.userbotAssertHealthy,
chatId: -100123,
close: mocks.userbotClose,
cleanupPrivateForum: mocks.userbotCleanupPrivateForum,
send: mocks.userbotSend,
});
mocks.proxyDrainUpdates.mockResolvedValue(undefined);
@ -133,10 +168,11 @@ describe("Telegram QA transport adapter", () => {
assertHealthy: mocks.userbotAssertHealthy,
chatId: 200,
close: mocks.userbotClose,
cleanupPrivateForum: mocks.userbotCleanupPrivateForum,
send: mocks.userbotSend,
};
});
mocks.userbotSend.mockResolvedValueOnce({ messageId: 10 });
mocks.userbotSend.mockResolvedValueOnce({ messageId: 10, senderId: 100, chatId: 200 });
const addInboundMessage = vi.fn().mockResolvedValue({ id: "in-1" });
const addOutboundMessage = vi.fn().mockResolvedValue({ id: "out-1" });
const adapter = await createTelegramQaTransportAdapter({
@ -277,9 +313,9 @@ describe("Telegram QA transport adapter", () => {
mocks.userbotSend
.mockImplementationOnce(async () => {
await onUpdate?.(preview);
return { messageId: 10 };
return { messageId: 10, senderId: 100, chatId: -100123 };
})
.mockResolvedValueOnce({ messageId: 12 });
.mockResolvedValueOnce({ messageId: 12, senderId: 100, chatId: -100123 });
const addInboundMessage = vi.fn().mockResolvedValue({ id: "in-1" });
const addOutboundMessage = vi.fn().mockResolvedValue({ id: "out-1" });
const editMessage = vi.fn();
@ -295,6 +331,7 @@ describe("Telegram QA transport adapter", () => {
});
expect(mocks.userbotSend).toHaveBeenCalledWith({
text: "@sut_bot reply exactly: QA-MARKER",
chatId: "-100123",
replyToMessageId: undefined,
});
expect(addInboundMessage).toHaveBeenCalledWith(
@ -321,6 +358,7 @@ describe("Telegram QA transport adapter", () => {
});
expect(mocks.userbotSend).toHaveBeenLastCalledWith({
text: "follow-up",
chatId: "-100123",
replyToMessageId: 11,
});
const edited = {
@ -399,6 +437,307 @@ describe("Telegram QA transport adapter", () => {
}
});
it("keeps DM, group, forum, and mixed leased participants distinct", async () => {
mocks.acquireQaCredentialLease.mockResolvedValueOnce({
payload: identityCredential,
heartbeat: mocks.leaseHeartbeat,
release: mocks.leaseRelease,
});
const updates: Array<(update: unknown) => Promise<void>> = [];
const sends = [100, 101].map((senderId) =>
vi.fn(async (input) => ({
messageId: 10,
senderId,
chatId: Number(input.chatId),
forumTopicId: input.forumTopicId,
})),
);
mocks.userbotStart.mockImplementation(async (params) => {
const index = updates.length;
updates.push(params.onUpdate);
return {
assertHealthy: mocks.userbotAssertHealthy,
chatId: -100123,
close: mocks.userbotClose,
send: sends[index],
};
});
let nextId = 0;
const addInboundMessage = vi.fn().mockImplementation(async () => ({ id: `in-${++nextId}` }));
const addOutboundMessage = vi.fn().mockImplementation(async () => ({ id: `out-${++nextId}` }));
const editMessage = vi.fn();
const adapter = await createTelegramQaTransportAdapter({
adapterOptions: {},
messages: { addInboundMessage, addOutboundMessage, editMessage },
} as never);
try {
const scenario = readQaScenarioById("telegram-participant-identity-inspection");
await expect(prepareFlow(adapter, scenario.execution.config)).resolves.toMatchObject({
telegramIdentityFixture: { participantAliases: ["primary", "second"], forumTopicId: 42 },
});
expect(mocks.userbotStart.mock.calls.map(([input]) => input.expectedUserId)).toEqual([
"100",
"101",
]);
expect(mocks.userbotStart).toHaveBeenNthCalledWith(
1,
expect.objectContaining({ observeChatIds: ["-100456"] }),
);
expect(adapter.createGatewayConfig({ baseUrl: "http://127.0.0.1:1234" })).toMatchObject({
channels: {
telegram: {
accounts: {
sut: {
allowFrom: ["100", "101"],
groups: { "-100456": { allowFrom: ["100", "101"] } },
},
},
},
},
});
await adapter.sendInbound({
conversation: { id: "dm", kind: "direct" },
senderId: "primary",
text: "dm",
});
await adapter.sendInbound({
conversation: { id: "room", kind: "group" },
senderId: "primary",
text: "group",
});
await adapter.sendInbound({
conversation: { id: "room", kind: "group" },
senderId: "second",
text: "mixed",
});
await adapter.sendInbound({
conversation: { id: "forum", kind: "group" },
threadId: "42",
senderId: "second",
text: "forum",
});
expect(sends.map((send) => send.mock.calls.map(([input]) => input.chatId))).toEqual([
["200", "-100123"],
["-100123", "-100456"],
]);
expect(sends[1]).toHaveBeenLastCalledWith({
chatId: "-100456",
forumTopicId: 42,
text: "forum",
replyToMessageId: undefined,
});
expect(addInboundMessage.mock.calls.map(([input]) => input.senderId)).toEqual([
"100",
"100",
"101",
"101",
]);
const update = {
kind: "message",
messageId: 11,
senderId: 200,
text: "reply",
timestamp: 1000,
entities: [],
};
// Deliver a late DM reply after two other native conversations have sent.
const [primaryUpdate, secondUpdate] = updates;
if (!primaryUpdate || !secondUpdate) {
throw new Error("Expected both leased participant observers to start.");
}
await primaryUpdate({ ...update, chatId: 200 });
await primaryUpdate({ ...update, chatId: -100123 });
await secondUpdate({ ...update, chatId: -100123 });
await primaryUpdate({ ...update, chatId: -100456, forumTopicId: 42 });
await primaryUpdate({ ...update, chatId: -100456, forumTopicId: 43 });
expect(addOutboundMessage.mock.calls.map(([input]) => input.to)).toEqual([
"dm:dm",
"group:room",
"thread:/v1/group/forum/42",
]);
await primaryUpdate({ ...update, kind: "edit", chatId: 200, text: "dm edit" });
expect(editMessage).toHaveBeenCalledWith(
expect.objectContaining({ messageId: "out-5", text: "dm edit" }),
);
await expect(
adapter.sendInbound({
conversation: { id: "other-room", kind: "group" },
senderId: "primary",
text: "remap",
}),
).rejects.toThrow("another logical conversation");
await expect(
adapter.sendInbound({
conversation: { id: "room", kind: "group" },
senderId: "missing",
text: "impersonate",
}),
).rejects.toThrow("named leased participant");
await expect(
adapter.sendInbound({
conversation: { id: "room", kind: "group" },
senderId: "second",
replyToId: "out-6",
text: "wrong observer",
}),
).rejects.toThrow("observed by this participant");
expect(adapter.buildAgentDelivery({ target: "group:forum", threadId: "42" })).toEqual({
channel: "telegram",
to: "-100456",
replyChannel: "telegram",
replyTo: "-100456",
threadId: "42",
});
} finally {
await adapter.cleanup?.();
await adapter.cleanupAfterGatewayStop?.();
}
expect(mocks.userbotClose).toHaveBeenCalledTimes(2);
expect(mocks.leaseRelease).toHaveBeenCalledOnce();
});
it("uses local Telegram.app participants without acquiring a shared credential", async () => {
const descriptor = {
file: "/private/descriptor.json",
mode: "private-production-local-apps",
forumGroupId: "-100456",
forumTopicId: 42,
topicTitle: "Private identity proof",
participants: [
{ alias: "primary", host: "mainframe", userId: "100" },
{ alias: "second", host: "macbook", userId: "101" },
],
};
mocks.readTelegramPrivateProductionDescriptor.mockReturnValueOnce(descriptor);
mocks.resolveTelegramPrivateProductionBot.mockResolvedValue({
id: "200",
token: "private-token",
username: "qa_bot",
});
mocks.requestTelegramPrivateAppTurn.mockResolvedValue({ replyText: "synthetic-marker" });
const addInboundMessage = vi.fn().mockResolvedValue({ id: "in-hitl" });
const addOutboundMessage = vi.fn().mockResolvedValue({ id: "out-hitl" });
const adapter = await createTelegramQaTransportAdapter({
adapterOptions: { credentialFile: "/private/descriptor.json" },
messages: { addInboundMessage, addOutboundMessage },
} as never);
try {
expect(mocks.acquireQaCredentialLease).not.toHaveBeenCalled();
expect(mocks.startApiProxy).not.toHaveBeenCalled();
expect(adapter.createGatewayConfig({ baseUrl: "http://127.0.0.1:1234" })).toMatchObject({
channels: {
telegram: {
accounts: {
sut: {
allowFrom: ["100", "101"],
groups: { "-100456": { allowFrom: ["100", "101"] } },
},
},
},
},
});
expect(
adapter.createGatewayConfig({ baseUrl: "http://127.0.0.1:1234" }).channels?.telegram
?.accounts?.sut,
).not.toHaveProperty("apiRoot");
await adapter.sendInbound({
conversation: { id: "forum", kind: "group" },
senderId: "second",
threadId: "42",
text: "@openclaw Reply exactly: synthetic-marker",
});
expect(mocks.requestTelegramPrivateAppTurn).toHaveBeenCalledWith({
descriptor,
destination: "forum-topic",
participant: descriptor.participants[1],
text: "@qa_bot Reply exactly: synthetic-marker",
});
expect(addInboundMessage).toHaveBeenCalledWith(expect.objectContaining({ senderId: "101" }));
expect(addOutboundMessage).toHaveBeenCalledWith(
expect.objectContaining({
replyToId: "in-hitl",
text: "synthetic-marker",
to: "thread:/v1/group/forum/42",
}),
);
expect(mocks.userbotSend).not.toHaveBeenCalled();
} finally {
await adapter.cleanup?.();
await adapter.cleanupAfterGatewayStop?.();
}
expect(mocks.userbotStart).not.toHaveBeenCalled();
expect(mocks.leaseRelease).not.toHaveBeenCalled();
});
it.each(["participants", "forumGroupId", "forumTopicId"] as const)(
"fails the cataloged identity scenario clearly when the lease lacks %s",
async (missing) => {
mocks.acquireQaCredentialLease.mockResolvedValueOnce({
payload: { ...identityCredential, [missing]: undefined },
heartbeat: mocks.leaseHeartbeat,
release: mocks.leaseRelease,
});
const adapter = await createTelegramQaTransportAdapter({
adapterOptions: {},
messages: {},
} as never);
try {
const scenario = readQaScenarioById("telegram-participant-identity-inspection");
await expect(prepareFlow(adapter, scenario.execution.config)).rejects.toThrow(
"requires distinct participants, forumGroupId, and forumTopicId",
);
expect(mocks.userbotSend).not.toHaveBeenCalled();
} finally {
await adapter.cleanup?.();
await adapter.cleanupAfterGatewayStop?.();
}
expect(mocks.leaseRelease).toHaveBeenCalledOnce();
},
);
it("rejects an unconfirmed participant receipt without recording synthetic identity", async () => {
mocks.userbotSend.mockResolvedValueOnce({ messageId: 10, senderId: 999, chatId: -100123 });
const addInboundMessage = vi.fn();
const adapter = await createTelegramQaTransportAdapter({
adapterOptions: {},
messages: { addInboundMessage },
} as never);
try {
await expect(
adapter.sendInbound({
conversation: { id: "room", kind: "group" },
senderId: "primary",
text: "identity",
}),
).rejects.toThrow("send receipt");
expect(addInboundMessage).not.toHaveBeenCalled();
} finally {
await adapter.cleanup?.();
await adapter.cleanupAfterGatewayStop?.();
}
});
it("retains participant recovery state and the lease when observer shutdown is unconfirmed", async () => {
const adapter = await createTelegramQaTransportAdapter({
adapterOptions: {},
messages: {},
} as never);
const root = mocks.createStateRoot.mock.results.at(-1)?.value;
mocks.userbotClose.mockRejectedValueOnce(new Error("exit unconfirmed"));
try {
await expect(adapter.cleanup?.()).rejects.toThrow("retained private state");
await expect(adapter.cleanupAfterGatewayStop?.()).rejects.toThrow(
"retained Telegram credential",
);
expect(fs.existsSync(root)).toBe(true);
expect(mocks.leaseRelease).not.toHaveBeenCalled();
expect(mocks.heartbeatStop).toHaveBeenCalledOnce();
} finally {
fs.rmSync(root, { recursive: true, force: true });
}
});
it("releases the lease when userbot startup fails", async () => {
const stateRoot = fs.mkdtempSync(path.join(os.tmpdir(), "telegram-adapter-failure-test-"));
mocks.createStateRoot.mockReturnValueOnce(stateRoot);

View file

@ -9,6 +9,11 @@ import {
acquireQaCredentialLease,
startQaCredentialLeaseHeartbeat,
} from "../shared/credential-lease.runtime.js";
import {
readTelegramPrivateProductionDescriptor,
requestTelegramPrivateAppTurn,
resolveTelegramPrivateProductionBot,
} from "./private-production.runtime.js";
import { buildTelegramQaConfig, waitForTelegramChannelRunning } from "./telegram-api.runtime.js";
import { TelegramUserbotDriver, type TelegramUserbotUpdate } from "./userbot-driver.runtime.js";
import {
@ -19,6 +24,28 @@ import {
type AdapterFactory = NonNullable<QaRunnerCliRegistration["adapterFactory"]>;
type FactoryContext = Parameters<AdapterFactory["create"]>[0];
type AdapterDefinition = Awaited<ReturnType<AdapterFactory["create"]>>;
type TelegramUserbotSkillRuntime = Awaited<ReturnType<typeof loadTelegramUserbotSkillRuntime>>;
type TelegramRuntimeCredential = Pick<
TelegramTestCredential,
| "forumGroupId"
| "forumTopicId"
| "groupId"
| "participants"
| "sutBotId"
| "sutToken"
| "sutUsername"
| "testerUserId"
> & {
environment: "production" | "test";
};
type TelegramRuntimeParticipant = {
alias: string;
credential?: TelegramRuntimeCredential;
mode: "hitl" | "userbot";
testerUserId: string;
};
const TELEGRAM_QA_DIAGNOSTIC_COUNT_LIMIT = 9_999;
@ -73,38 +100,83 @@ export async function createTelegramQaTransportAdapter(
context: FactoryContext,
): Promise<AdapterDefinition> {
const options = context.adapterOptions ?? {};
const skillRuntime = await loadTelegramUserbotSkillRuntime({ repoRoot: options.repoRoot });
const credentialLease = await acquireQaCredentialLease<TelegramTestCredential>({
kind: "telegram-test-userbot",
source: options.credentialSource || "convex",
role: options.credentialRole,
resolveEnvPayload: () => {
throw new Error("Telegram live QA requires a Convex-leased Test Server userbot.");
},
parsePayload: (payload) => skillRuntime.parseCredential(payload),
});
try {
assertQaGatewayCredentialLeaseQuarantine(credentialLease);
} catch (error) {
await credentialLease.release();
throw error;
}
const heartbeat = startQaCredentialLeaseHeartbeat(credentialLease);
const { buildQaTarget, parseQaTarget } = await import("openclaw/plugin-sdk/qa-channel-protocol");
const privateDescriptor = readTelegramPrivateProductionDescriptor(options.credentialFile);
const privateProduction = privateDescriptor !== undefined;
const skillRuntime = privateProduction
? undefined
: await loadTelegramUserbotSkillRuntime({ repoRoot: options.repoRoot });
const leasedRuntime = privateProduction
? undefined
: await (async () => {
const credentialLease = await acquireQaCredentialLease<TelegramRuntimeCredential>({
kind: "telegram-test-userbot",
source: options.credentialSource || "convex",
role: options.credentialRole,
resolveEnvPayload: () => {
throw new Error("Telegram live QA requires a Convex-leased userbot.");
},
parsePayload: (payload) => skillRuntime!.parseCredential(payload),
});
try {
assertQaGatewayCredentialLeaseQuarantine(credentialLease);
} catch (error) {
await credentialLease.release();
throw error;
}
return {
credential: credentialLease.payload,
credentialLease,
heartbeat: startQaCredentialLeaseHeartbeat(credentialLease),
};
})();
const privateBot = privateProduction
? await resolveTelegramPrivateProductionBot(process.env)
: undefined;
const credential: TelegramRuntimeCredential = privateDescriptor
? {
environment: "production",
forumGroupId: privateDescriptor.forumGroupId,
forumTopicId: privateDescriptor.forumTopicId,
groupId: privateDescriptor.forumGroupId,
participants: undefined,
sutBotId: privateBot!.id,
sutToken: privateBot!.token,
sutUsername: privateBot!.username,
testerUserId: privateDescriptor.participants[0]!.userId,
}
: leasedRuntime!.credential;
const leaseHealth = {
assertHealthy: () => heartbeat.throwIfFailed(),
whenUnhealthy: heartbeat.whenFailed,
assertHealthy: () => leasedRuntime?.heartbeat.throwIfFailed(),
whenUnhealthy: leasedRuntime?.heartbeat.whenFailed ?? new Promise<Error>(() => {}),
};
let leaseReleased = false;
const releaseCredentialLease = async () => {
if (leaseReleased) {
if (leaseReleased || !leasedRuntime) {
return;
}
await releaseTelegramCredential({ heartbeat, release: () => credentialLease.release() });
await releaseTelegramCredential({
heartbeat: leasedRuntime.heartbeat,
release: () => leasedRuntime.credentialLease.release(),
});
leaseReleased = true;
};
let stateRoot: string | undefined;
let apiProxy: Awaited<ReturnType<typeof skillRuntime.startApiProxy>> | undefined;
let apiProxy: Awaited<ReturnType<TelegramUserbotSkillRuntime["startApiProxy"]>> | undefined;
let userbot: TelegramUserbotDriver | undefined;
let participants: TelegramRuntimeParticipant[] = [];
const drivers: TelegramUserbotDriver[] = [];
const participantRoots: string[] = [];
let primaryAlias: string | undefined;
let participantCleanupUncertain = false;
const closeParticipants = async () => {
const results = await Promise.allSettled(drivers.map((driver) => driver.close()));
const errors = results.flatMap((result) =>
result.status === "rejected" ? [result.reason] : [],
);
participantCleanupUncertain ||= errors.length > 0;
return errors;
};
const observerState: TelegramQaObserverState = {
filteredCount: 0,
matchedCount: 0,
@ -113,19 +185,37 @@ export async function createTelegramQaTransportAdapter(
};
const accountId = options.sutAccountId?.trim() || "sut";
const directMessageOnly = options.transportPolicy?.directMessageOnly === true;
const agentDeliveryTarget = directMessageOnly
? credentialLease.payload.testerUserId
: credentialLease.payload.groupId;
let nativeChatId = Number(credentialLease.payload.groupId);
let logicalConversationId = credentialLease.payload.groupId;
let logicalConversationKind: "channel" | "direct" | "group" = "channel";
const nativeMessageIds = new Map<string, number>();
const busMessages = new Map<number, { id: string; update?: TelegramUserbotUpdate }>();
type Route = {
id: string;
kind: "channel" | "direct" | "group";
threadId?: string;
bound?: true;
};
const routes = new Map<string, Route>();
const nativeKey = (observer: number, chatId: number, messageId: number) =>
`${observer}:${chatId}:${messageId}`;
const routeKey = (observer: number, chatId: number, topic?: number) =>
`${observer}:${chatId}:${topic ?? ""}`;
const resetRoutes = () => {
const initialChatId = Number(directMessageOnly ? credential.sutBotId : credential.groupId);
routes.clear();
routes.set(routeKey(0, initialChatId), {
id: credential.groupId,
kind: directMessageOnly ? "direct" : "channel",
});
};
const nativeMessageIds = new Map<
string,
{ observer: number; chatId: number; messageId: number }
>();
const busMessages = new Map<string, { id: string; update?: TelegramUserbotUpdate }>();
let sendsInFlight = 0;
let deferredReplies: TelegramUserbotUpdate[] = [];
let deferredReplies: Array<{ update: TelegramUserbotUpdate; observer: number }> = [];
let localMessageId = 1;
const publishUpdate = async (update: TelegramUserbotUpdate) => {
const existing = busMessages.get(update.messageId);
const publishUpdate = async (update: TelegramUserbotUpdate, observer: number) => {
const key = nativeKey(observer, update.chatId, update.messageId);
const existing = busMessages.get(key);
if (update.kind === "edit" && existing) {
await context.messages.editMessage({
accountId,
@ -136,67 +226,145 @@ export async function createTelegramQaTransportAdapter(
existing.update = update;
return;
}
const route = routes.get(routeKey(observer, update.chatId, update.forumTopicId));
if (!route) {
return;
}
const outbound = await context.messages.addOutboundMessage({
accountId,
to: `${logicalConversationKind === "direct" ? "dm" : logicalConversationKind}:${logicalConversationId}`,
to: buildQaTarget({
chatType: route.kind,
conversationId: route.id,
threadId: route.threadId,
}),
senderId: String(update.senderId),
senderName: update.senderUsername,
text: update.text,
timestamp: update.timestamp,
replyToId: update.replyToMessageId ? busMessages.get(update.replyToMessageId)?.id : undefined,
replyToId: update.replyToMessageId
? busMessages.get(nativeKey(observer, update.chatId, update.replyToMessageId))?.id
: undefined,
});
nativeMessageIds.set(outbound.id, update.messageId);
busMessages.set(update.messageId, { id: outbound.id, update });
nativeMessageIds.set(outbound.id, {
observer,
chatId: update.chatId,
messageId: update.messageId,
});
busMessages.set(key, { id: outbound.id, update });
};
const observeUpdate = async (update: TelegramUserbotUpdate) => {
const observeUpdate = async (update: TelegramUserbotUpdate, observer: number) => {
observerState.updateCount += 1;
observerState.relevantUpdateKinds.add(update.kind);
if (
update.chatId !== nativeChatId ||
update.senderId !== Number(credentialLease.payload.sutBotId)
!routes.has(routeKey(observer, update.chatId, update.forumTopicId)) ||
update.senderId !== Number(credential.sutBotId)
) {
observerState.filteredCount += 1;
return;
}
observerState.matchedCount += 1;
if (sendsInFlight > 0 && update.replyToMessageId && !busMessages.has(update.replyToMessageId)) {
deferredReplies.push(update);
if (
sendsInFlight > 0 &&
update.replyToMessageId &&
!busMessages.has(nativeKey(observer, update.chatId, update.replyToMessageId))
) {
deferredReplies.push({ update, observer });
return;
}
await publishUpdate(update);
await publishUpdate(update, observer);
};
try {
stateRoot = skillRuntime.createStateRoot();
const restored = skillRuntime.restoreCredential(credentialLease.payload, stateRoot);
apiProxy = await skillRuntime.startApiProxy(leaseHealth);
await apiProxy.drainUpdates(restored.sutToken);
userbot = await TelegramUserbotDriver.start({
chatId: directMessageOnly ? `@${credentialLease.payload.sutUsername}` : restored.groupId,
driverEnv: restored.driverEnv,
leaseHealth,
userDriverPath: skillRuntime.userDriverPath,
onUpdate: observeUpdate,
});
nativeChatId = userbot.chatId;
participants = privateProduction
? privateDescriptor!.participants.map((participant) => ({
alias: participant.alias,
mode: "hitl" as const,
testerUserId: participant.userId,
}))
: [
{
alias: "primary",
credential,
mode: "userbot",
testerUserId: credential.testerUserId,
},
...(credential.participants ?? []).map((participant) => ({
alias: participant.alias,
credential: { ...credential, ...participant, participants: undefined },
mode: "userbot" as const,
testerUserId: participant.testerUserId,
})),
];
resetRoutes();
if (!privateProduction) {
stateRoot = skillRuntime!.createStateRoot();
const restored = skillRuntime!.restoreCredential(
// SAFETY: The leased credential passed the Telegram test-credential parser above.
credential as TelegramTestCredential,
stateRoot,
);
apiProxy = await skillRuntime!.startApiProxy(leaseHealth);
await apiProxy.drainUpdates(credential.sutToken);
userbot = await TelegramUserbotDriver.start({
chatId: directMessageOnly ? `@${credential.sutUsername}` : restored.groupId,
...(credential.forumGroupId ? { observeChatIds: [credential.forumGroupId] } : {}),
expectedUserId: credential.testerUserId,
driverEnv: restored.driverEnv,
leaseHealth,
userDriverPath: skillRuntime!.userDriverPath,
onUpdate: (update) => observeUpdate(update, 0),
});
drivers.push(userbot);
for (const [offset, participant] of participants.slice(1).entries()) {
if (!participant.credential) {
continue;
}
const root = skillRuntime!.createStateRoot();
participantRoots.push(root);
const restoredParticipant = skillRuntime!.restoreCredential(
// SAFETY: Userbot participants are derived from the parsed Telegram test credential.
participant.credential as TelegramTestCredential,
root,
);
drivers.push(
await TelegramUserbotDriver.start({
chatId: directMessageOnly ? `@${credential.sutUsername}` : restored.groupId,
expectedUserId: participant.testerUserId,
driverEnv: restoredParticipant.driverEnv,
leaseHealth,
userDriverPath: skillRuntime!.userDriverPath,
onUpdate: (update) => observeUpdate(update, offset + 1),
}),
);
}
}
} catch (error) {
const cleanupErrors: unknown[] = [];
try {
await userbot?.close();
} catch (cleanupError) {
cleanupErrors.push(cleanupError);
}
cleanupErrors.push(...(await closeParticipants()));
try {
await apiProxy?.close();
} catch (cleanupError) {
cleanupErrors.push(cleanupError);
}
if (stateRoot) {
fs.rmSync(stateRoot, { recursive: true, force: true });
if (!participantCleanupUncertain) {
for (const root of participantRoots) {
fs.rmSync(root, { recursive: true, force: true });
}
if (stateRoot) {
fs.rmSync(stateRoot, { recursive: true, force: true });
}
}
try {
await releaseCredentialLease();
if (participantCleanupUncertain && leasedRuntime) {
try {
await leasedRuntime.credentialLease.heartbeat();
} finally {
await leasedRuntime.heartbeat.stop();
}
} else {
await releaseCredentialLease();
}
} catch (cleanupError) {
cleanupErrors.push(cleanupError);
}
@ -208,10 +376,9 @@ export async function createTelegramQaTransportAdapter(
throw error;
}
if (!userbot || !apiProxy || !stateRoot) {
if (!privateProduction && (!userbot || !apiProxy || !stateRoot)) {
throw new Error("Telegram userbot runtime did not start.");
}
const activeUserbot = userbot;
const activeApiProxy = apiProxy;
const activeStateRoot = stateRoot;
let observerStopped = false;
@ -223,32 +390,166 @@ export async function createTelegramQaTransportAdapter(
requiredPluginIds: ["telegram"],
supportedActions: [],
assertTransportHealthy() {
activeUserbot.assertHealthy();
heartbeat.throwIfFailed();
for (const driver of drivers) {
driver.assertHealthy();
}
leasedRuntime?.heartbeat.throwIfFailed();
},
describeTransportState: () => describeTelegramQaObserverState(observerState),
async sendInbound(input) {
heartbeat.throwIfFailed();
logicalConversationId = input.conversation.id;
logicalConversationKind = input.conversation.kind;
const text = renderTelegramQaInboundText(input, credentialLease.payload.sutUsername);
const nativeReplyToId = input.replyToId ? nativeMessageIds.get(input.replyToId) : undefined;
leasedRuntime?.heartbeat.throwIfFailed();
let observer = participants.findIndex(
(participant) =>
participant.alias === input.senderId || participant.testerUserId === input.senderId,
);
if (observer < 0) {
if (participants.length > 1) {
throw new Error(
"Telegram QA sender requires primary or a named leased participant alias.",
);
}
primaryAlias ??= input.senderId;
if (!input.senderId || input.senderId !== primaryAlias) {
throw new Error(
"Telegram QA sender requires a named leased participant; labels cannot impersonate another user.",
);
}
observer = 0;
}
const participant = participants[observer];
if (!participant) {
throw new Error("Telegram QA participant is unavailable.");
}
if (directMessageOnly && input.conversation.kind !== "direct") {
throw new Error("Telegram QA direct-message-only policy rejects group sends.");
}
const forumTopicId = input.threadId === undefined ? undefined : Number(input.threadId);
if (
forumTopicId !== undefined &&
(!Number.isSafeInteger(forumTopicId) ||
forumTopicId <= 0 ||
input.conversation.kind === "direct")
) {
throw new Error(
"Telegram QA forum sends require a positive numeric topic and a group conversation.",
);
}
const chatId =
input.conversation.kind === "direct"
? Number(credential.sutBotId)
: Number(
forumTopicId ? (credential.forumGroupId ?? credential.groupId) : credential.groupId,
);
// Shared rooms use one observer. TDLib IDs and local-app attestations share that route.
const routeObserver = input.conversation.kind === "direct" ? observer : 0;
const routeId = routeKey(routeObserver, chatId, forumTopicId);
const previous = routes.get(routeId);
if (
previous?.bound &&
(previous.id !== input.conversation.id || previous.kind !== input.conversation.kind)
) {
throw new Error(
"Telegram QA chat/topic already belongs to another logical conversation; reset transport before reusing it.",
);
}
routes.set(routeId, { ...input.conversation, threadId: input.threadId, bound: true });
const reply = input.replyToId ? nativeMessageIds.get(input.replyToId) : undefined;
if (input.replyToId && (!reply || reply.observer !== observer || reply.chatId !== chatId)) {
throw new Error(
"Telegram QA reply requires a message observed by this participant in this chat.",
);
}
const text = renderTelegramQaInboundText(input, credential.sutUsername);
sendsInFlight += 1;
try {
const sent = await activeUserbot.send({ text, replyToMessageId: nativeReplyToId });
let sent: TelegramUserbotUpdate;
let messageObserver = observer;
let appProofReply: string | undefined;
if (participant.mode === "hitl") {
const appParticipant = privateDescriptor?.participants[observer];
if (!privateDescriptor || !appParticipant) {
throw new Error("Telegram local-app participant is unavailable.");
}
if (
input.conversation.kind === "group" &&
(forumTopicId !== privateDescriptor.forumTopicId ||
chatId !== Number(privateDescriptor.forumGroupId))
) {
throw new Error("Telegram local-app forum send targets the wrong private topic.");
}
messageObserver = routeObserver;
const proof = await requestTelegramPrivateAppTurn({
descriptor: privateDescriptor,
destination: input.conversation.kind === "direct" ? "bot-dm" : "forum-topic",
participant: appParticipant,
text,
});
appProofReply = proof.replyText;
sent = {
kind: "message",
chatId,
...(forumTopicId === undefined ? {} : { forumTopicId }),
messageId: localMessageId++,
senderId: Number(participant.testerUserId),
timestamp: Date.now(),
text,
entities: [],
};
} else {
const driver = drivers[observer];
if (!driver) {
throw new Error("Telegram QA participant driver is unavailable.");
}
sent = await driver.send({
text,
chatId: String(chatId),
...(forumTopicId === undefined ? {} : { forumTopicId }),
replyToMessageId: reply?.messageId,
});
}
if (
String(sent.senderId) !== participant.testerUserId ||
sent.chatId !== chatId ||
(forumTopicId !== undefined && sent.forumTopicId !== forumTopicId)
) {
throw new Error(
"Telegram send receipt does not match the configured participant and requested chat/topic.",
);
}
const message = await context.messages.addInboundMessage({
...input,
accountId,
senderId: credentialLease.payload.testerUserId,
senderId: participant.testerUserId,
});
nativeMessageIds.set(message.id, sent.messageId);
busMessages.set(sent.messageId, { id: message.id });
const readyReplies = deferredReplies.filter(
(update) => update.replyToMessageId && busMessages.has(update.replyToMessageId),
);
deferredReplies = deferredReplies.filter((update) => !readyReplies.includes(update));
for (const update of readyReplies) {
await publishUpdate(update);
nativeMessageIds.set(message.id, {
observer: messageObserver,
chatId,
messageId: sent.messageId,
});
busMessages.set(nativeKey(messageObserver, chatId, sent.messageId), { id: message.id });
if (appProofReply !== undefined) {
await observeUpdate(
{
kind: "message",
chatId,
...(forumTopicId === undefined ? {} : { forumTopicId }),
messageId: localMessageId++,
replyToMessageId: sent.messageId,
senderId: Number(credential.sutBotId),
senderUsername: credential.sutUsername,
timestamp: Date.now(),
text: appProofReply,
entities: [],
},
messageObserver,
);
}
// A shared-room reply can quote an ID in another user's private TDLib sequence.
// Preserve the reply without claiming a cross-account quote relationship.
const readyReplies = deferredReplies;
deferredReplies = [];
for (const entry of readyReplies) {
await publishUpdate(entry.update, entry.observer);
}
return message;
} finally {
@ -256,8 +557,8 @@ export async function createTelegramQaTransportAdapter(
}
},
resetTransport: () => {
logicalConversationId = credentialLease.payload.groupId;
logicalConversationKind = "channel";
resetRoutes();
primaryAlias = undefined;
nativeMessageIds.clear();
busMessages.clear();
deferredReplies = [];
@ -266,11 +567,25 @@ export async function createTelegramQaTransportAdapter(
observerState.matchedCount = 0;
observerState.relevantUpdateKinds.clear();
},
async prepareFlow() {
async prepareFlow({ config }) {
if (
config.requireParticipantIdentityFixture === true &&
(participants.length < 2 || !credential.forumGroupId || !credential.forumTopicId)
) {
throw new Error(
"Telegram participant identity proof requires distinct participants, forumGroupId, and forumTopicId.",
);
}
return {
telegramIdentityFixture: {
participantAliases: participants.map((participant) => participant.alias),
forumTopicId: credential.forumTopicId,
},
readTelegramMessages: () => {
activeUserbot.assertHealthy();
heartbeat.throwIfFailed();
for (const driver of drivers) {
driver.assertHealthy();
}
leasedRuntime?.heartbeat.throwIfFailed();
// Share the existing message lifetime; readers cannot mutate a later snapshot.
return [...busMessages.values()].flatMap(({ update }) =>
update ? [structuredClone(update)] : [],
@ -279,12 +594,18 @@ export async function createTelegramQaTransportAdapter(
};
},
createGatewayConfig: () =>
// SAFETY: The builder accepts an empty base and supplies every QA-owned config section.
buildTelegramQaConfig({} as OpenClawConfig, {
apiRoot: activeApiProxy.apiRoot,
apiRoot: activeApiProxy?.apiRoot,
directMessageOnly,
groupId: credentialLease.payload.groupId,
sutToken: credentialLease.payload.sutToken,
testerUserId: credentialLease.payload.testerUserId,
enableDirectMessages: true,
additionalTesterUserIds: participants
.slice(1)
.map((participant) => participant.testerUserId),
forumGroupId: credential.forumGroupId,
groupId: credential.groupId,
sutToken: credential.sutToken,
testerUserId: credential.testerUserId,
sutAccountId: accountId,
}),
waitReady: async ({ gateway, timeoutMs, pollIntervalMs }) =>
@ -292,30 +613,51 @@ export async function createTelegramQaTransportAdapter(
timeoutMs,
pollMs: pollIntervalMs,
}),
buildAgentDelivery: () => ({
channel: "telegram",
to: agentDeliveryTarget,
replyChannel: "telegram",
replyTo: agentDeliveryTarget,
}),
buildAgentDelivery: ({ target, threadId }) => {
const parsed = parseQaTarget(target);
const topic = threadId ?? parsed?.threadId;
const to =
parsed?.chatType === "direct" || directMessageOnly
? credential.testerUserId
: topic
? (credential.forumGroupId ?? credential.groupId)
: credential.groupId;
return {
channel: "telegram",
to,
replyChannel: "telegram",
replyTo: to,
...(topic ? { threadId: topic } : {}),
};
},
async handleAction() {
throw new Error("Telegram live QA adapter does not implement transport actions");
},
createReportNotes: () => ["Runs through the Telegram Test Server userbot adapter."],
createReportNotes: () => [
privateProduction
? "Runs through two operator-local Telegram.app participants; private UI attestations back the native send and reply observations."
: "Runs through the Telegram Test Server userbot adapter.",
],
async cleanup() {
if (observerStopped) {
return;
}
observerStopped = true;
try {
await activeUserbot.close();
} finally {
fs.rmSync(activeStateRoot, { recursive: true, force: true });
const errors: unknown[] = [];
errors.push(...(await closeParticipants()));
if (errors.length) {
throw new AggregateError(
errors,
"Telegram participant cleanup is unconfirmed; retained private state and lease.",
);
}
for (const root of [...(activeStateRoot ? [activeStateRoot] : []), ...participantRoots]) {
fs.rmSync(root, { recursive: true, force: true });
}
},
async cleanupAfterGatewayStop() {
const cleanupErrors: unknown[] = [];
if (!apiProxyClosed) {
if (!apiProxyClosed && activeApiProxy) {
try {
await activeApiProxy.close();
apiProxyClosed = true;
@ -323,19 +665,22 @@ export async function createTelegramQaTransportAdapter(
cleanupErrors.push(error);
}
}
if (await shouldRetainQaGatewayCredentialLease()) {
if (
leasedRuntime &&
(participantCleanupUncertain || (await shouldRetainQaGatewayCredentialLease()))
) {
try {
await credentialLease.heartbeat();
await leasedRuntime.credentialLease.heartbeat();
} catch (error) {
cleanupErrors.push(error);
}
try {
await heartbeat.stop();
await leasedRuntime.heartbeat.stop();
} catch (error) {
cleanupErrors.push(error);
}
throw new Error(
"retained Telegram credential lease for two hours because isolated SUT quiescence was not proven",
"retained Telegram credential lease for two hours because participant or isolated SUT quiescence was not proven",
cleanupErrors.length > 0 ? { cause: new AggregateError(cleanupErrors) } : undefined,
);
}

View file

@ -179,6 +179,7 @@ export async function runQaTelegramSuite(opts: TelegramQaSuiteOptions) {
],
adapterOptions: {
repoRoot: runOptions.repoRoot,
...(runOptions.credentialFile ? { credentialFile: runOptions.credentialFile } : {}),
...(runOptions.credentialRole ? { credentialRole: runOptions.credentialRole } : {}),
...(runOptions.credentialSource ? { credentialSource: runOptions.credentialSource } : {}),
...(runOptions.sutAccountId ? { sutAccountId: runOptions.sutAccountId } : {}),

View file

@ -27,7 +27,8 @@ export const telegramQaCliRegistration: LiveTransportQaCliRegistration =
roleDescription:
"Credential role for convex auth: maintainer or ci (default: ci in CI, maintainer otherwise)",
},
description: "Run Telegram Test Server QA with a Convex-leased real-user driver",
credentialFileHelp: "Private qualification-mode descriptor (contains no credentials)",
description: "Run Telegram Test Server QA with a Convex-leased user or private production apps",
listScenariosHelp: "Print available Telegram scenario ids and exit",
outputDirHelp: "Telegram QA artifact directory",
profileHelp: "Taxonomy profile for Telegram scenario selection (default: release)",

View file

@ -0,0 +1,321 @@
import { describe, expect, it, vi } from "vitest";
import { createQaBusState } from "../../bus-state.js";
import { readQaScenarioById, readQaScenarioPack } from "../../scenario-catalog.js";
import { runLoadedScenarioFlow } from "../../scenario-flow-runner.test-support.js";
import { selectQaFlowSuiteScenarios } from "../../suite-planning.js";
import { resolveTelegramQaScenarioIds } from "./scenario-selection.js";
const scenarioId = "telegram-participant-identity-inspection";
const primaryUserId = "710000001";
const additionalUserId = "710000002";
const botId = 710000003;
const forumGroupId = -100710000004;
const forumTopicId = 42;
const hmac = (digit: string) => `hmac-sha256:v1:${"a".repeat(32)}:${digit.repeat(64)}`;
type Fault =
| "wrong-transport"
| "wrong-provider"
| "missing-fixture"
| "missing-participant"
| "duplicate-alias"
| "missing-topic"
| "wrong-topic"
| "extra-run"
| "unknown-person"
| "room-principal"
| "wrong-run"
| "raw-principal"
| "missing-assurance"
| "raw-room"
| "prompt-leak"
| "oversized-context"
| "verified-generic"
| "wrong-cli-execution"
| "missing-human"
| "same-person"
| "changed-primary"
| "reused-context"
| "restart-drift"
| "restart-leak";
function runIdentityFlow(fault?: Fault) {
const state = createQaBusState();
const admittedRuns = ["previous-run"];
const turns: Array<ReturnType<typeof state.addInboundMessage>> = [];
const nativeReplies: Array<{
text: string;
chatId: number;
senderId: number;
forumTopicId?: number;
}> = [];
let restarted = false;
const inspect = (selector: { runId?: string; executionId?: string }) => {
const runId = selector.runId ?? selector.executionId?.replace("execution-", "");
const index = admittedRuns.indexOf(runId ?? "") - 1;
const turn = turns[index];
if (!turn || !runId) {
throw new Error("identity proof must inspect only a newly admitted transport turn");
}
const additional = turn.senderId === additionalUserId;
const principalRef =
fault === "raw-principal"
? turn.senderId
: hmac(
fault === "same-person"
? "b"
: fault === "changed-primary" && index === 1
? "e"
: additional
? "c"
: "b",
);
return {
run: { runId, executionId: `execution-${runId}` },
identity: {
state: "present",
context: {
contextId:
fault === "reused-context"
? "context-first"
: restarted && fault === "restart-drift"
? "context-replacement"
: `context-${runId}`,
executionId: `execution-${runId}`,
runId: fault === "wrong-run" ? "unrelated-run" : runId,
invoker: {
state: fault === "unknown-person" ? "unknown" : "present",
principal: {
kind: fault === "room-principal" ? "service" : "person",
principalRef,
domainRef: hmac("a"),
},
},
// Channel admission supplies no rawSourceRef; do not invent one in proof support.
ingress: { kind: "channel", state: "present" },
runtimeInstance: { runtimeRef: hmac("e") },
assurance:
fault === "missing-assurance"
? []
: [
{
kind: "channel-admission",
strength: "boundary-verified",
evidenceRef: hmac("d"),
},
],
applicableGrants: [],
...(fault === "oversized-context" ? { padding: "x".repeat(16384) } : {}),
},
},
decisionDisplays: [
{
action: { family: "decision", operation: "record" },
provenance: { state: fault === "verified-generic" ? "verified" : "unverified" },
},
],
...(fault === "raw-room" || (restarted && fault === "restart-leak")
? { leakedReference: String(botId) }
: {}),
...(fault === "prompt-leak" ? { leakedText: turn.text } : {}),
};
};
const call = vi.fn(
async (
method: string,
selector: { runId?: string; executionId?: string; decisionLimit: number },
) => {
expect(method).toBe("audit.run.inspect");
// audit.run.inspect has a closed request schema, distinct from audit CLI --limit.
expect(Object.keys(selector).toSorted()).toEqual([
"decisionLimit",
selector.runId === undefined ? "executionId" : "runId",
]);
expect(selector.decisionLimit).toBe(100);
return inspect(selector);
},
);
const restart = vi.fn(async (mutate: () => Promise<void>) => {
await mutate();
restarted = true;
});
const runQaCli = vi.fn(async (_env: unknown, args: string[], options?: { json?: boolean }) => {
expect(args[0]).toBe("audit");
if (args.includes("--kind")) {
return { events: admittedRuns.map((runId) => ({ runId })) };
}
expect(args.slice(0, 2)).toEqual(["audit", "--execution"]);
const executionId = args[2];
if (!executionId) {
throw new Error("identity CLI proof requires an exact execution selector");
}
const inspection = inspect({ executionId });
if (!options?.json) {
return fault === "missing-human"
? "Identity unavailable"
: `Invoker [present] ${inspection.identity.context.invoker.principal.principalRef}\nDecisions`;
}
return fault === "wrong-cli-execution"
? { ...inspection, run: { ...inspection.run, executionId: "foreign-execution" } }
: inspection;
});
return {
call,
restart,
runQaCli,
turns,
result: runLoadedScenarioFlow(scenarioId, {
state,
api: {
env: {
providerMode: fault === "wrong-provider" ? "live-frontier" : "mock-openai",
gateway: { call, restartAfterStateMutation: restart },
},
// Only the adapter's prepared, noncredential surface is supplied here.
telegramIdentityFixture:
fault === "missing-fixture"
? undefined
: {
participantAliases:
fault === "missing-participant"
? ["primary"]
: ["primary", fault === "duplicate-alias" ? "primary" : "guest"],
forumTopicId: fault === "missing-topic" ? undefined : forumTopicId,
},
readTelegramMessages: () => nativeReplies,
transport: {
id: fault === "wrong-transport" ? "qa-channel" : "telegram",
reset: async () => state.reset(),
sendInbound: async (input: Parameters<typeof state.addInboundMessage>[0]) => {
expect(["primary", "guest"]).toContain(input.senderId);
const inbound = state.addInboundMessage({
...input,
senderId: input.senderId === "primary" ? primaryUserId : additionalUserId,
});
turns.push(inbound);
return inbound;
},
waitForOutbound: async (input: {
conversation: { id: string; kind: string };
threadId?: string;
textIncludes: string;
}) => {
const inbound = turns.at(-1);
if (!inbound) {
throw new Error("identity proof must send a transport turn before inspecting it");
}
expect(input.conversation).toEqual(inbound.conversation);
expect(input.threadId).toBe(inbound.threadId);
const forum = inbound.conversation.kind === "group";
expect(inbound.threadId).toBe(forum ? String(forumTopicId) : undefined);
expect(inbound.text.startsWith("@openclaw ")).toBe(forum);
expect(inbound.text).toContain(`Reply exactly: ${input.textIncludes}`);
admittedRuns.push(`run-${turns.length}`);
if (fault === "extra-run") {
admittedRuns.push("unrelated-new-run");
}
nativeReplies.push({
text: input.textIncludes,
chatId: forum ? forumGroupId : botId,
senderId: botId,
...(forum
? { forumTopicId: fault === "wrong-topic" ? forumTopicId + 1 : forumTopicId }
: {}),
});
return state.addOutboundMessage({
accountId: "sut",
to: `${forum ? "group" : "dm"}:${inbound.conversation.id}`,
threadId: inbound.threadId,
text: input.textIncludes,
});
},
},
runQaCli,
},
}),
};
}
describe("Telegram participant identity executable flow", () => {
it("catalogs live participant identity qualification with the fixture gate", () => {
const scenarios = readQaScenarioPack().scenarios.filter(
(scenario) =>
scenario.execution.kind === "flow" &&
scenario.execution.channels?.includes("telegram") &&
scenario.execution.config?.requiredChannelDriver === "live" &&
scenario.execution.config.requireParticipantIdentityFixture === true,
);
expect(scenarios.map((scenario) => scenario.id)).toContain(scenarioId);
expect(
resolveTelegramQaScenarioIds({ providerMode: "mock-openai", scenarioIds: [scenarioId] }),
).toEqual([scenarioId]);
const scenario = readQaScenarioById(scenarioId);
expect(scenario.execution).toMatchObject({ suiteIsolation: "isolated", retryCount: 0 });
expect(scenario.gatewayConfigPatch).toMatchObject({
logging: { audit: { enabled: true, executionIdentity: true } },
});
expect(
selectQaFlowSuiteScenarios({
scenarios: [scenario],
channel: "telegram",
channelDriver: "crabline",
providerMode: "mock-openai",
primaryModel: "mock-openai/fixture",
}),
).toEqual([]);
});
it("executes both participant aliases through DM/forum, exact RPC/CLI inspection, and one restart", async () => {
const proof = runIdentityFlow();
await expect(proof.result).resolves.toMatchObject({ status: "pass" });
expect(proof.turns.map((turn) => [turn.senderId, turn.conversation.kind])).toEqual([
[primaryUserId, "direct"],
[primaryUserId, "group"],
[additionalUserId, "group"],
]);
expect(proof.restart).toHaveBeenCalledOnce();
expect(proof.call.mock.calls.map(([, selector]) => selector)).toEqual([
{ runId: "run-1", decisionLimit: 100 },
{ executionId: "execution-run-1", decisionLimit: 100 },
{ runId: "run-2", decisionLimit: 100 },
{ executionId: "execution-run-2", decisionLimit: 100 },
{ runId: "run-3", decisionLimit: 100 },
{ executionId: "execution-run-3", decisionLimit: 100 },
{ executionId: "execution-run-1", decisionLimit: 100 },
{ executionId: "execution-run-2", decisionLimit: 100 },
{ executionId: "execution-run-3", decisionLimit: 100 },
]);
expect(
proof.runQaCli.mock.calls.filter(([, args]) => args.includes("--execution")),
).toHaveLength(12);
});
it.each([
["wrong-transport", "requires the live Telegram adapter"],
["wrong-provider", "requires the live Telegram adapter"],
["missing-fixture", "requires distinct participant aliases"],
["missing-participant", "requires distinct participant aliases"],
["duplicate-alias", "requires distinct participant aliases"],
["missing-topic", "requires distinct participant aliases"],
["wrong-topic", "requested DM or leased forum topic"],
["extra-run", "exactly one newly admitted run"],
["unknown-person", "retain the admitted person"],
["room-principal", "retain the admitted person"],
["wrong-run", "retain the admitted person"],
["raw-principal", "bounded redacted identity"],
["missing-assurance", "bounded redacted identity"],
["raw-room", "bounded redacted identity"],
["prompt-leak", "bounded redacted identity"],
["oversized-context", "bounded redacted identity"],
["verified-generic", "bounded redacted identity"],
["wrong-cli-execution", "must agree with run discovery"],
["missing-human", "must agree with run discovery"],
["same-person", "distinguish the additional participant"],
["changed-primary", "preserve the same primary person"],
["reused-context", "three distinct execution contexts"],
["restart-drift", "changed or exposed private references after restart"],
["restart-leak", "changed or exposed private references after restart"],
] satisfies Array<[Fault, string]>)("rejects %s evidence", async (fault, message) => {
await expect(runIdentityFlow(fault).result).rejects.toThrow(message);
});
});

View file

@ -0,0 +1,142 @@
import fs from "node:fs";
import os from "node:os";
import path from "node:path";
import { afterEach, describe, expect, it, vi } from "vitest";
import {
readTelegramPrivateProductionDescriptor,
requestTelegramPrivateAppTurn,
resolveTelegramPrivateProductionBot,
} from "./private-production.runtime.js";
const fetchWithSsrFGuardMock = vi.hoisted(() => vi.fn());
vi.mock("openclaw/plugin-sdk/ssrf-runtime", () => ({
fetchWithSsrFGuard: fetchWithSsrFGuardMock,
}));
const roots: string[] = [];
function writeDescriptor() {
const root = fs.mkdtempSync(path.join(os.tmpdir(), "telegram-private-apps-test-"));
roots.push(root);
const file = path.join(root, "descriptor.json");
fs.writeFileSync(
file,
JSON.stringify({
mode: "private-production-local-apps",
forumGroupId: "-100456",
forumTopicId: 42,
topicTitle: "Private proof",
participants: [
{ alias: "primary", host: "mainframe", userId: "100" },
{ alias: "second", host: "macbook", userId: "101" },
],
}),
{ mode: 0o600 },
);
return file;
}
afterEach(() => {
fetchWithSsrFGuardMock.mockReset();
for (const root of roots.splice(0)) {
fs.rmSync(root, { force: true, recursive: true });
}
});
describe("Telegram private production local-app proof", () => {
it("parses two distinct operator-local participants without credential material", () => {
const file = writeDescriptor();
expect(readTelegramPrivateProductionDescriptor(file)).toEqual({
file,
mode: "private-production-local-apps",
forumGroupId: "-100456",
forumTopicId: 42,
topicTitle: "Private proof",
participants: [
{ alias: "primary", host: "mainframe", userId: "100" },
{ alias: "second", host: "macbook", userId: "101" },
],
});
});
it("resolves the bot through the guarded network boundary and releases it", async () => {
const release = vi.fn();
fetchWithSsrFGuardMock.mockResolvedValue({
response: new Response(
JSON.stringify({
ok: true,
result: { id: 700000001, username: "qa_bot" },
}),
{ status: 200 },
),
release,
});
await expect(
resolveTelegramPrivateProductionBot({ TELEGRAM_BOT_TOKEN: "test-token" }),
).resolves.toEqual({
id: "700000001",
token: "test-token",
username: "qa_bot",
});
expect(fetchWithSsrFGuardMock).toHaveBeenCalledWith({
url: "https://api.telegram.org/bottest-token/getMe",
init: { method: "POST" },
timeoutMs: 30_000,
maxRedirects: 0,
auditContext: "qa-lab-telegram-private-production-bot-api",
});
expect(release).toHaveBeenCalledOnce();
});
it("accepts a native UI acknowledgement without placing raw participant IDs in the handoff", async () => {
const file = writeDescriptor();
const descriptor = readTelegramPrivateProductionDescriptor(file)!;
const pending = requestTelegramPrivateAppTurn({
descriptor,
destination: "forum-topic",
participant: descriptor.participants[1]!,
text: "@qa_bot Reply exactly: marker",
});
const proofRoot = `${file}.app-proof`;
let requestPath: string | undefined;
for (let attempts = 0; attempts < 100 && !requestPath; attempts += 1) {
requestPath = fs.existsSync(proofRoot)
? fs
.readdirSync(proofRoot)
.map((entry) => path.join(proofRoot, entry))
.find((entry) => entry.endsWith(".request.json"))
: undefined;
if (!requestPath) {
await new Promise<void>((resolve) => {
setTimeout(resolve, 10);
});
}
}
expect(requestPath).toBeDefined();
const requestText = fs.readFileSync(requestPath!, "utf8");
expect(requestText).not.toContain('"100"');
expect(requestText).not.toContain('"101"');
const request = JSON.parse(requestText) as {
destination: string;
text: string;
token: string;
};
fs.writeFileSync(
path.join(proofRoot, `${request.token}.ack.json`),
JSON.stringify({
schemaVersion: 1,
token: request.token,
sentText: request.text,
replyText: "marker",
replyObservedIn: request.destination,
replyToRequestedMessage: true,
}),
{ mode: 0o600 },
);
await expect(pending).resolves.toEqual({ replyText: "marker" });
expect(fs.readdirSync(proofRoot)).toEqual([]);
});
});

View file

@ -0,0 +1,226 @@
import { randomUUID } from "node:crypto";
import fs from "node:fs";
import fsp from "node:fs/promises";
import path from "node:path";
import { fetchWithSsrFGuard } from "openclaw/plugin-sdk/ssrf-runtime";
import { isRecord } from "openclaw/plugin-sdk/string-coerce-runtime";
export type TelegramPrivateAppParticipant = {
alias: string;
host: string;
userId: string;
};
export type TelegramPrivateProductionDescriptor = {
file: string;
forumGroupId: string;
forumTopicId: number;
mode: "private-production-local-apps";
participants: TelegramPrivateAppParticipant[];
topicTitle: string;
};
type TelegramBotIdentity = {
id: string;
token: string;
username: string;
};
function requireString(value: Record<string, unknown>, key: string) {
const item = value[key];
if (typeof item !== "string" || !item.trim()) {
throw new Error(`Telegram private production descriptor has invalid ${key}.`);
}
return item.trim();
}
function parseParticipant(value: unknown): TelegramPrivateAppParticipant {
if (!isRecord(value)) {
throw new Error("Telegram private production participant is not an object.");
}
const participant = {
alias: requireString(value, "alias"),
host: requireString(value, "host"),
userId: requireString(value, "userId"),
};
if (!/^\d+$/u.test(participant.userId)) {
throw new Error("Telegram private production participant has invalid userId.");
}
return participant;
}
export function readTelegramPrivateProductionDescriptor(
file: string | undefined,
): TelegramPrivateProductionDescriptor | undefined {
if (!file?.trim()) {
return undefined;
}
const stat = fs.lstatSync(file);
if (!stat.isFile() || stat.isSymbolicLink()) {
throw new Error("Telegram private production descriptor must be a regular file.");
}
const value: unknown = JSON.parse(fs.readFileSync(file, "utf8"));
if (!isRecord(value) || value.mode !== "private-production-local-apps") {
throw new Error("Telegram credential descriptor must select private-production-local-apps.");
}
const participants = Array.isArray(value.participants)
? value.participants.map(parseParticipant)
: [];
if (
participants.length < 2 ||
participants[0]?.alias !== "primary" ||
new Set(participants.map((participant) => participant.alias)).size !== participants.length ||
new Set(participants.map((participant) => participant.userId)).size !== participants.length
) {
throw new Error(
"Telegram private production descriptor requires primary plus a distinct local-app participant.",
);
}
const forumGroupId = requireString(value, "forumGroupId");
if (!/^-\d+$/u.test(forumGroupId)) {
throw new Error("Telegram private production descriptor has invalid forumGroupId.");
}
const forumTopicId = value.forumTopicId;
if (
typeof forumTopicId !== "number" ||
!Number.isSafeInteger(forumTopicId) ||
forumTopicId <= 0
) {
throw new Error("Telegram private production descriptor has invalid forumTopicId.");
}
return {
file,
forumGroupId,
forumTopicId,
mode: value.mode,
participants,
topicTitle: requireString(value, "topicTitle"),
};
}
async function botApi(bot: TelegramBotIdentity, method: string) {
try {
const guarded = await fetchWithSsrFGuard({
url: `https://api.telegram.org/bot${bot.token}/${method}`,
init: { method: "POST" },
timeoutMs: 30_000,
maxRedirects: 0,
auditContext: "qa-lab-telegram-private-production-bot-api",
});
try {
const value: unknown = await guarded.response.json();
if (!guarded.response.ok || !isRecord(value) || value.ok !== true) {
throw new Error("request rejected");
}
return value.result;
} finally {
await guarded.release();
}
} catch {
throw new Error(`Telegram private production Bot API ${method} failed.`);
}
}
export async function resolveTelegramPrivateProductionBot(env: NodeJS.ProcessEnv) {
const token = env.TELEGRAM_BOT_TOKEN?.trim();
if (!token) {
throw new Error("Telegram private production proof requires TELEGRAM_BOT_TOKEN.");
}
const placeholder = { id: "", token, username: "" };
const result = await botApi(placeholder, "getMe");
if (!isRecord(result) || typeof result.id !== "number" || typeof result.username !== "string") {
throw new Error("Telegram private production bot identity is invalid.");
}
return { id: String(result.id), token, username: result.username };
}
function appProofRoot(descriptorFile: string) {
return `${descriptorFile}.app-proof`;
}
async function readAcknowledgement(params: {
acknowledgementPath: string;
destination: "bot-dm" | "forum-topic";
sentText: string;
token: string;
}) {
const deadline = Date.now() + 10 * 60_000;
while (Date.now() < deadline) {
try {
const stat = await fsp.lstat(params.acknowledgementPath);
if (!stat.isFile() || stat.isSymbolicLink()) {
throw new Error("Telegram.app proof acknowledgement must be a regular file.");
}
const value: unknown = JSON.parse(await fsp.readFile(params.acknowledgementPath, "utf8"));
if (
!isRecord(value) ||
value.schemaVersion !== 1 ||
value.token !== params.token ||
value.sentText !== params.sentText ||
value.replyObservedIn !== params.destination ||
value.replyToRequestedMessage !== true ||
typeof value.replyText !== "string" ||
!value.replyText.trim()
) {
throw new Error("Telegram.app proof acknowledgement is invalid.");
}
return value.replyText;
} catch (error) {
// SAFETY: Node filesystem rejections carry errno codes; only ENOENT is retryable here.
if ((error as NodeJS.ErrnoException).code !== "ENOENT") {
throw error;
}
}
await new Promise<void>((resolve) => {
setTimeout(resolve, 250);
});
}
throw new Error("Timed out waiting for Telegram.app send and native reply proof.");
}
export async function requestTelegramPrivateAppTurn(params: {
descriptor: TelegramPrivateProductionDescriptor;
destination: "bot-dm" | "forum-topic";
participant: TelegramPrivateAppParticipant;
text: string;
}) {
const root = appProofRoot(params.descriptor.file);
await fsp.mkdir(root, { recursive: true, mode: 0o700 });
await fsp.chmod(root, 0o700);
const token = randomUUID();
const requestPath = path.join(root, `${token}.request.json`);
const acknowledgementPath = path.join(root, `${token}.ack.json`);
await fsp.writeFile(
requestPath,
`${JSON.stringify(
{
schemaVersion: 1,
token,
participant: { alias: params.participant.alias, host: params.participant.host },
destination: params.destination,
...(params.destination === "forum-topic"
? { topicTitle: params.descriptor.topicTitle }
: {}),
text: params.text,
},
null,
2,
)}\n`,
{ flag: "wx", mode: 0o600 },
);
process.stdout.write(`TELEGRAM_PRIVATE_APP_SEND_REQUIRED ${token}\n`);
try {
const replyText = await readAcknowledgement({
acknowledgementPath,
destination: params.destination,
sentText: params.text,
token,
});
return { replyText };
} finally {
await Promise.all([
fsp.rm(requestPath, { force: true }),
fsp.rm(acknowledgementPath, { force: true }),
]);
}
}

View file

@ -25,4 +25,12 @@ describe("resolveTelegramQaRunOptions", () => {
"supports only --credential-source convex",
);
});
it("preserves the private production credential descriptor", () => {
expect(
resolveTelegramQaRunOptions({
credentialFile: "/private/telegram-production.json",
}),
).toMatchObject({ credentialFile: "/private/telegram-production.json" });
});
});

View file

@ -39,6 +39,7 @@ export function resolveTelegramQaRunOptions(
scenarioIds: opts.scenarioIds,
listScenarios: opts.listScenarios,
sutAccountId: opts.sutAccountId,
credentialFile: opts.credentialFile,
credentialSource,
credentialRole: opts.credentialRole?.trim(),
};

View file

@ -55,6 +55,26 @@ describe("Telegram QA API boundary", () => {
});
});
it("omits apiRoot for a production Bot API qualification", () => {
const config = buildTelegramQaConfig(
{},
{
groupId: "-10042",
sutAccountId: "sut",
sutToken: "secret-token",
testerUserId: "100",
additionalTesterUserIds: ["101"],
enableDirectMessages: true,
},
);
expect(config.channels?.telegram?.accounts?.sut).not.toHaveProperty("apiRoot");
expect(config.channels?.telegram?.accounts?.sut).toMatchObject({
allowFrom: ["100", "101"],
groups: { "-10042": { allowFrom: ["100", "101"] } },
});
});
it("waits for the selected Telegram account to connect", async () => {
const call = vi
.fn()

View file

@ -23,14 +23,18 @@ const TELEGRAM_QA_DEFAULT_READY_TIMEOUT_MS = 45_000;
export function buildTelegramQaConfig(
baseCfg: OpenClawConfig,
params: {
apiRoot: string;
apiRoot?: string;
directMessageOnly?: boolean;
enableDirectMessages?: boolean;
additionalTesterUserIds?: string[];
forumGroupId?: string;
groupId: string;
sutAccountId: string;
sutToken: string;
testerUserId: string;
},
): OpenClawConfig {
const testerUserIds = [params.testerUserId, ...(params.additionalTesterUserIds ?? [])];
return {
...baseCfg,
agents: {
@ -71,19 +75,25 @@ export function buildTelegramQaConfig(
[params.sutAccountId]: {
enabled: true,
botToken: params.sutToken,
apiRoot: params.apiRoot,
...(params.directMessageOnly
? { dmPolicy: "allowlist", allowFrom: [params.testerUserId] }
...(params.apiRoot ? { apiRoot: params.apiRoot } : {}),
...(params.directMessageOnly || params.enableDirectMessages
? { dmPolicy: "allowlist", allowFrom: testerUserIds }
: { dmPolicy: "disabled" }),
groups: {
[params.groupId]: {
groupPolicy: "allowlist",
allowFrom: [params.testerUserId],
// Concurrent leases share this group and QA sender. Only this
// bot's mentions or reply chain may trigger an agent turn.
requireMention: true,
},
},
groups: Object.fromEntries(
uniqueStrings([
params.groupId,
...(params.forumGroupId ? [params.forumGroupId] : []),
]).map((groupId) => [
groupId,
{
groupPolicy: "allowlist",
allowFrom: testerUserIds,
// Concurrent leases share this group and QA sender. Only this
// bot's mentions or reply chain may trigger an agent turn.
requireMention: true,
},
]),
),
},
},
},

View file

@ -58,6 +58,10 @@ describe("Telegram userbot driver runtime", () => {
"print(json.dumps({'type':'ready','chatId':-1001,'user':{'id':100}}), flush=True)",
"for line in sys.stdin:",
" request = json.loads(line)",
" if request['method'] == 'cleanup-private-forum':",
" print(json.dumps({'type':'response','id':request['id'],'result':{'ok':True,'status':'deleted'}}), flush=True)",
" continue",
" assert request['chatId'] == '-2002' and request['forumTopicId'] == 42",
" message_id = 10 + int(request['id'])",
" update = {'kind':'message','chatId':-1001,'messageId':message_id + 1,'senderId':200,'timestamp':1000,'text':request['text'],'entities':entities,'contentType':'messagePhoto'}",
" print(json.dumps({'type':'update','update':update}), flush=True)",
@ -67,7 +71,7 @@ describe("Telegram userbot driver runtime", () => {
" for rich in [rich_message, {**rich_message, 'is_full':False}, None]:",
" update = {**update, 'richMessage':rich, 'text':'x', 'entities':[], 'contentType':'messageRichMessage' if rich else 'messageText'}",
" print(json.dumps({'type':'update','update':update}), flush=True)",
" result = {'chatId':-1001,'messageId':message_id,'senderId':100,'timestamp':1000,'text':request['text'],'entities':entities,'contentType':'messageText'}",
" result = {'chatId':-1001,'messageId':message_id,'senderId':100,'timestamp':1000,'text':request['text'],'entities':entities,'contentType':'messageText','forumTopicId':request['forumTopicId']}",
" print(json.dumps({'type':'response','id':request['id'],'result':result}), flush=True)",
].join("\n"),
);
@ -75,6 +79,7 @@ describe("Telegram userbot driver runtime", () => {
const leaseFailure = new Promise<Error>(() => {});
const driver = await TelegramUserbotDriver.start({
chatId: "-1001",
expectedUserId: "100",
driverEnv: {},
leaseHealth: { assertHealthy() {}, whenUnhealthy: leaseFailure },
userDriverPath: scriptPath,
@ -84,13 +89,16 @@ describe("Telegram userbot driver runtime", () => {
});
expect(driver.chatId).toBe(-1001);
try {
await expect(driver.send({ text })).resolves.toMatchObject({
messageId: 11,
senderId: 100,
contentType: "messageText",
text,
entities,
});
await expect(driver.send({ text, chatId: "-2002", forumTopicId: 42 })).resolves.toMatchObject(
{
forumTopicId: 42,
messageId: 11,
senderId: 100,
contentType: "messageText",
text,
entities,
},
);
await vi.waitFor(() => expect(updates).toHaveLength(6));
expect(updates).toMatchObject([
{
@ -120,6 +128,7 @@ describe("Telegram userbot driver runtime", () => {
]);
expect(updates[5]).not.toHaveProperty("richMessage");
expect(() => driver.assertHealthy()).not.toThrow();
await expect(driver.cleanupPrivateForum()).resolves.toBeUndefined();
} finally {
await driver.close();
}

View file

@ -18,12 +18,26 @@ export type TelegramUserbotUpdate = {
kind: "edit" | "message";
messageId: number;
replyToMessageId?: number;
forumTopicId?: number;
threadId?: number;
senderId: number;
senderUsername?: string;
text: string;
timestamp: number;
};
type PendingUserbotCommand =
| {
kind: "send";
reject(error: Error): void;
resolve(value: TelegramUserbotUpdate): void;
}
| {
kind: "cleanup-private-forum";
reject(error: Error): void;
resolve(): void;
};
function isUtf16Boundary(text: string, offset: number) {
const before = text.charCodeAt(offset - 1);
const after = text.charCodeAt(offset);
@ -107,6 +121,8 @@ function parseUserbotUpdate(value: unknown): TelegramUserbotUpdate {
...(typeof value.replyToMessageId === "number"
? { replyToMessageId: value.replyToMessageId }
: {}),
...(typeof value.forumTopicId === "number" ? { forumTopicId: value.forumTopicId } : {}),
...(typeof value.threadId === "number" ? { threadId: value.threadId } : {}),
...(typeof value.senderUsername === "string" ? { senderUsername: value.senderUsername } : {}),
};
}
@ -131,12 +147,10 @@ function waitForChildExit(child: ChildProcessWithoutNullStreams, timeoutMs: numb
export class TelegramUserbotDriver {
private activeChatId: number | undefined;
private activeUserId: number | undefined;
private closing = false;
private commandId = 0;
private readonly pending = new Map<
string,
{ reject(error: Error): void; resolve(value: TelegramUserbotUpdate): void }
>();
private readonly pending = new Map<string, PendingUserbotCommand>();
private readyReject: (error: Error) => void = () => undefined;
private readyResolve: () => void = () => undefined;
private readonly ready: Promise<void>;
@ -181,16 +195,28 @@ export class TelegramUserbotDriver {
static async start(params: {
chatId: string;
observeChatIds?: string[];
expectedUserId?: string;
driverEnv: Record<string, string>;
leaseHealth: { assertHealthy(): void; whenUnhealthy: Promise<Error> };
onUpdate(update: TelegramUserbotUpdate): Promise<void> | void;
userDriverPath: string;
}): Promise<TelegramUserbotDriver> {
params.leaseHealth.assertHealthy();
const child = spawn("python3", [params.userDriverPath, "serve", "--chat", params.chatId], {
env: { ...process.env, ...params.driverEnv },
stdio: ["pipe", "pipe", "pipe"],
});
const child = spawn(
"python3",
[
params.userDriverPath,
"serve",
"--chat",
params.chatId,
...(params.observeChatIds ?? []).flatMap((chatId) => ["--observe-chat", chatId]),
],
{
env: { ...process.env, ...params.driverEnv },
stdio: ["pipe", "pipe", "pipe"],
},
);
const driver = new TelegramUserbotDriver(
child,
(update) => params.onUpdate(update),
@ -203,9 +229,15 @@ export class TelegramUserbotDriver {
timer.unref?.();
try {
await driver.ready;
if (
params.expectedUserId !== undefined &&
String(driver.activeUserId) !== params.expectedUserId
) {
throw new Error("Telegram userbot authorization does not match the leased participant.");
}
return driver;
} catch (error) {
child.kill("SIGTERM");
await driver.close();
throw error;
} finally {
clearTimeout(timer);
@ -231,6 +263,8 @@ export class TelegramUserbotDriver {
return;
}
this.activeChatId = chatId;
this.activeUserId =
isRecord(message.user) && typeof message.user.id === "number" ? message.user.id : undefined;
this.readyResolve();
return;
}
@ -263,6 +297,17 @@ export class TelegramUserbotDriver {
pending.reject(new Error("Telegram userbot emitted an invalid command result."));
return;
}
if (pending.kind === "cleanup-private-forum") {
if (
message.result.ok !== true ||
(message.result.status !== "deleted" && message.result.status !== "not-created")
) {
pending.reject(new Error("Telegram userbot did not confirm private forum cleanup."));
return;
}
pending.resolve();
return;
}
try {
pending.resolve(parseUserbotUpdate({ ...message.result, kind: "message" }));
} catch (error) {
@ -298,20 +343,35 @@ export class TelegramUserbotDriver {
return this.activeChatId;
}
async send(params: { replyToMessageId?: number; text: string }): Promise<TelegramUserbotUpdate> {
async send(params: {
chatId?: string;
forumTopicId?: number;
replyToMessageId?: number;
text: string;
}): Promise<TelegramUserbotUpdate> {
this.leaseHealth.assertHealthy();
this.assertHealthy();
this.commandId += 1;
const id = String(this.commandId);
const result = new Promise<TelegramUserbotUpdate>((resolve, reject) => {
this.pending.set(id, { resolve, reject });
this.pending.set(id, { kind: "send", resolve, reject });
});
this.child.stdin.write(
`${JSON.stringify({ id, method: "send", text: params.text, replyToMessageId: params.replyToMessageId })}\n`,
);
this.child.stdin.write(`${JSON.stringify({ id, method: "send", ...params })}\n`);
return await result;
}
async cleanupPrivateForum() {
this.leaseHealth.assertHealthy();
this.assertHealthy();
this.commandId += 1;
const id = String(this.commandId);
const result = new Promise<void>((resolve, reject) => {
this.pending.set(id, { kind: "cleanup-private-forum", resolve, reject });
});
this.child.stdin.write(`${JSON.stringify({ id, method: "cleanup-private-forum" })}\n`);
await result;
}
async close() {
if (this.closing) {
return;
@ -323,7 +383,9 @@ export class TelegramUserbotDriver {
}
if (!(await waitForChildExit(this.child, 2_000))) {
this.child.kill("SIGKILL");
await waitForChildExit(this.child, 2_000);
if (!(await waitForChildExit(this.child, 2_000))) {
throw new Error("Telegram userbot process exit is unconfirmed.");
}
}
await this.updateChain;
}

View file

@ -4,17 +4,23 @@ import { pathToFileURL } from "node:url";
import { isRecord } from "openclaw/plugin-sdk/string-coerce-runtime";
import { resolvePreferredOpenClawTmpDir } from "openclaw/plugin-sdk/temp-path";
export type TelegramTestCredential = {
type TelegramUserCredential = {
testerUserId: string;
tdlibArchiveBase64: string;
tdlibArchiveSha256: string;
tdlibVersion: string;
};
export type TelegramTestCredential = TelegramUserCredential & {
environment: "test";
groupId: string;
schemaVersion: 1;
sutBotId: string;
sutToken: string;
sutUsername: string;
tdlibArchiveBase64: string;
tdlibArchiveSha256: string;
tdlibVersion: string;
testerUserId: string;
forumGroupId?: string;
forumTopicId?: number;
participants?: Array<TelegramUserCredential & { alias: string }>;
};
type RestoredTelegramTestCredential = TelegramTestCredential & {
@ -132,7 +138,32 @@ export async function loadTelegramUserbotSkillRuntime(params?: {
createStateRoot: () =>
fs.mkdtempSync(path.join(resolvePreferredOpenClawTmpDir(), "openclaw-qa-telegram-")),
parseCredential(value) {
return parseCredentialResult(Reflect.apply(parseCredential, undefined, [value]));
const normalized: unknown = Reflect.apply(parseCredential, undefined, [value]);
const credential = parseCredentialResult(normalized);
// The skill parser owns validation; retain only its normalized optional fixture fields.
if (isRecord(normalized)) {
if (typeof normalized.forumGroupId === "string") {
credential.forumGroupId = normalized.forumGroupId;
}
if (typeof normalized.forumTopicId === "number") {
credential.forumTopicId = normalized.forumTopicId;
}
if (Array.isArray(normalized.participants)) {
credential.participants = normalized.participants.map((participant: unknown) => {
if (!isRecord(participant)) {
throw new Error("Telegram userbot parser returned an invalid participant.");
}
return {
alias: requireString(participant, "alias"),
testerUserId: requireString(participant, "testerUserId"),
tdlibArchiveBase64: requireString(participant, "tdlibArchiveBase64"),
tdlibArchiveSha256: requireString(participant, "tdlibArchiveSha256"),
tdlibVersion: requireString(participant, "tdlibVersion"),
};
});
}
}
return credential;
},
restoreCredential(value, stateRoot) {
return parseRestoredCredential(

View file

@ -0,0 +1,172 @@
import { describe, expect, it } from "vitest";
import { createQaBusState } from "../../bus-state.js";
import { readQaScenarioById } from "../../scenario-catalog.js";
import { runLoadedScenarioFlow } from "../../scenario-flow-runner.test-support.js";
const scenarioId = "whatsapp-participant-identity-inspection";
const driverPhone = "+15550000001";
const groupJid = "120363000000000000@g.us";
type Failure =
| "missing-group"
| "extra-run"
| "unknown-person"
| "room-principal"
| "wrong-run"
| "wrong-execution"
| "raw-phone"
| "raw-group"
| "verified-generic"
| "different-person"
| "missing-human"
| "changed-after-restart";
async function runIdentityFlow(failure?: Failure) {
const state = createQaBusState();
const admittedRuns: string[] = ["previous-run"];
const inspectedExecutions = new Set<string>();
let restarted = false;
const result = await runLoadedScenarioFlow(scenarioId, {
state,
api: {
transport: {
id: "whatsapp",
reset: async () => state.reset(),
sendInbound: async (input: Parameters<typeof state.addInboundMessage>[0]) =>
state.addInboundMessage({ ...input, senderId: driverPhone }),
waitForOutbound: async () => {
const inbound = state.getSnapshot().messages.at(-1);
if (!inbound || inbound.direction !== "inbound") {
throw new Error("identity proof must send a transport turn before inspecting it");
}
const isGroup = inbound.conversation.kind === "group";
expect(inbound.text.startsWith("openclawqa ")).toBe(isGroup);
admittedRuns.push(isGroup ? "group-run" : "dm-run");
if (failure === "extra-run") {
admittedRuns.push("unrelated-new-run");
}
const marker = inbound.text.split("Reply exactly: ")[1];
if (!marker) {
throw new Error("identity proof must request an exact reply marker");
}
return state.addOutboundMessage({
accountId: "sut",
to: `${isGroup ? "group" : "dm"}:${inbound.conversation.id}`,
text: marker,
});
},
},
env: {
providerMode: "mock-openai",
gateway: {
restartAfterStateMutation: async (mutate: () => Promise<void>) => {
await mutate();
restarted = true;
},
},
},
// The real adapter prepares this from its lease; support tests never acquire one.
whatsappScenarioContext: {
runtimeEnv: {
driverPhoneE164: driverPhone,
sutPhoneE164: "+15550000002",
...(failure === "missing-group" ? {} : { groupJid }),
},
},
runQaCli: async (_env: unknown, args: string[], options?: { json?: boolean }) => {
if (args.includes("--kind")) {
return { events: admittedRuns.map((runId) => ({ runId })) };
}
const exact = args.includes("--execution");
const selector = args[args.indexOf(exact ? "--execution" : "--run") + 1];
if (!selector) {
throw new Error("identity proof must select a run or execution");
}
const runId = exact ? selector.replace("execution-", "") : selector;
expect(admittedRuns).toContain(runId);
expect(runId).not.toBe("previous-run");
if (exact) {
inspectedExecutions.add(selector);
}
if (!options?.json) {
return failure === "missing-human"
? "Identity unavailable"
: "Invoker [present]\nDecisions";
}
const isGroup = runId === "group-run";
return {
identity: {
state: "present",
context: {
contextId: `context-${runId}`,
executionId:
failure === "wrong-execution" && exact
? "unrelated-execution"
: `execution-${runId}`,
runId: failure === "wrong-run" ? "unrelated-run" : runId,
invoker: {
state: failure === "unknown-person" ? "unknown" : "present",
principal: {
kind: failure === "room-principal" ? "service" : "person",
principalRef:
(failure === "different-person" && isGroup) ||
(failure === "changed-after-restart" && restarted)
? "principal:other-person"
: "principal:driver",
},
},
ingress: { kind: "channel", state: "present" },
},
},
decisionDisplays: [
{
action: { family: "decision", operation: "record" },
provenance: {
state: failure === "verified-generic" ? "verified" : "unverified",
},
},
],
...(failure === "raw-phone" ? { leakedReference: driverPhone.slice(1) } : {}),
...(failure === "raw-group" ? { leakedReference: groupJid } : {}),
};
},
},
});
return { result, inspectedExecutions, restarted };
}
describe("WhatsApp participant identity executable flow", () => {
it("selects the real transport lane and inspects both executions across restart", async () => {
const scenario = readQaScenarioById(scenarioId);
expect(scenario.execution).toMatchObject({
kind: "flow",
channel: "whatsapp",
suiteIsolation: "isolated",
config: { requiredChannelDriver: "live", requiredProviderMode: "mock-openai" },
});
expect(scenario.gatewayConfigPatch).toMatchObject({
logging: { audit: { executionIdentity: true } },
});
const proof = await runIdentityFlow();
expect(proof.result.status).toBe("pass");
expect(proof.inspectedExecutions).toEqual(new Set(["execution-dm-run", "execution-group-run"]));
expect(proof.restarted).toBe(true);
});
it.each([
["missing-group", "requires groupJid"],
["extra-run", "exactly one newly admitted run"],
["unknown-person", "retain the admitted person"],
["room-principal", "retain the admitted person"],
["wrong-run", "retain the admitted person"],
["wrong-execution", "must agree with run discovery"],
["raw-phone", "exclude raw route and participant references"],
["raw-group", "exclude raw route and participant references"],
["verified-generic", "keep generic decisions unverified"],
["different-person", "same participant"],
["missing-human", "must agree with run discovery"],
["changed-after-restart", "changed or exposed private references after restart"],
] satisfies Array<[Failure, string]>)("rejects %s evidence", async (failure, error) => {
await expect(runIdentityFlow(failure)).rejects.toThrow(error);
});
});

View file

@ -11,6 +11,54 @@ import { runLoadedScenarioFlow } from "./scenario-flow-runner.test-support.js";
import { createRestartFlowFixture } from "./scenario-restart-flow.test-support.js";
describe("qa scenario catalog causality", () => {
it("exposes the message tool directly for delivery decision inspection", () => {
const scenario = readQaScenarioById("message-delivery-decision-inspection");
expect(scenario.gatewayConfigPatch).toMatchObject({
tools: {
toolSearch: false,
alsoAllow: ["message"],
},
});
});
it("requires one host-owned fallback for message suppression and no duplicate after restart", () => {
const scenario = requireFlowScenario(
readQaScenarioById("message-delivery-decision-inspection"),
);
const suppressionActions = scenario.execution.flow?.steps[1]?.actions ?? [];
const restartActions = scenario.execution.flow?.steps[2]?.actions ?? [];
const outboundCount =
"state.getSnapshot().messages.filter((message) => message.direction === 'outbound').length";
expect(scenario.execution.config?.expectedFallbackText).toBe(
"The tool run finished, but no final summary was produced. I did not repeat any completed actions.",
);
expect(suppressionActions).toEqual(
expect.arrayContaining([
expect.objectContaining({
waitForOutbound: {
conversation: { id: "qa-message-suppression-room", kind: "direct" },
textIncludes: { ref: "config.expectedFallbackText" },
timeoutMs: 60000,
},
saveAs: "suppressionOutbound",
}),
]),
);
expect(suppressionActions.map(readFlowAssertExpression)).toContain(
`${outboundCount} === suppressionOutboundStart + 1 && suppressionOutbound.text === config.expectedFallbackText`,
);
expect(restartActions.map(readFlowAssertExpression)).toContain(
`${outboundCount} === suppressionOutboundStart + 1`,
);
expect([...suppressionActions, ...restartActions].map(readFlowAssertExpression)).not.toContain(
`${outboundCount} === suppressionOutboundStart`,
);
expect(JSON.stringify(scenario.execution.flow)).not.toContain("waitForNoOutbound");
expect(JSON.stringify(scenario.execution.flow)).not.toContain("visibleOutbound=0");
});
it("treats denied Telegram admission as silent transport suppression", () => {
for (const scenarioId of [
"telegram-policy-hot-reload",

View file

@ -0,0 +1,215 @@
import { describe, expect, it, vi } from "vitest";
import { readQaScenarioById } from "./scenario-catalog.js";
import { runLoadedScenarioFlow } from "./scenario-flow-runner.test-support.js";
import { selectQaFlowSuiteScenarios, splitModelRef } from "./suite-planning.js";
const SCENARIO = "live-frontier-execution-identity";
type Fault =
| "none"
| "mock-lane"
| "model-fallback"
| "missing-identity"
| "raw-runtime"
| "wrong-run"
| "enforced-admission"
| "prompt-leak"
| "wrong-cli-execution"
| "reused-context"
| "restart-drift";
function runIdentityFlow(fault: Fault = "none") {
let turn = 0;
let marker = "";
let restarted = false;
const contexts = new Map<
string,
{
runId: string;
contextId: string;
executionId: string;
ingress: { state: string; kind: string; boundary: string };
invoker: { state: string };
coverageState: string;
agentPrincipal: { kind: string; principalRef: string };
agentDefinition: { definitionRef: string };
trustDomain: { state: string; domainRef: string };
runtimeInstance: { state: string; kind: string; runtimeRef: string };
}
>();
const call = vi.fn(async (method: string, params: Record<string, string>) => {
if (method === "agent") {
turn += 1;
if (!params.message) {
throw new Error("live identity proof must send an agent message");
}
marker = params.message.replace("Reply exactly: ", "");
const runId = `run-${turn}`;
contexts.set(runId, {
runId,
contextId: fault === "reused-context" ? "context-1" : `context-${turn}`,
executionId: `execution-${turn}`,
ingress: {
state: "present",
kind: "gateway-client",
boundary: "gateway.ws.authenticated-connect",
},
invoker: { state: "absent" },
coverageState: "unattributed",
agentPrincipal: { kind: "agent", principalRef: "qa" },
agentDefinition: { definitionRef: "qa" },
trustDomain: {
state: "present",
domainRef: `hmac-sha256:v1:${"a".repeat(32)}:${"b".repeat(64)}`,
},
runtimeInstance: {
state: "present",
kind: "embedded",
runtimeRef:
fault === "raw-runtime"
? "private-runtime"
: `hmac-sha256:v1:${"a".repeat(32)}:${"c".repeat(64)}`,
},
});
return { status: "accepted", runId };
}
if (method === "chat.history") {
return {
messages: [
{
role: "assistant",
provider: fault === "model-fallback" ? "mock-openai" : "fixture-live",
model: "fixture-model",
stopReason: "stop",
content: [{ type: "text", text: marker }],
},
],
};
}
if (method !== "audit.run.inspect") {
throw new Error(`unexpected Gateway method ${method}`);
}
const context = params.runId
? contexts.get(params.runId)
: [...contexts.values()].find((entry) => entry.executionId === params.executionId);
if (!context) {
throw new Error("inspection selected an unknown execution");
}
return {
run: {
runId: fault === "wrong-run" ? "foreign-run" : context.runId,
executionId: context.executionId,
},
identity:
fault === "missing-identity"
? { state: "unsupported" }
: {
state: "present",
context: {
...context,
...(restarted && fault === "restart-drift" ? { contextId: "replacement" } : {}),
},
},
decisionDisplays: [
{
provenance: { state: "verified", producer: "run-admission" },
decision: {
outcome: fault === "enforced-admission" ? "allowed" : "not-applicable",
reasonCode: "run_admission_identity_not_evaluated",
},
},
],
...(fault === "prompt-leak" ? { privateText: marker } : {}),
};
});
const restart = vi.fn(async (mutate: () => Promise<void>) => {
await mutate();
restarted = true;
});
const runQaCli = vi.fn(async (_env: unknown, args: string[]) => {
expect(args.slice(0, 2)).toEqual(["audit", "--execution"]);
const executionId = args[2];
if (!executionId) {
throw new Error("live identity proof must select an execution");
}
const inspection = await call("audit.run.inspect", { executionId });
if (!args.includes("--json")) {
return "Identity Invoker [absent] Decisions run_admission_identity_not_evaluated not-applicable";
}
if (fault === "wrong-cli-execution") {
return { ...inspection, run: { executionId: "foreign-execution" } };
}
return inspection;
});
return {
call,
restart,
runQaCli,
result: runLoadedScenarioFlow(SCENARIO, {
api: {
env: {
providerMode: fault === "mock-lane" ? "mock-openai" : "live-frontier",
primaryModel: "fixture-live/fixture-model",
gateway: { call, restartAfterStateMutation: restart },
},
splitModelRef,
waitForAgentRun: async (_env: unknown, runId: string) => {
expect(contexts.has(runId)).toBe(true);
return { status: "ok" };
},
waitForAgentHistoryReply: async (
_env: unknown,
_session: string,
predicate: (text: string) => boolean,
) => {
expect(predicate(marker)).toBe(true);
return marker;
},
runQaCli,
},
}),
};
}
describe("live-frontier execution identity qualification", () => {
it("rejects mock selection and admits the operator-selected live model", () => {
const scenario = readQaScenarioById(SCENARIO);
const selection = { scenarios: [scenario], primaryModel: "fixture-live/fixture-model" };
expect(selectQaFlowSuiteScenarios({ ...selection, providerMode: "mock-openai" })).toEqual([]);
expect(() =>
selectQaFlowSuiteScenarios({
...selection,
providerMode: "mock-openai",
scenarioIds: [SCENARIO],
}),
).toThrow("providerMode=live-frontier");
expect(selectQaFlowSuiteScenarios({ ...selection, providerMode: "live-frontier" })).toEqual([
scenario,
]);
expect(scenario.gatewayConfigPatch).toMatchObject({
logging: { audit: { enabled: true, executionIdentity: true } },
});
});
it("executes the registered flow through admission, CLI inspection, and replacement readback", async () => {
const fixture = runIdentityFlow();
await expect(fixture.result).resolves.toMatchObject({ status: "pass" });
expect(fixture.call.mock.calls.filter(([method]) => method === "agent")).toHaveLength(2);
expect(fixture.runQaCli).toHaveBeenCalledTimes(4);
expect(fixture.restart).toHaveBeenCalledOnce();
});
it.each([
["mock-lane", "requires live-frontier"],
["model-fallback", "selected live provider and model"],
["missing-identity", "test condition was not met"],
["raw-runtime", "pseudonymized runtime identity"],
["wrong-run", "exact authenticated ingress"],
["enforced-admission", "overstated admission authority"],
["prompt-leak", "exposed private content"],
["wrong-cli-execution", "different live execution"],
["reused-context", "reused one execution identity"],
["restart-drift", "changed across Gateway replacement"],
] as const)("rejects %s evidence from the same executable flow", async (fault, message) => {
await expect(runIdentityFlow(fault).result).rejects.toThrow(message);
});
});

View file

@ -152,7 +152,10 @@ For `kind: "telegram"`, broker `admin/add` validates that payload includes:
For `kind: "telegram-test-userbot"`, broker `admin/add` accepts only Test
Server schema version 1 with numeric chat, bot, and tester ids; a bot token and
username; a base64 TDLib archive and SHA-256 hash; and a TDLib version.
username; a base64 TDLib archive and SHA-256 hash; and a TDLib version. Optional
participant-identity fixtures add a negative `forumGroupId`, positive numeric
`forumTopicId`, and independently authorized `participants` with distinct
lowercase aliases, tester ids, TDLib archives, hashes, and versions.
For `kind: "buzz"`, broker `admin/add` validates that payload includes:

View file

@ -230,42 +230,114 @@ function normalizeTelegramTestUserbotCredentialPayload(
`Credential payload for kind "${kind}" must use schemaVersion 1 and environment "test".`,
);
}
const normalizeUser = (user: Record<string, unknown>) => {
const testerUserId = requirePayloadString(user, "testerUserId", kind, createFailure);
if (!TELEGRAM_USER_ID_RE.test(testerUserId)) {
throwPayloadError(
createFailure,
`Credential payload for kind "${kind}" has invalid tester identity.`,
);
}
const tdlibArchiveBase64 = requirePayloadString(
user,
"tdlibArchiveBase64",
kind,
createFailure,
);
if (!BASE64_RE.test(tdlibArchiveBase64) || tdlibArchiveBase64.length % 4 !== 0) {
throwPayloadError(
createFailure,
`Credential payload for kind "${kind}" has invalid tdlibArchiveBase64.`,
);
}
const tdlibArchiveSha256 = requirePayloadString(
user,
"tdlibArchiveSha256",
kind,
createFailure,
).toLowerCase();
if (!SHA256_HEX_RE.test(tdlibArchiveSha256)) {
throwPayloadError(
createFailure,
`Credential payload for kind "${kind}" has invalid tdlibArchiveSha256.`,
);
}
return {
testerUserId,
tdlibArchiveBase64,
tdlibArchiveSha256,
tdlibVersion: requirePayloadString(user, "tdlibVersion", kind, createFailure),
};
};
const groupId = requirePayloadString(payload, "groupId", kind, createFailure);
const sutBotId = requirePayloadString(payload, "sutBotId", kind, createFailure);
const testerUserId = requirePayloadString(payload, "testerUserId", kind, createFailure);
if (!TELEGRAM_CHAT_ID_RE.test(groupId)) {
throwPayloadError(createFailure, `Credential payload for kind "${kind}" has invalid groupId.`);
}
if (!TELEGRAM_USER_ID_RE.test(sutBotId) || !TELEGRAM_USER_ID_RE.test(testerUserId)) {
if (!TELEGRAM_USER_ID_RE.test(sutBotId)) {
throwPayloadError(
createFailure,
`Credential payload for kind "${kind}" has invalid bot or tester identity.`,
`Credential payload for kind "${kind}" has invalid bot identity.`,
);
}
const tdlibArchiveBase64 = requirePayloadString(
payload,
"tdlibArchiveBase64",
kind,
createFailure,
);
if (!BASE64_RE.test(tdlibArchiveBase64) || tdlibArchiveBase64.length % 4 !== 0) {
const primary = normalizeUser(payload);
const forumGroupId =
payload.forumGroupId === undefined
? undefined
: requirePayloadString(payload, "forumGroupId", kind, createFailure);
if (forumGroupId && !/^-\d+$/u.test(forumGroupId)) {
throwPayloadError(
createFailure,
`Credential payload for kind "${kind}" has invalid tdlibArchiveBase64.`,
`Credential payload for kind "${kind}" has invalid forumGroupId.`,
);
}
const tdlibArchiveSha256 = requirePayloadString(
payload,
"tdlibArchiveSha256",
kind,
createFailure,
).toLowerCase();
if (!SHA256_HEX_RE.test(tdlibArchiveSha256)) {
const forumTopicId = payload.forumTopicId;
if (
forumTopicId !== undefined &&
(!Number.isSafeInteger(forumTopicId) || Number(forumTopicId) <= 0)
) {
throwPayloadError(
createFailure,
`Credential payload for kind "${kind}" has invalid tdlibArchiveSha256.`,
`Credential payload for kind "${kind}" has invalid forumTopicId.`,
);
}
let participants: Array<ReturnType<typeof normalizeUser> & { alias: string }> | undefined;
if (payload.participants !== undefined) {
if (!Array.isArray(payload.participants)) {
throwPayloadError(
createFailure,
`Credential payload for kind "${kind}" has invalid participants.`,
);
}
const aliases = new Set(["primary"]);
const identities = new Set([primary.testerUserId]);
participants = payload.participants.map((value) => {
if (!value || typeof value !== "object" || Array.isArray(value)) {
throwPayloadError(
createFailure,
`Credential payload for kind "${kind}" has invalid participant.`,
);
}
const participant = value as Record<string, unknown>;
const alias = requirePayloadString(participant, "alias", kind, createFailure);
if (!/^[a-z][a-z0-9-]*$/u.test(alias) || aliases.has(alias)) {
throwPayloadError(
createFailure,
`Credential payload for kind "${kind}" requires distinct lowercase participant aliases.`,
);
}
const user = normalizeUser(participant);
if (identities.has(user.testerUserId)) {
throwPayloadError(
createFailure,
`Credential payload for kind "${kind}" requires distinct participant identities.`,
);
}
aliases.add(alias);
identities.add(user.testerUserId);
return { alias, ...user };
});
}
return {
schemaVersion: 1,
environment: "test",
@ -276,10 +348,10 @@ function normalizeTelegramTestUserbotCredentialPayload(
"",
),
sutBotId,
testerUserId,
tdlibArchiveBase64,
tdlibArchiveSha256,
tdlibVersion: requirePayloadString(payload, "tdlibVersion", kind, createFailure),
...primary,
...(forumGroupId ? { forumGroupId } : {}),
...(forumTopicId === undefined ? {} : { forumTopicId: Number(forumTopicId) }),
...(participants ? { participants } : {}),
} satisfies Record<string, unknown>;
}

View file

@ -19,6 +19,7 @@ scenario:
qa-channel:
allowFrom: ["*"]
tools:
toolSearch: false
alsoAllow: [message]
plugins:
allow: [qa-lab, qa-channel]
@ -35,10 +36,10 @@ scenario:
alsoAllow: [message]
successCriteria:
- A QA Channel reply records queued, platform-started, and delivered as attribution-only owner-native receipts.
- A message-tool metadata echo records one exact private suppression fact without a visible outbound message.
- A message-tool metadata echo records one exact private suppression fact and delivers exactly one host-owned fallback reply.
- Public inspection reports the private generic fact as explicitly unknown without exposing its reason.
- JSON and human CLI inspection expose no raw sender, conversation, message content, or platform message id.
- Both receipt sets remain stable after a lifecycle-owned Gateway restart.
- Both receipt sets remain stable after a lifecycle-owned Gateway restart without adding another outbound message.
docsRefs:
- docs/gateway/audit.md
- docs/cli/audit.md
@ -57,6 +58,7 @@ scenario:
summary: Drive delivered and intentionally suppressed channel turns through an ephemeral Gateway and mock provider, then inspect exact decision receipts before and after restart.
config:
requiredProviderMode: mock-openai
expectedFallbackText: "The tool run finished, but no final summary was produced. I did not repeat any completed actions."
flow:
steps:
@ -176,12 +178,14 @@ flow:
- assert:
expr: "suppressionPrivacySentinels.every((value) => typeof value === 'string' && value.length > 0)"
message: suppression QA message did not expose complete real privacy sentinels
- waitForNoOutbound:
sinceIndex: { ref: suppressionOutboundStart }
quietMs: 1500
- waitForOutbound:
conversation: { id: qa-message-suppression-room, kind: direct }
textIncludes: { ref: config.expectedFallbackText }
timeoutMs: 60000
saveAs: suppressionOutbound
- assert:
expr: "state.getSnapshot().messages.filter((message) => message.direction === 'outbound').length === suppressionOutboundStart"
message: suppression interval produced a visible outbound QA message
expr: "state.getSnapshot().messages.filter((message) => message.direction === 'outbound').length === suppressionOutboundStart + 1 && suppressionOutbound.text === config.expectedFallbackText"
message: suppression interval did not produce exactly one host-owned fallback reply
- call: waitForCondition
saveAs: suppressionRunId
args:
@ -227,9 +231,9 @@ flow:
expr: "suppressionPrivacySentinels.every((sentinel) => !JSON.stringify(suppressionInspect).includes(sentinel) && !suppressionText.includes(sentinel))"
message: suppression JSON or human inspection exposed a raw sender, conversation, inbound message id, or content sentinel
- assert:
expr: "state.getSnapshot().messages.filter((message) => message.direction === 'outbound').length === suppressionOutboundStart"
message: suppression path did not retain exact zero visible outbound messages
detailsExpr: "`suppression facts=1; publicCoverage=unknown; selectorIds=nonempty-unique; visibleOutbound=0; privacy=json+human-pass`"
expr: "state.getSnapshot().messages.filter((message) => message.direction === 'outbound').length === suppressionOutboundStart + 1"
message: suppression path did not retain exactly one visible fallback reply
detailsExpr: "`suppression facts=1; publicCoverage=unknown; selectorIds=nonempty-unique; visibleOutbound=1; privacy=json+human-pass`"
- name: preserves delivery and suppression receipts across restart
actions:
@ -280,6 +284,6 @@ flow:
expr: "deliveryPrivacySentinels.every((sentinel) => !JSON.stringify(deliveryAfterRestart).includes(sentinel) && !deliveryTextAfterRestart.includes(sentinel)) && suppressionPrivacySentinels.every((sentinel) => !JSON.stringify(suppressionAfterRestart).includes(sentinel) && !suppressionTextAfterRestart.includes(sentinel))"
message: restarted JSON or human inspection exposed a raw sender, conversation, message id, or content sentinel
- assert:
expr: "state.getSnapshot().messages.filter((message) => message.direction === 'outbound').length === suppressionOutboundStart"
message: suppression path produced a visible outbound message after restart
expr: "state.getSnapshot().messages.filter((message) => message.direction === 'outbound').length === suppressionOutboundStart + 1"
message: suppression path added another outbound message after restart
detailsExpr: "`restart-stable decisions=byte-identical; selectorIds=nonempty-unique; privacy=json+human-pass`"

View file

@ -0,0 +1,196 @@
title: Telegram DM and forum participant identity inspection
scenario:
id: telegram-participant-identity-inspection
surface: channels
coverage:
primary:
- gateway.identity-and-presence-apis
objective: Verify real Telegram DM and forum-topic admissions preserve one person identity across routes, distinguish a second participant, and retain bounded redacted execution identity across one Gateway restart.
gatewayConfigPatch:
logging:
audit:
enabled: true
executionIdentity: true
successCriteria:
- The primary QA user receives a DM reply and both participants receive replies in the configured forum topic.
- Each turn has one newly admitted run and distinct execution context; the primary person's pseudonym is stable across DM and forum while the additional person's differs.
- audit.run.inspect and audit CLI JSON and human output agree, retain bounded pseudonyms, and expose no raw participant, room, or message content.
- Exact execution identity survives one lifecycle-owned Gateway restart, and generic decision displays remain unverified.
docsRefs:
- docs/gateway/audit.md
- docs/cli/audit.md
- docs/concepts/qa-e2e-automation/channel-qa-reference.md
codeRefs:
- extensions/qa-lab/src/live-transports/telegram/adapter.runtime.ts
- extensions/qa-lab/src/live-transports/telegram/private-production.runtime.ts
execution:
kind: flow
channel: telegram
suiteIsolation: isolated
isolationReason: Enables identity before admission and inspects the same isolated audit database across a Gateway replacement restart.
timeoutMs: 900000
retryCount: 0
summary: Run with qa telegram --scenario telegram-participant-identity-inspection --provider-mode mock-openai. The default mode requires a Convex-leased Test Server fixture with distinct participants and a forum topic. An explicitly supplied private-production descriptor uses two operator-local Telegram.app participants and private native-observation acknowledgements without leasing a production user. Uses the live adapter and no live model; incomplete fixtures fail without fallback.
config:
requiredChannelDriver: live
requiredProviderMode: mock-openai
requireParticipantIdentityFixture: true
turns:
- conversation: { id: qa-telegram-identity-dm, kind: direct }
participantIndex: 0
forum: false
- conversation: { id: qa-telegram-identity-forum, kind: group }
participantIndex: 0
forum: true
- conversation: { id: qa-telegram-identity-forum, kind: group }
participantIndex: 1
forum: true
flow:
steps:
- name: inspects Telegram people across DM and forum executions
actions:
- assert:
expr: "transport.id === 'telegram' && env.providerMode === config.requiredProviderMode"
message: Telegram participant identity proof requires the live Telegram adapter and mock-openai provider
- assert:
expr: "typeof telegramIdentityFixture !== 'undefined' && telegramIdentityFixture.participantAliases?.[0] === 'primary' && telegramIdentityFixture.participantAliases.length >= 2 && new Set(telegramIdentityFixture.participantAliases).size === telegramIdentityFixture.participantAliases.length && Number.isSafeInteger(telegramIdentityFixture.forumTopicId) && telegramIdentityFixture.forumTopicId > 0"
message: Telegram participant identity proof requires distinct participant aliases and forumTopicId from the prepared fixture
- call: waitForGatewayHealthy
args: [{ ref: env }, 60000]
- call: waitForTransportReady
args: [{ ref: env }, 60000]
- resetTransport: true
- set: proofs
value: []
- set: privateReferences
value: []
- set: isSafeInspection
value:
lambda:
params: [value, privateRefs]
expr: >-
value.identity?.state === 'present' &&
Buffer.byteLength(JSON.stringify(value.identity.context), 'utf8') <= 16384 &&
[value.identity.context.invoker.principal?.principalRef, value.identity.context.invoker.principal?.domainRef, value.identity.context.runtimeInstance.runtimeRef, ...(value.identity.context.ingress.sourceRef === undefined ? [] : [value.identity.context.ingress.sourceRef]), ...value.identity.context.assurance.map((item) => item.evidenceRef), ...value.identity.context.applicableGrants.map((item) => item.grantRef)].every((ref) => typeof ref === 'string' && /^hmac-sha256:v1:[a-f0-9]{32}:[a-f0-9]{64}$/.test(ref)) &&
value.identity.context.assurance.length <= 16 && value.identity.context.applicableGrants.length <= 16 &&
value.identity.context.assurance.some((item) => item.kind === 'channel-admission' && item.strength === 'boundary-verified') &&
value.decisionDisplays.length <= 100 && !Object.hasOwn(value, 'decisions') &&
value.decisionDisplays.some((receipt) => receipt.action.family === 'decision') &&
value.decisionDisplays.filter((receipt) => receipt.action.family === 'decision').every((receipt) => receipt.provenance.state === 'unverified') &&
privateRefs.every((ref) => !JSON.stringify(value).includes(ref))
- forEach:
items: { ref: config.turns }
item: turn
index: turnIndex
actions:
- set: marker
value: { expr: "`QA-TELEGRAM-IDENTITY-${turnIndex}-${randomUUID()}`" }
- set: runsBefore
value:
expr: "[...new Set((await runQaCli(env, ['audit', '--kind', 'agent_run', '--limit', '500', '--json'], { timeoutMs: 60000, json: true })).events.map((event) => event.runId).filter((id) => typeof id === 'string'))]"
- sendInbound:
conversation: { ref: turn.conversation }
senderId:
{ expr: "telegramIdentityFixture.participantAliases[turn.participantIndex]" }
threadId:
{
expr: "turn.forum ? String(telegramIdentityFixture.forumTopicId) : undefined",
}
text: { expr: "`${turn.forum ? '@openclaw ' : ''}Reply exactly: ${marker}`" }
saveAs: inbound
- waitForOutbound:
conversation: { ref: turn.conversation }
threadId:
{
expr: "turn.forum ? String(telegramIdentityFixture.forumTopicId) : undefined",
}
textIncludes: { ref: marker }
timeoutMs: 60000
- call: waitForCondition
saveAs: nativeReplies
args:
- lambda:
expr: "(replies => replies.length ? replies : undefined)(readTelegramMessages().filter((reply) => reply.text.includes(marker)))"
- 60000
- 250
- assert:
expr: "nativeReplies.every((reply) => turn.forum ? reply.forumTopicId === telegramIdentityFixture.forumTopicId && reply.chatId < 0 : !reply.forumTopicId && reply.chatId > 0)"
message: Telegram reply must be observed in the requested DM or leased forum topic
- set: privateReferences
value:
expr: "[...new Set([...privateReferences, inbound.senderId, turn.conversation.id, marker, ...nativeReplies.flatMap((reply) => [String(reply.chatId), String(reply.senderId)])])]"
- call: waitForCondition
saveAs: newRunIds
args:
- lambda:
async: true
expr: "(ids => ids.length ? ids : undefined)([...new Set((await runQaCli(env, ['audit', '--kind', 'agent_run', '--limit', '500', '--json'], { timeoutMs: 60000, json: true })).events.map((event) => event.runId).filter((id) => typeof id === 'string' && !runsBefore.includes(id)))])"
- 60000
- 250
- assert:
expr: "newRunIds.length === 1"
message: Telegram turn must produce exactly one newly admitted run
- call: waitForCondition
saveAs: runInspection
args:
- lambda:
async: true
expr: "env.gateway.call('audit.run.inspect', { runId: newRunIds[0], decisionLimit: 100 }).then((value) => value.identity?.state === 'present' && value.decisionDisplays.some((receipt) => receipt.action.family === 'decision') ? value : undefined)"
- 60000
- 250
- assert:
expr: "runInspection.run.runId === newRunIds[0] && runInspection.run.executionId === runInspection.identity.context.executionId && runInspection.identity.context.runId === newRunIds[0] && runInspection.identity.context.invoker.state === 'present' && runInspection.identity.context.invoker.principal?.kind === 'person' && runInspection.identity.context.ingress.kind === 'channel' && runInspection.identity.context.ingress.state === 'present'"
message: Telegram audit inspection must retain the admitted person and channel ingress for this exact run
- set: exactInspection
value:
expr: "await env.gateway.call('audit.run.inspect', { executionId: runInspection.identity.context.executionId, decisionLimit: 100 })"
- set: cliInspection
value:
expr: "await runQaCli(env, ['audit', '--execution', runInspection.identity.context.executionId, '--explain', '--json'], { timeoutMs: 60000, json: true })"
- set: humanInspection
value:
expr: "await runQaCli(env, ['audit', '--execution', runInspection.identity.context.executionId, '--explain'], { timeoutMs: 60000 })"
- assert:
expr: "[exactInspection, cliInspection].every((value) => value.run.executionId === runInspection.identity.context.executionId && JSON.stringify(value.identity.context) === JSON.stringify(runInspection.identity.context)) && humanInspection.includes('Invoker [present]') && humanInspection.includes(runInspection.identity.context.invoker.principal.principalRef) && humanInspection.includes('Decisions')"
message: Telegram exact RPC and CLI inspections must agree with run discovery
- assert:
expr: "[runInspection, exactInspection, cliInspection].every((value) => isSafeInspection(value, privateReferences)) && privateReferences.every((ref) => !humanInspection.includes(ref))"
message: Telegram inspection must retain bounded redacted identity and unverified generic decisions
- set: proofs
value:
expr: "[...proofs, { context: exactInspection.identity.context, senderId: inbound.senderId }]"
- assert:
expr: "proofs.length === 3 && new Set(proofs.map((proof) => proof.context.executionId)).size === 3 && new Set(proofs.map((proof) => proof.context.contextId)).size === 3 && proofs[0].senderId === proofs[1].senderId && proofs[1].senderId !== proofs[2].senderId && proofs[0].context.invoker.principal.principalRef === proofs[1].context.invoker.principal.principalRef && proofs[1].context.invoker.principal.principalRef !== proofs[2].context.invoker.principal.principalRef"
message: Telegram DM and forum must preserve the same primary person and distinguish the additional participant in three distinct execution contexts
detailsExpr: "proofs.map((proof) => `run=${proof.context.runId}; execution=${proof.context.executionId}`).join('; ')"
- name: preserves exact Telegram identity across one Gateway restart
actions:
- call: env.gateway.restartAfterStateMutation
args:
- lambda:
async: true
expr: Promise.resolve()
- call: waitForGatewayHealthy
args: [{ ref: env }, 60000]
- call: waitForTransportReady
args: [{ ref: env }, 60000]
- forEach:
items: { ref: proofs }
item: proof
actions:
- set: restartedRpc
value:
expr: "await env.gateway.call('audit.run.inspect', { executionId: proof.context.executionId, decisionLimit: 100 })"
- set: restartedCli
value:
expr: "await runQaCli(env, ['audit', '--execution', proof.context.executionId, '--explain', '--json'], { timeoutMs: 60000, json: true })"
- set: restartedHuman
value:
expr: "await runQaCli(env, ['audit', '--execution', proof.context.executionId, '--explain'], { timeoutMs: 60000 })"
- assert:
expr: "[restartedRpc, restartedCli].every((value) => isSafeInspection(value, privateReferences) && value.run.executionId === proof.context.executionId && JSON.stringify(value.identity.context) === JSON.stringify(proof.context)) && restartedHuman.includes('Invoker [present]') && restartedHuman.includes(proof.context.invoker.principal.principalRef) && restartedHuman.includes('Decisions') && privateReferences.every((ref) => !restartedHuman.includes(ref))"
message: Telegram exact participant identity changed or exposed private references after restart
detailsExpr: "`restart-stable executions=${proofs.length}`"

View file

@ -0,0 +1,149 @@
title: WhatsApp DM and group participant identity inspection
scenario:
id: whatsapp-participant-identity-inspection
surface: channels
coverage:
primary:
- gateway.identity-and-presence-apis
secondary:
- whatsapp.outbound-text-sends
objective: Verify one real WhatsApp participant keeps the same pseudonym across admitted DM and group turns, with distinct executions and exact audit inspection across restart.
gatewayConfigPatch:
logging:
audit:
executionIdentity: true
successCriteria:
- A leased driver account receives a marker reply through both the DM and mentioned group paths.
- Each turn resolves to one new run and exact execution with a present person invoker and channel ingress.
- The participant pseudonym stays equal across routes; phone numbers, native JIDs, and logical rooms are absent from inspection.
- Generic decision facts remain unverified, and exact JSON plus human inspection survives Gateway restart.
docsRefs:
- docs/gateway/audit.md
- docs/cli/audit.md
- docs/concepts/qa-e2e-automation/whatsapp-and-credentials.md
codeRefs:
- extensions/qa-lab/src/live-transports/whatsapp/adapter.runtime.ts
- extensions/whatsapp/src/inbound/access-control.ts
execution:
kind: flow
channel: whatsapp
suiteIsolation: isolated
isolationReason: Enables identity before admission and inspects the same isolated audit database across a Gateway replacement restart.
timeoutMs: 300000
retryCount: 0
summary: Run with the live WhatsApp driver and mock-openai provider. Requires two dedicated linked Web sessions, their distinct E.164 numbers, and a dedicated groupJid containing both accounts. No live model is needed.
config:
requiredChannelDriver: live
requiredProviderMode: mock-openai
turns:
- conversation: { id: qa-whatsapp-identity-dm, kind: direct }
prefix: ""
marker: QA-WHATSAPP-IDENTITY-DM-OK
- conversation: { id: qa-whatsapp-identity-group, kind: group }
prefix: "openclawqa "
marker: QA-WHATSAPP-IDENTITY-GROUP-OK
flow:
steps:
- name: inspects one WhatsApp person across DM and group executions
actions:
- assert:
expr: "transport.id === 'whatsapp' && env.providerMode === config.requiredProviderMode"
message: WhatsApp participant identity proof requires the live WhatsApp adapter and mock-openai provider
- assert:
expr: "typeof whatsappScenarioContext.runtimeEnv.groupJid === 'string' && whatsappScenarioContext.runtimeEnv.groupJid.length > 0"
message: WhatsApp participant identity proof requires groupJid in the credential payload
- call: waitForGatewayHealthy
args: [{ ref: env }, 60000]
- call: waitForTransportReady
args: [{ ref: env }, 60000]
- resetTransport: true
- set: proofs
value: []
- set: privateReferences
value:
expr: "[whatsappScenarioContext.runtimeEnv.driverPhoneE164, whatsappScenarioContext.runtimeEnv.sutPhoneE164, whatsappScenarioContext.runtimeEnv.groupJid, ...config.turns.map((turn) => turn.conversation.id)].flatMap((value) => [value, value.replace(/^\\+/, '')])"
- forEach:
items: { ref: config.turns }
item: turn
actions:
- set: runsBefore
value:
expr: "[...new Set((await runQaCli(env, ['audit', '--kind', 'agent_run', '--limit', '500', '--json'], { timeoutMs: 60000, json: true })).events.map((event) => event.runId).filter((id) => typeof id === 'string'))]"
- sendInbound:
conversation: { ref: turn.conversation }
senderId: qa-whatsapp-identity-driver
senderName: QA WhatsApp Driver
text: { expr: "`${turn.prefix}Reply exactly: ${turn.marker}`" }
- waitForOutbound:
conversation: { ref: turn.conversation }
textIncludes: { ref: turn.marker }
timeoutMs: 60000
- call: waitForCondition
saveAs: newRunIds
args:
- lambda:
async: true
expr: "(ids => ids.length ? ids : undefined)([...new Set((await runQaCli(env, ['audit', '--kind', 'agent_run', '--limit', '500', '--json'], { timeoutMs: 60000, json: true })).events.map((event) => event.runId).filter((id) => typeof id === 'string' && !runsBefore.includes(id)))])"
- 60000
- 250
- assert:
expr: "newRunIds.length === 1"
message: WhatsApp turn must produce exactly one newly admitted run
- call: waitForCondition
saveAs: runInspection
args:
- lambda:
async: true
expr: "runQaCli(env, ['audit', '--run', newRunIds[0], '--explain', '--json'], { timeoutMs: 60000, json: true }).then((value) => value.identity?.state === 'present' && value.decisionDisplays.some((receipt) => receipt.action.family === 'decision') ? value : undefined)"
- 60000
- 250
- assert:
expr: "runInspection.identity.context.runId === newRunIds[0] && typeof runInspection.identity.context.executionId === 'string' && runInspection.identity.context.invoker.state === 'present' && runInspection.identity.context.invoker.principal?.kind === 'person' && typeof runInspection.identity.context.invoker.principal.principalRef === 'string' && runInspection.identity.context.ingress.kind === 'channel' && runInspection.identity.context.ingress.state === 'present'"
message: WhatsApp audit inspection must retain the admitted person and channel ingress for this run
- set: exactInspection
value:
expr: "await runQaCli(env, ['audit', '--execution', runInspection.identity.context.executionId, '--explain', '--json'], { timeoutMs: 60000, json: true })"
- set: humanInspection
value:
expr: "await runQaCli(env, ['audit', '--execution', runInspection.identity.context.executionId, '--explain'], { timeoutMs: 60000 })"
- assert:
expr: "exactInspection.identity.state === 'present' && JSON.stringify(exactInspection.identity.context) === JSON.stringify(runInspection.identity.context) && humanInspection.includes('Invoker [present]') && humanInspection.includes('Decisions')"
message: WhatsApp exact execution JSON and human inspection must agree with run discovery
- assert:
expr: "exactInspection.decisionDisplays.some((receipt) => receipt.action.family === 'decision' && receipt.provenance.state === 'unverified') && exactInspection.decisionDisplays.filter((receipt) => receipt.action.family === 'decision').every((receipt) => receipt.provenance.state === 'unverified') && privateReferences.every((value) => !JSON.stringify(exactInspection).includes(value) && !humanInspection.includes(value))"
message: WhatsApp inspection must exclude raw route and participant references and keep generic decisions unverified
- set: proofs
value:
expr: "[...proofs, exactInspection.identity.context]"
- assert:
expr: "proofs.length === 2 && new Set(proofs.map((proof) => proof.executionId)).size === 2 && new Set(proofs.map((proof) => proof.contextId)).size === 2 && new Set(proofs.map((proof) => proof.invoker.principal.principalRef)).size === 1"
message: WhatsApp DM and group must have distinct execution contexts for the same participant
detailsExpr: "proofs.map((proof) => `run=${proof.runId}; execution=${proof.executionId}`).join('; ')"
- name: preserves exact WhatsApp identity across Gateway restart
actions:
- call: env.gateway.restartAfterStateMutation
args:
- lambda:
async: true
expr: Promise.resolve()
- call: waitForGatewayHealthy
args: [{ ref: env }, 60000]
- call: waitForTransportReady
args: [{ ref: env }, 60000]
- forEach:
items: { ref: proofs }
item: proof
actions:
- set: restartedInspection
value:
expr: "await runQaCli(env, ['audit', '--execution', proof.executionId, '--explain', '--json'], { timeoutMs: 60000, json: true })"
- set: restartedHuman
value:
expr: "await runQaCli(env, ['audit', '--execution', proof.executionId, '--explain'], { timeoutMs: 60000 })"
- assert:
expr: "restartedInspection.identity.state === 'present' && JSON.stringify(restartedInspection.identity.context) === JSON.stringify(proof) && restartedHuman.includes('Invoker [present]') && restartedHuman.includes('Decisions') && privateReferences.every((value) => !JSON.stringify(restartedInspection).includes(value) && !restartedHuman.includes(value))"
message: WhatsApp exact participant identity changed or exposed private references after restart
detailsExpr: "`restart-stable executions=${proofs.length}`"

View file

@ -10,6 +10,7 @@ scenario:
successCriteria:
- A real local agent turn against the deterministic mock provider records one bounded execution identity context.
- Fresh and existing-install restarts keep execution-identity storage absent before explicit opt-in, and global audit disable prevents new identity rows.
- Real config.patch enable, disable, and reenable operations retain the same Gateway boot and socket, change subsequent admissions, preserve old contexts, and never backfill disabled runs.
- The audit CLI renders text and JSON projections for the exact run, including every identity field and the truthful non-enforcement admission receipt.
- A replacement Gateway process returns byte-equivalent normalized identity context JSON for the same run.
- Two real public-ingress turns that omit runId share the session correlation but retain distinct execution and context ids.
@ -32,7 +33,7 @@ scenario:
kind: script
parallelSafe: true
path: test/e2e/qa-lab/runtime/agent-run-identity-inspection.ts
summary: Starts an ephemeral Gateway and mock provider, runs local and repeated public-ingress turns, proves run ambiguity plus exact text/JSON selection, replaces the Gateway process, and compares normalized context bytes.
summary: Requires a current private-QA dist build (OPENCLAW_BUILD_PRIVATE_QA=1 pnpm build). Starts an ephemeral Gateway and mock provider from that build, runs local and repeated public-ingress turns, proves run ambiguity plus exact text/JSON selection, replaces the Gateway process, and compares normalized context bytes.
args:
- --artifact-base
- ${outputDir}

View file

@ -10,6 +10,8 @@ scenario:
objective: Verify exact-bound cron, task, and flow owner lifecycle rows are explained without generic decision-fact copies.
successCriteria:
- A mapped hook transform suppression returns HTTP 204 before admission and creates no execution identity.
- Admitted mapped webhook POSTs before and after Gateway replacement retain distinct immutable contexts with one stable pseudonymized ingress source, absent invoker evidence, and unattributed admission coverage.
- Mapping IDs, request IDs, and webhook body sentinels stay absent from persisted contexts, JSON and human inspection, and audit-specific Gateway logs before and after replacement.
- A real forced scheduled agent run binds one exact context/execution to its cron receipt and task row.
- A real Gateway agent action binds one exact context/execution to its CLI task row; focused owner tests cover bound flow rows.
- JSON and human audit inspection expose only bounded attribution-only lifecycle displays with closed producers, accept the cron cursor prefix, and omit task/prompt content.
@ -27,10 +29,11 @@ scenario:
- src/tasks/task-flow-registry.store.sqlite.ts
- src/gateway/hooks-mapping.ts
- test/e2e/qa-lab/runtime/autonomous-task-lifecycle-receipts.ts
- test/e2e/qa-lab/runtime/autonomous-task-lifecycle-receipts.fixtures.ts
execution:
kind: script
path: test/e2e/qa-lab/runtime/autonomous-task-lifecycle-receipts.ts
summary: Starts an ephemeral Gateway and mock provider, proves pre-admission suppression, runs cron and CLI task lifecycles, replaces the Gateway, and compares inspection output.
summary: Requires a current private-QA dist build (OPENCLAW_BUILD_PRIVATE_QA=1 pnpm build). Starts an ephemeral Gateway and mock provider from that build, proves pre-admission suppression and admitted mapped webhook identity/privacy, runs cron and CLI task lifecycles, replaces the Gateway, and compares persisted contexts and inspection output.
args:
- --artifact-base
- ${outputDir}

View file

@ -0,0 +1,159 @@
title: Live frontier execution identity inspection
scenario:
id: live-frontier-execution-identity
surface: gateway
coverage:
secondary:
- gateway.identity-and-presence-apis
objective: Inspect distinct admitted executions from two real frontier turns in one session, with exact CLI selection and immutable identity after restart.
gatewayConfigPatch:
logging:
audit:
enabled: true
executionIdentity: true
successCriteria:
- Only the live-frontier provider lane is eligible; the selected model comes from operator input or the existing live defaults.
- Each completed turn has a matching assistant reply from the selected provider and model, plus a distinct execution and context id.
- Run discovery, exact RPC inspection, and human and JSON CLI inspection agree without claiming admission authorization or retaining prompt content.
- Exact context bytes survive a lifecycle-owned Gateway replacement.
docsRefs:
- docs/gateway/audit.md
- docs/cli/audit.md
codeRefs:
- extensions/qa-lab/src/suite-runtime-agent-process.ts
- src/gateway/server-methods/audit.ts
- src/commands/audit.ts
execution:
kind: flow
channel: qa-channel
providerMode: live-frontier
suiteIsolation: isolated
isolationReason: Enables identity collection in isolated state and replaces its Gateway before exact readback.
retryCount: 0
summary: Run `pnpm openclaw qa suite --provider-mode live-frontier --scenario live-frontier-execution-identity`; needs authorized provider auth and the source-owned live model selection. Does not require external channel credentials.
config:
requiredProviderMode: live-frontier
flow:
steps:
- name: inspects two real frontier executions
actions:
- assert:
expr: "env.providerMode === config.requiredProviderMode"
message: execution identity qualification requires live-frontier
- set: selected
value: { expr: "splitModelRef(env.primaryModel)" }
- assert:
expr: "selected?.provider && selected?.model && !['mock-openai', 'aimock'].includes(selected.provider)"
message: execution identity qualification requires a real provider and model
- call: waitForGatewayHealthy
args: [{ ref: env }, 60000]
- set: sessionKey
value: { expr: "`agent:qa:live-identity-${randomUUID()}`" }
- set: inspections
value: []
- forEach:
items: [first, second]
item: turn
actions:
- set: marker
value: { expr: "`QA-LIVE-IDENTITY-${turn}-${randomUUID()}`" }
- set: started
value:
expr: "await env.gateway.call('agent', { agentId: 'qa', sessionKey, message: `Reply exactly: ${marker}`, deliver: false, provider: selected.provider, model: selected.model, idempotencyKey: randomUUID() })"
- assert:
expr: "started.status === 'accepted' && typeof started.runId === 'string' && started.runId.length > 0"
message: live agent turn was not admitted
- call: waitForAgentRun
saveAs: terminal
args:
[
{ ref: env },
{ expr: "started.runId" },
{ expr: "liveTurnTimeoutMs(env, 60000)" },
]
- assert:
expr: "['ok', 'completed', 'succeeded'].includes(terminal.status)"
message: live agent turn did not complete successfully
- call: waitForAgentHistoryReply
args:
- ref: env
- ref: sessionKey
- lambda:
params: [text]
expr: "text.trim() === marker"
- expr: liveTurnTimeoutMs(env, 60000)
- set: history
value: { expr: "await env.gateway.call('chat.history', { sessionKey, limit: 12 })" }
- set: assistant
value:
{
expr: "history.messages.filter((message) => message.role === 'assistant').at(-1)",
}
- assert:
expr: "assistant?.provider === selected.provider && assistant?.model === selected.model && assistant?.stopReason !== 'error' && assistant?.content?.some((part) => part.type === 'text' && part.text.trim() === marker)"
message: completed reply did not come from the selected live provider and model
- call: waitForCondition
saveAs: inspection
args:
- lambda:
async: true
expr: "env.gateway.call('audit.run.inspect', { runId: started.runId }).then((value) => value.identity.state === 'present' ? value : undefined)"
- 60000
- 250
- set: context
value: { expr: "inspection.identity.context" }
- assert:
expr: "inspection.run.runId === started.runId && inspection.run.executionId === context.executionId && context.runId === started.runId && context.contextId && context.executionId && context.ingress.state === 'present' && context.ingress.kind === 'gateway-client' && context.ingress.boundary === 'gateway.ws.authenticated-connect' && context.invoker.state === 'absent' && context.coverageState === 'unattributed'"
message: live execution inspection lost its exact authenticated ingress or fabricated an invoker
- assert:
expr: "context.agentPrincipal.kind === 'agent' && context.agentPrincipal.principalRef === 'qa' && context.agentDefinition.definitionRef === 'qa' && context.trustDomain.state === 'present' && context.runtimeInstance.state === 'present' && ['gateway', 'embedded', 'plugin-harness'].includes(context.runtimeInstance.kind) && [context.trustDomain.domainRef, context.runtimeInstance.runtimeRef].every((ref) => /^hmac-sha256:v1:[a-f0-9]{32}:[a-f0-9]{64}$/.test(ref)) && JSON.stringify(context).length <= 16384"
message: live execution lost bounded agent or pseudonymized runtime identity
- assert:
expr: "inspection.decisionDisplays.some((display) => display.provenance.state === 'verified' && display.provenance.producer === 'run-admission' && display.decision.outcome === 'not-applicable' && display.decision.reasonCode === 'run_admission_identity_not_evaluated') && !Object.hasOwn(inspection, 'decisions') && !JSON.stringify(inspection).includes(marker)"
message: live inspection overstated admission authority or exposed private content
- set: exact
value:
{
expr: "await runQaCli(env, ['audit', '--execution', context.executionId, '--explain', '--json'], { timeoutMs: 60000, json: true })",
}
- assert:
expr: "exact.run.executionId === context.executionId && JSON.stringify(exact.identity.context) === JSON.stringify(context)"
message: exact CLI inspection selected a different live execution
- set: human
value:
{
expr: "await runQaCli(env, ['audit', '--execution', context.executionId, '--explain'], { timeoutMs: 60000 })",
}
- assert:
expr: "['Identity', 'Invoker [absent]', 'Decisions', 'run_admission_identity_not_evaluated', 'not-applicable'].every((label) => human.includes(label)) && !human.includes(marker)"
message: human live inspection omitted non-enforcement evidence or leaked the prompt
- set: inspections
value: { expr: "[...inspections, context]" }
- assert:
expr: "inspections.length === 2 && new Set(inspections.map((context) => context.executionId)).size === 2 && new Set(inspections.map((context) => context.contextId)).size === 2"
message: repeated live turns reused one execution identity
detailsExpr: "`live turns=2; distinct executions=${inspections.length}; exact RPC and CLI identity inspected`"
- name: retains live execution identity across Gateway replacement
actions:
- call: env.gateway.restartAfterStateMutation
args:
- lambda:
async: true
expr: "Promise.resolve()"
- call: waitForGatewayHealthy
args: [{ ref: env }, 60000]
- forEach:
items: { ref: inspections }
item: before
actions:
- set: after
value:
{
expr: "await env.gateway.call('audit.run.inspect', { executionId: before.executionId })",
}
- assert:
expr: "after.run.executionId === before.executionId && after.identity.state === 'present' && JSON.stringify(after.identity.context) === JSON.stringify(before)"
message: live execution identity changed across Gateway replacement
detailsExpr: "`restart-stable exact executions=${inspections.length}`"

View file

@ -12,7 +12,7 @@ scenario:
objective: Prove a forced pre-admission memory flush cannot own the parent session lifecycle and the channel turn still receives a reply.
successCriteria:
- A synthetic channel turn creates the parent session.
- The next turn forces a pre-admission memory flush through the real Gateway child and mock provider.
- The next turn delivers its foreground reply, then records the forced memory flush through the real Gateway child and mock provider.
- The parent session remains terminally healthy and the second turn receives a visible reply.
gatewayConfigPatch:
agents:
@ -31,7 +31,7 @@ scenario:
execution:
kind: flow
runtime: openclaw
summary: Force a memory-flush child before a synthetic channel turn, then verify parent lifecycle and delivery.
summary: Force post-delivery memory maintenance for a synthetic channel turn, then verify its eventual fact, parent lifecycle, and delivery.
channel: qa-channel
config:
conversationPrefix: restart-recovery-memory-flush
@ -103,17 +103,17 @@ flow:
timeoutMs:
expr: liveTurnTimeoutMs(env, 60000)
saveAs: proofReply
- call: readRawQaSessionStore
saveAs: store
- call: waitForCondition
saveAs: sessionEntry
args:
- ref: env
- set: sessionEntry
value:
expr: "store[sessionKey]"
- lambda:
async: true
expr: "readRawQaSessionStore(env).then((store) => { const entry = store[sessionKey]; return entry?.memoryFlush?.kind === 'succeeded' ? entry : undefined; })"
- 60000
- 100
- assert:
expr: "sessionEntry?.memoryFlush?.kind === 'succeeded'"
message:
expr: "`forced memory flush did not succeed: ${JSON.stringify({ sessionKey, storeKeys: Object.keys(store), sessionEntry })}`"
message: forced post-delivery memory flush did not succeed
- assert:
expr: "sessionEntry?.status === 'done' && sessionEntry?.abortedLastRun !== true"
message:

View file

@ -22,6 +22,11 @@ HOST_BUILD="${OPENCLAW_CODEX_NPM_PLUGIN_HOST_BUILD:-1}"
PACKAGE_TGZ="${OPENCLAW_CURRENT_PACKAGE_TGZ:-}"
PROFILE_FILE="${OPENCLAW_CODEX_NPM_PLUGIN_PROFILE_FILE:-${OPENCLAW_TESTBOX_PROFILE_FILE:-$HOME/.openclaw-testbox-live.profile}}"
CODEX_PLUGIN_SPEC="${OPENCLAW_CODEX_NPM_PLUGIN_SPEC:-}"
AUDIT_IDENTITY="${OPENCLAW_CODEX_NPM_PLUGIN_AUDIT_IDENTITY:-0}"
case "$AUDIT_IDENTITY" in
0|1) ;;
*) echo "OPENCLAW_CODEX_NPM_PLUGIN_AUDIT_IDENTITY must be 0 or 1" >&2; exit 1 ;;
esac
CODEX_PLUGIN_MOUNT=()
CODEX_PLUGIN_PACK_DIR=""
CODEX_PLUGIN_REGISTRY_PACKAGE=""
@ -214,6 +219,7 @@ if ! docker_e2e_run_with_harness \
-e OPENCLAW_CODEX_NPM_PLUGIN_FORCE_UNSAFE_INSTALL="${OPENCLAW_CODEX_NPM_PLUGIN_FORCE_UNSAFE_INSTALL:-1}" \
-e OPENCLAW_CODEX_NPM_PLUGIN_MODEL="${OPENCLAW_CODEX_NPM_PLUGIN_MODEL:-openai/gpt-5.4}" \
-e OPENCLAW_CODEX_NPM_PLUGIN_SPEC="$CODEX_PLUGIN_SPEC" \
-e OPENCLAW_CODEX_NPM_PLUGIN_AUDIT_IDENTITY="$AUDIT_IDENTITY" \
-e OPENCLAW_CODEX_NPM_PLUGIN_REGISTRY_PACKAGE="$CODEX_PLUGIN_REGISTRY_PACKAGE" \
-e OPENCLAW_CODEX_NPM_PLUGIN_REGISTRY_TARBALL="$CODEX_PLUGIN_REGISTRY_TARBALL" \
-e OPENCLAW_CODEX_NPM_PLUGIN_REGISTRY_VERSION="$CODEX_PLUGIN_REGISTRY_VERSION" \
@ -295,6 +301,7 @@ dump_debug_logs() {
/tmp/openclaw-codex-agent-turn1.err \
/tmp/openclaw-codex-agent-turn2.json \
/tmp/openclaw-codex-agent-turn2.err \
/tmp/openclaw-codex-audit-gateway.log \
/tmp/openclaw-codex-followthrough.json \
/tmp/openclaw-codex-followthrough.log \
/tmp/openclaw-codex-followthrough.err \
@ -305,12 +312,14 @@ dump_debug_logs() {
}
registry_pid=""
audit_gateway_pid=""
debug_logs_dumped=0
cleanup_scenario() {
local status=$?
trap - EXIT
set +e
openclaw_e2e_stop_process "${registry_pid:-}"
openclaw_e2e_stop_process "${audit_gateway_pid:-}"
if [ "$status" -ne 0 ] && [ "$debug_logs_dumped" -eq 0 ]; then
dump_debug_logs "$status"
fi
@ -456,6 +465,17 @@ run_agent_turn \
node scripts/e2e/lib/codex-npm-plugin-live/assertions.mjs assert-agent-turn "$SUCCESS_MARKER" "$SESSION_ID" "$MODEL_REF"
if [ "${OPENCLAW_CODEX_NPM_PLUGIN_AUDIT_IDENTITY:-0}" = "1" ]; then
echo "Inspecting persisted Codex execution identity through the installed package Gateway..."
audit_package_root="$(openclaw_e2e_package_root "$NPM_CONFIG_PREFIX")"
audit_package_entry="$(openclaw_e2e_package_entrypoint "$audit_package_root")"
audit_gateway_pid="$(openclaw_e2e_start_gateway "$audit_package_entry" 18789 /tmp/openclaw-codex-audit-gateway.log)"
openclaw_e2e_wait_gateway_ready "$audit_gateway_pid" /tmp/openclaw-codex-audit-gateway.log
node scripts/e2e/lib/codex-npm-plugin-live/assertions.mjs assert-audit "$SUCCESS_MARKER"
openclaw_e2e_stop_process "$audit_gateway_pid"
audit_gateway_pid=""
fi
FOLLOWTHROUGH_SESSION_ID="${SESSION_ID}-followthrough"
FOLLOWTHROUGH_PROGRESS_MARKER="${SUCCESS_MARKER}-FOLLOWTHROUGH-PROGRESS"
FOLLOWTHROUGH_COMPLETE_MARKER="${SUCCESS_MARKER}-FOLLOWTHROUGH-COMPLETE"

View file

@ -1,5 +1,6 @@
// Assertions for Codex npm plugin live E2E scenarios.
import { createHash } from "node:crypto";
import { execFileSync } from "node:child_process";
import { createHash, randomUUID } from "node:crypto";
import fs from "node:fs";
import path from "node:path";
import { DatabaseSync } from "node:sqlite";
@ -21,6 +22,7 @@ import {
stateDir,
} from "../codex-install-utils.mjs";
import { assertCodexReleasePackageContract } from "../codex-release-package-assertions.mjs";
import { inspectCodexAudit } from "./audit-inspection.mjs";
const command = process.argv[2];
const allowBetaCompatDiagnostics =
@ -289,6 +291,19 @@ function configure() {
const state = stateDir();
const cfgPath = configPath();
const cfg = fs.existsSync(cfgPath) ? readJson(cfgPath) : {};
if (process.env.OPENCLAW_CODEX_NPM_PLUGIN_AUDIT_IDENTITY === "1") {
cfg.logging = {
...cfg.logging,
audit: { ...cfg.logging?.audit, enabled: true, executionIdentity: true },
};
cfg.gateway = {
...cfg.gateway,
mode: "local",
bind: "loopback",
port: 18789,
auth: { mode: "token", token: randomUUID() },
};
}
cfg.plugins = {
...cfg.plugins,
enabled: true,
@ -819,6 +834,47 @@ function assertAgentTurn() {
});
}
function assertAudit() {
const marker = process.argv[3];
if (!marker) {
throw new Error("assert-audit requires a private reply marker");
}
const cliPath = process.env.OPENCLAW_E2E_CLI_BIN;
if (!cliPath) {
throw new Error("assert-audit requires the installed OPENCLAW_E2E_CLI_BIN");
}
// CLI runs mint independent run ids. This fresh state contains exactly the three
// completed local turns; select retained identity keys, never session/route guesses.
const database = new DatabaseSync(path.join(stateDir(), "state", "openclaw.sqlite"), {
readOnly: true,
});
let selectors;
try {
selectors = database
.prepare(
"SELECT run_id AS runId, execution_id AS executionId, context_id AS contextId FROM execution_identity_contexts ORDER BY created_at, execution_id LIMIT 4",
)
.all();
} finally {
database.close();
}
const result = inspectCodexAudit({
selectors,
expectedExecutions: 3,
privateValues: [marker],
query: (args) =>
JSON.parse(
execFileSync(process.execPath, [cliPath, "audit", ...args, "--json"], {
encoding: "utf8",
timeout: 120_000,
maxBuffer: MAX_TEXT_FILE_BYTES,
stdio: ["ignore", "pipe", "pipe"],
}),
),
});
console.log(`codex_audit_identity: ${JSON.stringify(result)}`);
}
function assertFollowthrough() {
const progressMarker = process.argv[3];
const completeMarker = process.argv[4];
@ -929,6 +985,7 @@ const commands = {
"print-codex-bin": printCodexBin,
"assert-preflight": assertPreflight,
"assert-agent-turn": assertAgentTurn,
"assert-audit": assertAudit,
"assert-followthrough": assertFollowthrough,
"assert-uninstalled": assertUninstalled,
"assert-agent-error": assertAgentError,

View file

@ -0,0 +1,104 @@
// Inspect the package Gateway's public audit surface after direct-local Codex turns.
import assert from "node:assert/strict";
const HMAC_REF = /^hmac-sha256:v1:[a-f0-9]{32}:[a-f0-9]{64}$/u;
export function inspectCodexAudit({ query, selectors, expectedExecutions, privateValues }) {
const inspect = (...args) => {
const result = query(args);
const encoded = JSON.stringify(result);
for (const value of privateValues) {
assert(!encoded.includes(value), "Codex audit exposed private workspace content");
}
assert(!Object.hasOwn(result, "decisions"), "Codex audit exposed raw decision receipts");
return result;
};
assert.equal(selectors.length, expectedExecutions, "Codex turns did not retain three identities");
const executionIds = new Set();
const contextIds = new Set();
for (const candidate of selectors) {
const { runId } = candidate;
assert.equal(typeof runId, "string");
assert(runId.length > 0);
assert.equal(typeof candidate.executionId, "string");
assert(candidate.executionId.length > 0);
const discovery = inspect("--run", runId, "--explain", "--limit", "50");
assert.equal(discovery.run?.runId, runId, "Codex audit discovered the wrong run");
assert.equal(discovery.run?.status, "known", "Codex audit run was not retained");
assert.equal(
discovery.identity?.state,
"present",
"Codex run discovery did not select its execution",
);
assert.equal(discovery.identity.context.executionId, candidate.executionId);
assert.equal(discovery.identity.context.contextId, candidate.contextId);
assert.equal(
discovery.nextExecutionCursor,
undefined,
"Codex execution discovery was truncated",
);
const exact = inspect("--execution", candidate.executionId, "--explain", "--limit", "100");
assert.equal(exact.schemaVersion, 1);
assert.equal(exact.run?.status, "known");
assert.equal(exact.run?.runId, runId);
assert.equal(exact.run?.executionId, candidate.executionId);
assert.equal(exact.identity?.state, "present", "Codex execution identity is missing");
const context = exact.identity.context;
assert.equal(context.runId, runId);
assert.equal(context.executionId, candidate.executionId);
assert.equal(context.contextId, candidate.contextId);
assert.equal(typeof context.contextId, "string");
assert(context.contextId.length > 0);
assert.deepEqual(context.invoker, { state: "absent" }, "local CLI must not invent a person");
assert.equal(context.ingress?.kind, "local-cli");
assert.equal(context.ingress?.state, "present");
assert.equal(context.agentPrincipal?.kind, "agent");
assert.equal(context.agentPrincipal?.principalRef, "main");
assert.equal(context.trustDomain?.state, "present");
assert.equal(context.trustDomain?.kind, "gateway-cell");
assert.match(context.trustDomain.domainRef, HMAC_REF);
assert.equal(context.agentPrincipal.domainRef, context.trustDomain.domainRef);
assert.equal(context.agentDefinition?.definitionRef, "main");
assert.equal(context.runtimeInstance?.kind, "plugin-harness");
assert.equal(context.runtimeInstance?.state, "present");
assert.match(context.runtimeInstance.runtimeRef, HMAC_REF);
assert.equal(context.coverageState, "unattributed");
assert.deepEqual(context.applicableGrants, []);
assert(
context.assurance?.some(
(evidence) =>
evidence.kind === "runtime-binding" && evidence.strength === "boundary-verified",
),
"Codex runtime binding assurance is missing",
);
for (const evidence of context.assurance) {
assert.match(evidence.evidenceRef, HMAC_REF);
}
const admissions = exact.decisionDisplays?.filter(
(receipt) =>
receipt.provenance?.state === "verified" && receipt.provenance.producer === "run-admission",
);
assert.equal(admissions?.length, 1, "Codex admission receipt is missing or duplicated");
const admission = admissions[0];
assert.deepEqual(admission.action.family, "run");
assert.deepEqual(admission.action.operation, "admission");
assert.equal(admission.decision?.outcome, "not-applicable");
assert.equal(admission.decision?.reasonCode, "run_admission_identity_not_evaluated");
assert.equal(admission.enforcement?.coverageState, "unattributed");
assert.equal(admission.enforcement?.grantCount, 0);
assert.equal(admission.enforcement?.policyCount, 0);
assert.equal(exact.nextDecisionCursor, undefined, "Codex decision inspection was truncated");
assert(
exact.decisionDisplays.every(
(receipt) => receipt.provenance?.producer !== "operator-approval",
),
"full-access Codex execution must not manufacture operator approval",
);
executionIds.add(context.executionId);
contextIds.add(context.contextId);
}
assert.equal(executionIds.size, expectedExecutions, "Codex turns reused an execution id");
assert.equal(contextIds.size, expectedExecutions, "Codex turns reused a context id");
return { executionCount: executionIds.size };
}

View file

@ -0,0 +1,156 @@
// Installed-package identity proof; reads only the scenario's isolated audit state.
import assert from "node:assert/strict";
import fs from "node:fs";
import path from "node:path";
import { DatabaseSync } from "node:sqlite";
import { fileURLToPath } from "node:url";
const PSEUDONYM = /^hmac-sha256:v1:[a-f0-9]{32}:[a-f0-9]{64}$/u;
export function readIdentityRows(stateDir) {
const file = path.join(stateDir, "state", "openclaw.sqlite");
if (!fs.existsSync(file)) {
return [];
}
const db = new DatabaseSync(file, { readOnly: true });
try {
if (
!db
.prepare("SELECT 1 FROM sqlite_schema WHERE name = ? AND type = 'table'")
.get("execution_identity_contexts")
) {
return [];
}
// Two rows suffice to reject cross-test contamination or duplicate admission.
return db
.prepare("SELECT context_json FROM execution_identity_contexts LIMIT 2")
.all()
.map((row) => row.context_json);
} finally {
db.close();
}
}
function assertIdentityPrivacy(text, needles) {
for (const needle of needles) {
if (needle && text.includes(needle)) {
throw new Error("execution identity exposed private fixture data");
}
}
}
export function assertIdentityProjection(result, persisted, needles) {
assertIdentityPrivacy(JSON.stringify(result), needles);
assertIdentityPrivacy(persisted, needles);
const persistedContext = JSON.parse(persisted);
const expectedRunId = persistedContext.runId;
assert.ok(
typeof expectedRunId === "string" && expectedRunId.length > 0,
"missing persisted run id",
);
assert.equal(result.identity?.state, "present", "installed CLI omitted execution identity");
assert.equal(result.run?.runId, expectedRunId, "installed CLI selected another run");
assert.equal(Object.hasOwn(result, "decisions"), false, "private receipts reached the CLI");
const context = result.identity.context;
assert.equal(context.runId, expectedRunId);
assert.equal(context.schemaVersion, 1);
for (const id of [context.contextId, context.executionId]) {
assert.ok(typeof id === "string" && id.length > 0, "missing opaque identity id");
}
assert.equal(result.run.executionId, context.executionId);
assert.deepEqual(context.ingress, {
kind: "local-cli",
boundary: "agent-command.local",
state: "present",
});
assert.deepEqual(context.invoker, { state: "absent" });
assert.equal(context.coverageState, "unattributed");
assert.equal(context.agentPrincipal?.principalRef, "main");
assert.equal(context.agentDefinition?.definitionRef, "main");
assert.equal(context.representedSubject, undefined);
assert.equal(context.sponsor, undefined);
assert.equal(context.lineage, undefined);
assert.match(context.trustDomain?.domainRef, PSEUDONYM);
assert.equal(context.agentPrincipal.domainRef, context.trustDomain.domainRef);
assert.match(context.runtimeInstance?.runtimeRef, PSEUDONYM);
for (const item of context.assurance) {
assert.match(item.evidenceRef, PSEUDONYM);
}
for (const grant of context.applicableGrants) {
assert.match(grant.grantRef, PSEUDONYM);
}
const admission = result.decisionDisplays?.find(
(display) =>
display.provenance?.state === "verified" && display.provenance.producer === "run-admission",
);
assert.equal(admission?.decision?.outcome, "not-applicable");
assert.equal(admission?.decision?.reasonCode, "run_admission_identity_not_evaluated");
assert.notEqual(admission?.enforcement?.coverageState, "enforced");
assert.equal(JSON.stringify(context), persisted, "CLI context differs from persisted bytes");
return context.executionId;
}
function isolatedStateDir() {
const home = process.env.OPENCLAW_TEST_STATE_HOME;
assert.ok(home, "missing isolated test HOME");
assert.equal(process.env.HOME, home);
assert.equal(process.env.OPENCLAW_HOME, home);
const stateDir = path.join(home, ".openclaw");
assert.equal(process.env.OPENCLAW_STATE_DIR, stateDir);
assert.equal(process.env.OPENCLAW_CONFIG_PATH, path.join(stateDir, "openclaw.json"));
return stateDir;
}
function main() {
const [command, file, beforeFile] = process.argv.slice(2);
const stateDir = isolatedStateDir();
if (command === "clean-home") {
assert.deepEqual(fs.readdirSync(stateDir), [], "package proof inherited existing state");
return;
}
const rows = readIdentityRows(stateDir);
if (command === "empty") {
assert.equal(rows.length, 0, "identity state existed before the opted-in turn");
return;
}
if (command === "run-id") {
assert.equal(rows.length, 1, "expected exactly one admitted execution identity");
const runId = JSON.parse(rows[0]).runId;
assert.ok(typeof runId === "string" && runId.length > 0, "missing persisted run id");
process.stdout.write(runId);
return;
}
assert.equal(command, "verify");
assert.equal(rows.length, 1, "expected exactly one admitted execution identity");
const cfg = JSON.parse(fs.readFileSync(process.env.OPENCLAW_CONFIG_PATH, "utf8"));
const channel = cfg.channels?.[process.env.OPENCLAW_NPM_ONBOARD_CHANNEL];
const needles = [
process.env.HOME,
stateDir,
process.env.OPENCLAW_TEST_WORKSPACE_DIR,
process.env.OPENAI_API_KEY,
process.env.OPENCLAW_GATEWAY_TOKEN,
channel?.token,
channel?.botToken,
channel?.appToken,
cfg.models?.providers?.openai?.baseUrl,
process.env.SUCCESS_MARKER,
"Return the success marker from the test server.",
];
const result = JSON.parse(fs.readFileSync(file, "utf8"));
const executionId = assertIdentityProjection(result, rows[0], needles);
if (beforeFile) {
const before = JSON.parse(fs.readFileSync(beforeFile, "utf8"));
assertIdentityProjection(before, rows[0], needles);
assert.equal(
JSON.stringify(result.identity.context),
JSON.stringify(before.identity.context),
"execution identity changed across Gateway restart",
);
}
process.stdout.write(executionId);
}
if (process.argv[1] && fileURLToPath(import.meta.url) === path.resolve(process.argv[1])) {
main();
}

View file

@ -17,6 +17,9 @@ TARGET_ROOT_DIR="$(cd "${OPENCLAW_DOCKER_E2E_REPO_ROOT:-$ROOT_DIR}" && pwd)"
ONBOARD_ASSERTIONS="$(openclaw_resolve_frozen_target_file "$TARGET_ROOT_DIR" \
scripts/e2e/lib/npm-onboard-channel-agent/assertions.mjs \
"$ROOT_DIR/scripts/e2e/lib/npm-onboard-channel-agent/assertions.mjs")"
ONBOARD_IDENTITY_ASSERTIONS="$(openclaw_resolve_frozen_target_file "$TARGET_ROOT_DIR" \
scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs \
"$ROOT_DIR/scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs")"
# The assertion and its config producer are one target-owned contract; mixing
# generations can make a valid frozen package appear to change its default model.
ONBOARD_MOCK_OPENAI_CONFIG="$(openclaw_resolve_frozen_target_file "$TARGET_ROOT_DIR" \
@ -84,6 +87,7 @@ if ! docker_e2e_run_with_harness \
-e "OPENCLAW_NPM_ONBOARD_STATUS_TEXT_MAX_BYTES=$STATUS_TEXT_MAX_BYTES" \
-e "OPENCLAW_TEST_STATE_SCRIPT_B64=$OPENCLAW_TEST_STATE_SCRIPT_B64" \
-v "$ONBOARD_ASSERTIONS:/app/scripts/e2e/lib/npm-onboard-channel-agent/assertions.mjs:ro" \
-v "$ONBOARD_IDENTITY_ASSERTIONS:/app/scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs:ro" \
-v "$ONBOARD_MOCK_OPENAI_CONFIG:/app/scripts/e2e/lib/fixtures/mock-openai-config.mjs:ro" \
"${DOCKER_E2E_PACKAGE_ARGS[@]}" \
-i "$IMAGE_NAME" bash -s >"$run_log" 2>&1 <<'EOF'; then
@ -92,6 +96,9 @@ set -Eeuo pipefail
source scripts/lib/openclaw-e2e-instance.sh
source scripts/e2e/lib/prepublish-plugin-registry.sh
openclaw_e2e_eval_test_state_from_b64 "${OPENCLAW_TEST_STATE_SCRIPT_B64:?missing OPENCLAW_TEST_STATE_SCRIPT_B64}"
export OPENCLAW_TEST_STATE_HOME
identity_assertions=scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs
node "$identity_assertions" clean-home
export NPM_CONFIG_PREFIX="$HOME/.npm-global"
export PATH="$NPM_CONFIG_PREFIX/bin:$PATH"
export OPENAI_API_KEY="sk-openclaw-npm-onboard-e2e"
@ -106,6 +113,7 @@ MOCK_REQUEST_LOG="$scenario_tmp/mock-openai-requests.jsonl"
export SUCCESS_MARKER MOCK_REQUEST_LOG
mock_pid=""
plugin_registry_pid=""
gateway_pid=""
case "$CHANNEL" in
telegram)
@ -134,6 +142,7 @@ case "$CHANNEL" in
esac
cleanup() {
openclaw_e2e_stop_process "${gateway_pid:-}"
openclaw_e2e_stop_process "${mock_pid:-}"
openclaw_e2e_stop_process "${plugin_registry_pid:-}"
rm -rf "$scenario_tmp"
@ -157,6 +166,8 @@ dump_debug_logs() {
/tmp/openclaw-agent.err \
/tmp/openclaw-agent.json \
/tmp/openclaw-mock-openai.log \
"$scenario_tmp/gateway-before.log" \
"$scenario_tmp/gateway-after.log" \
"$MOCK_REQUEST_LOG" \
"$OPENCLAW_HOME/.openclaw/openclaw.json" \
"$OPENCLAW_HOME/.openclaw/agents/main/agent/auth-profiles.json"
@ -254,6 +265,9 @@ fi
node scripts/e2e/lib/npm-onboard-channel-agent/assertions.mjs configure-mock-model "$MOCK_PORT"
node scripts/e2e/lib/npm-onboard-channel-agent/assertions.mjs assert-mock-model-config "$MOCK_PORT"
node "$identity_assertions" empty
openclaw config set logging.audit.enabled true
openclaw config set logging.audit.executionIdentity true
echo "Running local agent turn against mocked OpenAI..."
if openclaw agent --local \
@ -272,6 +286,23 @@ if [ "$agent_status" -ne 0 ]; then
fi
node scripts/e2e/lib/npm-onboard-channel-agent/assertions.mjs assert-agent-turn "$SUCCESS_MARKER" "$MOCK_REQUEST_LOG"
run_id="$(node "$identity_assertions" run-id)"
# The local CLI flushes its audit writer before exiting. Inspect once after that
# boundary; polling a read-only inspector would hide lost admission writes.
entry="$(openclaw_e2e_package_entrypoint "$package_root")"
export OPENCLAW_SKIP_CHANNELS=1 OPENCLAW_SKIP_GMAIL_WATCHER=1 OPENCLAW_SKIP_CANVAS_HOST=1
gateway_pid="$(openclaw_e2e_start_gateway "$entry" "$PORT" "$scenario_tmp/gateway-before.log")"
openclaw_e2e_wait_gateway_ready "$gateway_pid" "$scenario_tmp/gateway-before.log" 300 "$PORT"
openclaw audit --run "$run_id" --explain --json >"$scenario_tmp/identity-before.json"
execution_id="$(node "$identity_assertions" verify "$scenario_tmp/identity-before.json")"
openclaw_e2e_stop_process "$gateway_pid"
gateway_pid=""
gateway_pid="$(openclaw_e2e_start_gateway "$entry" "$PORT" "$scenario_tmp/gateway-after.log")"
openclaw_e2e_wait_gateway_ready "$gateway_pid" "$scenario_tmp/gateway-after.log" 300 "$PORT"
openclaw audit --execution "$execution_id" --explain --json >"$scenario_tmp/identity-after.json"
node "$identity_assertions" verify "$scenario_tmp/identity-after.json" "$scenario_tmp/identity-before.json" >/dev/null
echo "Installed CLI execution identity survived Gateway restart with private fixture data omitted."
echo "npm tarball onboard/channel/agent Docker E2E passed for $CHANNEL"
EOF

View file

@ -0,0 +1,248 @@
import type { DatabaseSync } from "node:sqlite";
import { afterAll, beforeAll, describe, expect, it, vi } from "vitest";
import { useAutoCleanupTempDirTracker } from "../../test/helpers/temp-dir.js";
import {
closeOpenClawStateDatabaseAsync,
closeOpenClawStateDatabaseForTest,
openOpenClawStateDatabase,
} from "../state/openclaw-state-db.js";
import { recordAuditEventInDatabase } from "./audit-event-store.js";
import { recordExecutionDecisionFactInDatabase } from "./execution-decision-facts.js";
import { receipt } from "./execution-decision-facts.test-support.js";
import { ExecutionDecisionCursorError } from "./execution-decision-receipts.js";
import { createExecutionIdentityAdmissionToken } from "./execution-identity-admission.js";
import { inspectExecutionIdentityRun } from "./execution-identity-context.js";
import {
prepareExecutionIdentityContextAtAdmission,
recordDeniedApprovalForRun,
} from "./execution-identity.test-support.js";
import { bindExecutionOwnerLifecycleMetadata } from "./execution-owner-lifecycle-binding-store.js";
const tempDirs = useAutoCleanupTempDirTracker(afterAll);
const now = 200;
const stages = [
{ prefix: "a", owner: "operator_approvals" },
{ prefix: "m", owner: "audit_events" },
{ prefix: "g", owner: "tool-policy" },
{ prefix: "c", owner: "cron_run_receipts" },
{ prefix: "t", owner: "task_runs" },
{ prefix: "f", owner: "flow_runs" },
] as const;
type Stage = (typeof stages)[number]["prefix"];
let options: { env: { OPENCLAW_STATE_DIR: string } };
let db: DatabaseSync;
function inspect(decisionCursor: string, executionId = "execution-local") {
return inspectExecutionIdentityRun(
{ executionId, decisionCursor, decisionLimit: 1 },
{ ...options, now },
);
}
beforeAll(async () => {
vi.spyOn(Date, "now").mockReturnValue(now);
options = { env: { OPENCLAW_STATE_DIR: tempDirs.make("decision-cursors-") } };
const database = openOpenClawStateDatabase(options);
db = database.db;
for (const scope of ["local", "foreign", "sibling"]) {
const context = prepareExecutionIdentityContextAtAdmission(
{
runId: scope === "sibling" ? "run-local" : `run-${scope}`,
agentId: "main",
ingress: { kind: "local-cli", boundary: "agent-command.local", state: "present" },
runtime: { kind: "embedded" },
},
{
...options,
contextId: `context-${scope}`,
executionId: `execution-${scope}`,
runtimeInstanceId: `runtime-${scope}`,
now: 50,
},
);
const token = createExecutionIdentityAdmissionToken(context.runId, {
contextId: context.contextId,
executionId: context.executionId,
now: context.createdAt,
});
for (const index of [1, 2]) {
const id = `${scope}-${index}`;
await recordDeniedApprovalForRun(context.runId, options, `approval-${id}`, context);
expect(
recordAuditEventInDatabase(
{
sourceId: `message-${id}`,
sourceSequence: 1,
occurredAt: now,
kind: "message",
action: "message.outbound.finished",
status: "succeeded",
outcome: "sent",
actorType: "agent",
actorId: "main",
agentId: "main",
runId: context.runId,
executionIdentityToken: token,
direction: "outbound",
channel: "qa-channel",
conversationKind: "direct",
resultCount: 1,
},
{ ...options, database },
),
).toBeDefined();
expect(
recordExecutionDecisionFactInDatabase(
{
...receipt(`generic-${id}`, now),
contextId: context.contextId,
executionId: context.executionId,
runId: context.runId,
},
{ ...options, database, now },
),
).toBe("inserted");
db.prepare(`INSERT INTO cron_run_receipts (
receipt_id, store_key, job_id, config_revision, agent_id, request_run_id,
status, owner_pid, started_at_ms, finished_at_ms
) VALUES (?, 'default', ?, 'revision-1', 'main', ?, 'ok', 1, ?, ?)`).run(
`cron-${id}`,
`job-${id}`,
context.runId,
now,
now,
);
db.prepare(`INSERT INTO task_runs (
task_id, runtime, owner_key, scope_kind, task, status, delivery_status,
notify_policy, created_at
) VALUES (?, 'cron', 'fixture-owner', 'system', 'fixture task', 'succeeded',
'not_applicable', 'silent', ?)`).run(`task-${id}`, now);
db.prepare(`INSERT INTO flow_runs (
flow_id, owner_key, status, notify_policy, goal, created_at, updated_at
) VALUES (?, 'fixture-owner', 'succeeded', 'silent', 'fixture goal', ?, ?)`).run(
`flow-${id}`,
now,
now,
);
for (const ownerKind of ["cron", "task", "flow"] as const) {
expect(
bindExecutionOwnerLifecycleMetadata({
db,
ownerKind,
ownerId: `${ownerKind}-${id}`,
binding: context,
}),
).toBe("bound");
}
}
}
});
afterAll(async () => {
await closeOpenClawStateDatabaseAsync();
closeOpenClawStateDatabaseForTest();
vi.restoreAllMocks();
});
const removeAnchor: Record<Stage, () => void> = {
a: () => {
db.prepare("DELETE FROM operator_approvals WHERE approval_id = ?").run("approval-local-1");
},
m: () => {
db.prepare(
"DELETE FROM audit_events WHERE sequence = (SELECT MIN(sequence) FROM audit_events)",
).run();
},
g: () => {
db.prepare("DELETE FROM execution_decision_facts WHERE receipt_id = ?").run("generic-local-1");
},
c: () => {
db.prepare("DELETE FROM cron_run_receipts WHERE receipt_id = ?").run("cron-local-1");
},
t: () => {
db.prepare("DELETE FROM task_runs WHERE task_id = ?").run("task-local-1");
},
f: () => {
db.prepare("DELETE FROM flow_runs WHERE flow_id = ?").run("flow-local-1");
},
};
describe("inspection decision cursor rejection matrix", () => {
it.each(stages)(
"rejects malformed and unretained $prefix anchors through inspection",
async ({ prefix, owner }) => {
const first = await inspect(`${prefix}:0:0`);
expect(first.decisions).toHaveLength(1);
expect(first.decisions[0]?.source.owner).toBe(owner);
const cursor = first.nextDecisionCursor;
expect(cursor).toMatch(new RegExp(`^${prefix}:${now}:[1-9]\\d*$`));
if (!cursor) {
throw new Error("Expected an owner cursor with another row at the same timestamp");
}
const second = await inspect(cursor);
expect(second.decisions).toHaveLength(1);
expect(second.decisions[0]?.source.owner).toBe(owner);
expect(second.decisions[0]?.receiptId).not.toBe(first.decisions[0]?.receiptId);
for (const suffix of [
"-1:1",
"1:-1",
"1.5:1",
"1:1.5",
"01:1",
"1:01",
"9007199254740992:1",
"1:9007199254740992",
"1",
"1:1:extra",
]) {
await expect(inspect(`${prefix}:${suffix}`)).rejects.toThrow(
"invalid execution decision cursor",
);
}
const rowId = cursor.split(":")[2];
for (const invalid of [
`${prefix}:${now + 1}:${rowId}`,
`${prefix}:${now}:9007199254740991`,
]) {
await expect(inspect(invalid)).rejects.toThrow(
"decision cursor is no longer retained; restart inspection without --cursor",
);
}
await expect(inspect(cursor, "execution-foreign")).rejects.toBeInstanceOf(
ExecutionDecisionCursorError,
);
await expect(inspect(cursor, "execution-foreign")).rejects.toThrow(
"decision cursor is no longer retained; restart inspection without --cursor",
);
if (prefix === "a") {
// Approvals page the run correlation and expose a mismatched execution
// only as unknown evidence; the other owners scope their cursor itself.
const sibling = await inspect(cursor, "execution-sibling");
expect(sibling.decisions).toHaveLength(1);
expect(sibling.decisions[0]).toMatchObject({
contextId: "context-sibling",
executionId: "execution-sibling",
runId: "run-local",
decision: {
outcome: "unknown",
reasonCode: "operator_approval_execution_link_mismatch",
},
enforcement: { coverageState: "unknown", grantRefs: [] },
missingEvidence: ["decision.execution_link"],
});
} else {
await expect(inspect(cursor, "execution-sibling")).rejects.toThrow(
"decision cursor is no longer retained; restart inspection without --cursor",
);
}
removeAnchor[prefix]();
await expect(inspect(cursor)).rejects.toThrow(
"decision cursor is no longer retained; restart inspection without --cursor",
);
expect((await inspect(`${prefix}:0:0`)).decisions[0]?.receiptId).toBe(
second.decisions[0]?.receiptId,
);
},
);
});

View file

@ -1,6 +1,7 @@
import fs from "node:fs";
import { afterEach, describe, expect, it } from "vitest";
import { useAutoCleanupTempDirTracker } from "../../test/helpers/temp-dir.js";
import { assertSqliteSchemaContains } from "../infra/sqlite-schema-contract.js";
import {
closeOpenClawStateDatabaseAsync,
closeOpenClawStateDatabaseForTest,
@ -14,6 +15,11 @@ import {
resolveOperatorApproval,
} from "./operator-approval-store.js";
import { insertOperatorApprovalInDatabase as insertOperatorApprovalNative } from "./operator-approval-store.kernel.js";
import {
getOperatorApprovalDetailed as getOlderOperatorApproval,
OLDER_OPERATOR_APPROVAL_SCHEMA_SQL,
resolveOperatorApproval as resolveOlderOperatorApproval,
} from "./operator-approval-store.older-reader.test-support.js";
type NewOperatorApproval = Parameters<typeof insertOperatorApproval>[0]["approval"];
const tempDirs = useAutoCleanupTempDirTracker(afterEach);
@ -259,4 +265,86 @@ describe("operator approval execution identity", () => {
});
}
});
it("preserves companion identity through an older approval reader write and candidate reopen", async () => {
const options = databaseOptions();
await insertOperatorApproval({
approval: approval("older-reader", token()),
databaseOptions: options,
});
const candidate = openOpenClawStateDatabase(options);
const version = candidate.db.prepare("PRAGMA user_version").get();
const binding = candidate.db
.prepare("SELECT * FROM operator_approval_execution_identities")
.all();
expect(binding).toEqual([
{
approval_id: "older-reader",
source_context_id: "context-1",
source_execution_id: "execution-1",
},
]);
await closeOpenClawStateDatabaseAsync();
closeOpenClawStateDatabaseForTest();
// The pinned reader predates companion identities. Its original decoder and
// decision transition reopen through the current shared database owner.
expect(
getOlderOperatorApproval({ id: "older-reader", nowMs: 2_000, databaseOptions: options }),
).toMatchObject({ outcome: "found", record: { status: "pending", decision: null } });
const older = openOpenClawStateDatabase(options);
assertSqliteSchemaContains(older.db, older.path, OLDER_OPERATOR_APPROVAL_SCHEMA_SQL);
expect(
resolveOlderOperatorApproval({
id: "older-reader",
decision: "allow-once",
resolver: { kind: "device", id: "older-reviewer" },
expectedKind: "exec",
runtimeEpoch: "runtime-a",
nowMs: 2_000,
databaseOptions: options,
}),
).toMatchObject({ outcome: "resolved", record: { decision: "allow-once" } });
expect(
getOlderOperatorApproval({ id: "older-reader", nowMs: 2_000, databaseOptions: options }),
).toMatchObject({ outcome: "found", record: { status: "allowed", decision: "allow-once" } });
expect(older.db.prepare("PRAGMA user_version").get()).toEqual(version);
await closeOpenClawStateDatabaseAsync();
closeOpenClawStateDatabaseForTest();
expect(
await getOperatorApprovalDetailed({
id: "older-reader",
nowMs: 3_000,
databaseOptions: options,
}),
).toMatchObject({
outcome: "found",
record: { decision: "allow-once", resolver: { id: "older-reviewer" } },
});
expect(
await consumeOperatorApprovalAllowOnce({
id: "older-reader",
consumerId: "candidate-consumer",
expectedKind: "exec",
runtimeEpoch: "runtime-a",
nowMs: 3_000,
databaseOptions: options,
}),
).toMatchObject({ outcome: "consumed" });
expect(
await getOperatorApprovalDetailed({
id: "older-reader",
nowMs: 3_000,
databaseOptions: options,
}),
).toMatchObject({ outcome: "found", record: { consumedBy: "candidate-consumer" } });
const reopened = openOpenClawStateDatabase(options).db;
expect(reopened.prepare("SELECT * FROM operator_approval_execution_identities").all()).toEqual(
binding,
);
expect(reopened.prepare("PRAGMA user_version").get()).toEqual(version);
expect(reopened.prepare("PRAGMA integrity_check").all()).toEqual([{ integrity_check: "ok" }]);
expect(reopened.prepare("PRAGMA foreign_key_check").all()).toEqual([]);
});
});

View file

@ -0,0 +1,613 @@
import { normalizeNullableString as normalizeString } from "@openclaw/normalization-core/string-coerce";
import { validateApprovalPresentation } from "../../packages/gateway-protocol/src/approval-result-validators.js";
import { isWellFormedApprovalId } from "../../packages/gateway-protocol/src/schema/approval-id.js";
// Frozen pre-companion approval reader from aa7cf44c75a123a2724f20b05cd10d66cf7e65f3.
// get/resolve and their validation/query helpers below are unmodified historical source.
// Database, protocol, and nullable-string normalization dependencies are current;
// normalization is equivalent for the historical string/null/undefined input domain.
// This qualifies the older reader, not an older binary's schema-version admission.
import type { ApprovalPresentation } from "../../packages/gateway-protocol/src/schema/approvals.js";
import {
buildApprovalResolutionRef,
isApprovalResolutionRef,
} from "../infra/approval-resolution-ref.js";
import {
executeSqliteQuerySync,
executeSqliteQueryTakeFirstSync,
getNodeSqliteKysely,
} from "../infra/kysely-sync.js";
import {
type OpenClawStateDatabaseOptions,
openOpenClawStateDatabase,
runOpenClawStateWriteTransaction,
} from "../state/openclaw-state-db.js";
import type {
GetOperatorApprovalResult,
OperatorApprovalDatabase,
OperatorApprovalDecision,
OperatorApprovalKind,
OperatorApprovalRecord,
OperatorApprovalResolver,
OperatorApprovalResolverKind,
OperatorApprovalRow,
OperatorApprovalStatus,
OperatorApprovalTerminalReason,
ResolveOperatorApprovalResult,
} from "./operator-approval-store.types.js";
const OPERATOR_APPROVAL_MAX_AUDIENCE_SESSION_KEYS = 64;
const OPERATOR_APPROVAL_DECISIONS = new Set<OperatorApprovalDecision>([
"allow-once",
"allow-always",
"deny",
]);
const OPERATOR_APPROVAL_KINDS = new Set<OperatorApprovalKind>(["exec", "plugin", "system-agent"]);
const OPERATOR_APPROVAL_STATUSES = new Set<OperatorApprovalStatus>([
"pending",
"allowed",
"denied",
"expired",
"cancelled",
]);
const OPERATOR_APPROVAL_TERMINAL_REASONS = new Set<OperatorApprovalTerminalReason>([
"user",
"timeout",
"malformed-verdict",
"no-route",
"run-aborted",
"gateway-restart",
"storage-corrupt",
]);
const OPERATOR_APPROVAL_RESOLVER_KINDS = new Set<OperatorApprovalResolverKind>([
"device",
"channel",
"runtime",
"system",
]);
function parseApprovalPresentation(raw: string): ApprovalPresentation | null {
try {
const value: unknown = JSON.parse(raw);
return validateApprovalPresentation(value) ? value : null;
} catch {
return null;
}
}
function parseStringArray(raw: string): string[] | null {
try {
const value: unknown = JSON.parse(raw);
if (
!Array.isArray(value) ||
value.some((entry) => typeof entry !== "string" || !entry.trim())
) {
return null;
}
return value as string[];
} catch {
return null;
}
}
function requireString(value: string, label: string): string {
const normalized = normalizeString(value);
if (!normalized) {
throw new Error(`${label} must not be empty`);
}
return normalized;
}
function requireApprovalId(value: string): string {
if (!isWellFormedApprovalId(value)) {
throw new Error("operator approval id must be non-empty, well-formed Unicode, and not . or ..");
}
return value;
}
function isValidTimestamp(value: number): boolean {
return Number.isSafeInteger(value) && value >= 0;
}
function clampAuditTimestamp(nowMs: number, ...minimums: Array<number | null>): number {
return Math.max(nowMs, ...minimums.filter((value): value is number => value !== null));
}
function hasValidLifecycleTuple(params: {
row: OperatorApprovalRow;
status: OperatorApprovalStatus;
decision: OperatorApprovalDecision | null;
terminalReason: OperatorApprovalTerminalReason | null;
resolverKind: OperatorApprovalResolverKind | null;
}): boolean {
const { row, status, decision, terminalReason, resolverKind } = params;
const noConsumption = row.consumed_at_ms === null && row.consumed_by === null;
if (status === "pending") {
return (
decision === null &&
terminalReason === null &&
row.resolved_at_ms === null &&
resolverKind === null &&
row.resolver_id === null &&
noConsumption
);
}
if (row.resolved_at_ms === null || resolverKind === null) {
return false;
}
if (status === "allowed") {
const validConsumption =
decision === "allow-once"
? noConsumption || (row.consumed_at_ms !== null && Boolean(row.consumed_by?.trim()))
: noConsumption;
return (
(decision === "allow-once" || decision === "allow-always") &&
terminalReason === "user" &&
validConsumption
);
}
if (decision !== "deny" || !noConsumption) {
return false;
}
if (status === "denied") {
return (
terminalReason === "user" ||
terminalReason === "malformed-verdict" ||
terminalReason === "no-route" ||
terminalReason === "storage-corrupt"
);
}
if (status === "expired") {
return terminalReason === "timeout";
}
return (
status === "cancelled" &&
(terminalReason === "run-aborted" || terminalReason === "gateway-restart")
);
}
function decodeOperatorApprovalRow(row: OperatorApprovalRow): OperatorApprovalRecord | null {
const presentation = parseApprovalPresentation(row.presentation_json);
const reviewerDeviceIds = parseStringArray(row.reviewer_device_ids_json);
const audienceSessionKeys = parseStringArray(row.audience_session_keys_json);
const kind = row.kind as OperatorApprovalKind;
const status = row.status as OperatorApprovalStatus;
const decision = row.decision as OperatorApprovalDecision | null;
const terminalReason = row.terminal_reason as OperatorApprovalTerminalReason | null;
const resolverKind = row.resolver_kind as OperatorApprovalResolverKind | null;
if (
!presentation ||
!isWellFormedApprovalId(row.approval_id) ||
!isApprovalResolutionRef(row.resolution_ref) ||
!reviewerDeviceIds ||
!audienceSessionKeys ||
audienceSessionKeys.length > OPERATOR_APPROVAL_MAX_AUDIENCE_SESSION_KEYS ||
!OPERATOR_APPROVAL_KINDS.has(kind) ||
!OPERATOR_APPROVAL_STATUSES.has(status) ||
!isValidTimestamp(row.created_at_ms) ||
!isValidTimestamp(row.expires_at_ms) ||
!isValidTimestamp(row.updated_at_ms) ||
row.expires_at_ms < row.created_at_ms ||
row.updated_at_ms < row.created_at_ms ||
(row.resolved_at_ms !== null &&
(!isValidTimestamp(row.resolved_at_ms) ||
row.resolved_at_ms < row.created_at_ms ||
row.resolved_at_ms > row.updated_at_ms)) ||
(row.consumed_at_ms !== null &&
(!isValidTimestamp(row.consumed_at_ms) ||
row.resolved_at_ms === null ||
row.consumed_at_ms < row.resolved_at_ms ||
row.consumed_at_ms > row.updated_at_ms)) ||
(row.requested_by_device_token_auth !== 0 && row.requested_by_device_token_auth !== 1) ||
(decision !== null && !OPERATOR_APPROVAL_DECISIONS.has(decision)) ||
(terminalReason !== null && !OPERATOR_APPROVAL_TERMINAL_REASONS.has(terminalReason)) ||
(resolverKind !== null && !OPERATOR_APPROVAL_RESOLVER_KINDS.has(resolverKind))
) {
return null;
}
if (
presentation.kind !== kind ||
row.resolution_ref !==
buildApprovalResolutionRef({ approvalId: row.approval_id, approvalKind: kind }) ||
!hasValidLifecycleTuple({ row, status, decision, terminalReason, resolverKind }) ||
(status === "allowed" &&
(!decision || !Array.prototype.includes.call(presentation.allowedDecisions, decision)))
) {
return null;
}
return {
id: row.approval_id,
resolutionRef: row.resolution_ref,
kind,
status,
presentation,
requester: {
deviceId: row.requested_by_device_id,
clientId: row.requested_by_client_id,
deviceTokenAuth: row.requested_by_device_token_auth === 1,
},
reviewerDeviceIds,
source: {
agentId: row.source_agent_id,
sessionKey: row.source_session_key,
sessionId: row.source_session_id,
runId: row.source_run_id,
toolCallId: row.source_tool_call_id,
toolName: row.source_tool_name,
},
audienceSessionKeys,
runtimeEpoch: row.runtime_epoch,
createdAtMs: row.created_at_ms,
expiresAtMs: row.expires_at_ms,
updatedAtMs: row.updated_at_ms,
decision,
terminalReason,
resolvedAtMs: row.resolved_at_ms,
resolver:
resolverKind === null
? null
: {
kind: resolverKind,
id: row.resolver_id,
},
consumedAtMs: row.consumed_at_ms,
consumedBy: row.consumed_by,
};
}
function selectOperatorApprovalRow(
database: ReturnType<typeof openOpenClawStateDatabase>,
id: string,
): OperatorApprovalRow | undefined {
const stateDb = getNodeSqliteKysely<OperatorApprovalDatabase>(database.db);
return executeSqliteQueryTakeFirstSync(
database.db,
stateDb.selectFrom("operator_approvals").selectAll().where("approval_id", "=", id),
);
}
function selectOperatorApprovalRowByLocator(
database: ReturnType<typeof openOpenClawStateDatabase>,
locator: string,
): OperatorApprovalRow | undefined {
const stateDb = getNodeSqliteKysely<OperatorApprovalDatabase>(database.db);
const rows = executeSqliteQuerySync(
database.db,
stateDb
.selectFrom("operator_approvals")
.selectAll()
.where((eb) => eb.or([eb("approval_id", "=", locator), eb("resolution_ref", "=", locator)]))
.limit(2),
).rows;
return rows.length === 1 ? rows[0] : undefined;
}
function matchesExpectedApprovalOwner(params: {
row: OperatorApprovalRow;
expectedKind?: OperatorApprovalKind;
runtimeEpoch?: string;
}): boolean {
return (
(params.expectedKind === undefined || params.row.kind === params.expectedKind) &&
(params.runtimeEpoch === undefined || params.row.runtime_epoch === params.runtimeEpoch)
);
}
function denyCorruptPendingRow(params: {
database: ReturnType<typeof openOpenClawStateDatabase>;
id: string;
nowMs: number;
createdAtMs: number;
}): void {
const auditTimestampMs = clampAuditTimestamp(params.nowMs, params.createdAtMs);
const stateDb = getNodeSqliteKysely<OperatorApprovalDatabase>(params.database.db);
executeSqliteQuerySync(
params.database.db,
stateDb
.updateTable("operator_approvals")
.set({
status: "denied",
decision: "deny",
terminal_reason: "storage-corrupt",
resolved_at_ms: auditTimestampMs,
resolver_kind: "system",
resolver_id: null,
updated_at_ms: auditTimestampMs,
})
.where("approval_id", "=", params.id)
.where("status", "=", "pending"),
);
}
function expirePendingRow(params: {
database: ReturnType<typeof openOpenClawStateDatabase>;
id: string;
nowMs: number;
createdAtMs: number;
}): OperatorApprovalRow | undefined {
const auditTimestampMs = clampAuditTimestamp(params.nowMs, params.createdAtMs);
const stateDb = getNodeSqliteKysely<OperatorApprovalDatabase>(params.database.db);
executeSqliteQuerySync(
params.database.db,
stateDb
.updateTable("operator_approvals")
.set({
status: "expired",
decision: "deny",
terminal_reason: "timeout",
resolved_at_ms: auditTimestampMs,
resolver_kind: "system",
resolver_id: null,
updated_at_ms: auditTimestampMs,
})
.where("approval_id", "=", params.id)
.where("status", "=", "pending")
.where("expires_at_ms", "<=", params.nowMs),
);
return selectOperatorApprovalRow(params.database, params.id);
}
function requireDecodedRecord(row: OperatorApprovalRow): OperatorApprovalRecord {
const record = decodeOperatorApprovalRow(row);
if (!record) {
throw new Error(`operator approval '${row.approval_id}' became corrupt during a transaction`);
}
return record;
}
export function getOperatorApprovalDetailed(params: {
id: string;
allowTransportRef?: boolean;
nowMs?: number;
databaseOptions?: OpenClawStateDatabaseOptions;
}): GetOperatorApprovalResult {
const locator = requireApprovalId(params.id);
return runOpenClawStateWriteTransaction((database) => {
const nowMs = params.nowMs ?? Date.now();
let row = params.allowTransportRef
? selectOperatorApprovalRowByLocator(database, locator)
: selectOperatorApprovalRow(database, locator);
if (!row) {
return { outcome: "not-found" };
}
const id = row.approval_id;
if (row.status === "pending" && row.expires_at_ms <= nowMs) {
row = expirePendingRow({ database, id, nowMs, createdAtMs: row.created_at_ms });
if (!row) {
return { outcome: "not-found" };
}
}
const record = decodeOperatorApprovalRow(row);
if (record) {
return { outcome: "found", record };
}
denyCorruptPendingRow({ database, id, nowMs, createdAtMs: row.created_at_ms });
return params.allowTransportRef ? { outcome: "corrupt", id } : { outcome: "corrupt" };
}, params.databaseOptions);
}
export function resolveOperatorApproval(params: {
id: string;
decision: OperatorApprovalDecision;
resolver: OperatorApprovalResolver;
expectedKind?: OperatorApprovalKind;
runtimeEpoch?: string;
nowMs?: number;
databaseOptions?: OpenClawStateDatabaseOptions;
}): ResolveOperatorApprovalResult {
const id = requireApprovalId(params.id);
const resolverId = normalizeString(params.resolver.id);
const runtimeEpoch =
params.runtimeEpoch === undefined
? undefined
: requireString(params.runtimeEpoch, "operator approval runtime epoch");
return runOpenClawStateWriteTransaction((database) => {
const nowMs = params.nowMs ?? Date.now();
let row = selectOperatorApprovalRow(database, id);
if (!row) {
return { outcome: "not-found" };
}
if (!matchesExpectedApprovalOwner({ row, expectedKind: params.expectedKind, runtimeEpoch })) {
return { outcome: "not-found" };
}
let record = decodeOperatorApprovalRow(row);
if (!record) {
denyCorruptPendingRow({ database, id, nowMs, createdAtMs: row.created_at_ms });
return { outcome: "corrupt" };
}
if (record.status !== "pending") {
return {
outcome: "already-resolved",
retry: record.decision === params.decision ? "same" : "conflict",
record,
};
}
if (record.expiresAtMs <= nowMs) {
row = expirePendingRow({ database, id, nowMs, createdAtMs: row.created_at_ms });
if (!row) {
return { outcome: "not-found" };
}
record = requireDecodedRecord(row);
return { outcome: "expired", record };
}
if (!Array.prototype.includes.call(record.presentation.allowedDecisions, params.decision)) {
return { outcome: "decision-not-allowed", record };
}
const auditTimestampMs = clampAuditTimestamp(nowMs, record.createdAtMs);
const stateDb = getNodeSqliteKysely<OperatorApprovalDatabase>(database.db);
let resolveQuery = stateDb
.updateTable("operator_approvals")
.set({
status: params.decision === "deny" ? "denied" : "allowed",
decision: params.decision,
terminal_reason: "user",
resolved_at_ms: auditTimestampMs,
resolver_kind: params.resolver.kind,
resolver_id: resolverId,
updated_at_ms: auditTimestampMs,
})
.where("approval_id", "=", id)
.where("status", "=", "pending")
.where("expires_at_ms", ">", nowMs);
if (params.expectedKind !== undefined) {
resolveQuery = resolveQuery.where("kind", "=", params.expectedKind);
}
if (runtimeEpoch !== undefined) {
resolveQuery = resolveQuery.where("runtime_epoch", "=", runtimeEpoch);
}
const result = executeSqliteQuerySync(database.db, resolveQuery);
row = selectOperatorApprovalRow(database, id);
if (!row) {
return { outcome: "not-found" };
}
record = requireDecodedRecord(row);
if (result.numAffectedRows === 1n) {
return { outcome: "resolved", record };
}
if (record.status === "pending" && record.expiresAtMs <= nowMs) {
const expiredRow = expirePendingRow({
database,
id,
nowMs,
createdAtMs: record.createdAtMs,
});
if (!expiredRow) {
return { outcome: "not-found" };
}
return { outcome: "expired", record: requireDecodedRecord(expiredRow) };
}
return {
outcome: "already-resolved",
retry: record.decision === params.decision ? "same" : "conflict",
record,
};
}, params.databaseOptions);
}
export const OLDER_OPERATOR_APPROVAL_SCHEMA_SQL = `
CREATE TABLE IF NOT EXISTS operator_approvals (
approval_id TEXT NOT NULL PRIMARY KEY CHECK (
length(approval_id) > 0 AND approval_id NOT IN ('.', '..')
),
resolution_ref TEXT NOT NULL CHECK (
length(resolution_ref) = 43 AND resolution_ref NOT GLOB '*[^A-Za-z0-9_-]*'
),
kind TEXT NOT NULL CHECK (kind IN ('exec', 'plugin', 'system-agent')),
status TEXT NOT NULL CHECK (status IN ('pending', 'allowed', 'denied', 'expired', 'cancelled')),
presentation_json TEXT NOT NULL,
requested_by_device_id TEXT,
requested_by_client_id TEXT,
requested_by_device_token_auth INTEGER NOT NULL DEFAULT 0,
reviewer_device_ids_json TEXT NOT NULL,
source_agent_id TEXT,
source_session_key TEXT,
source_session_id TEXT,
source_run_id TEXT,
source_tool_call_id TEXT,
source_tool_name TEXT,
audience_session_keys_json TEXT NOT NULL,
runtime_epoch TEXT NOT NULL,
created_at_ms INTEGER NOT NULL,
expires_at_ms INTEGER NOT NULL,
updated_at_ms INTEGER NOT NULL,
decision TEXT CHECK (decision IN ('allow-once', 'allow-always', 'deny')),
terminal_reason TEXT CHECK (
terminal_reason IN (
'user',
'timeout',
'malformed-verdict',
'no-route',
'run-aborted',
'gateway-restart',
'storage-corrupt'
)
),
resolved_at_ms INTEGER,
resolver_kind TEXT CHECK (resolver_kind IN ('device', 'channel', 'runtime', 'system')),
resolver_id TEXT,
consumed_at_ms INTEGER,
consumed_by TEXT,
CHECK (expires_at_ms >= created_at_ms),
CHECK (updated_at_ms >= created_at_ms),
CHECK (resolved_at_ms IS NULL OR resolved_at_ms >= created_at_ms),
CHECK (resolved_at_ms IS NULL OR resolved_at_ms <= updated_at_ms),
CHECK (consumed_at_ms IS NULL OR consumed_at_ms >= resolved_at_ms),
CHECK (consumed_at_ms IS NULL OR consumed_at_ms <= updated_at_ms),
CHECK (requested_by_device_token_auth IN (0, 1)),
CHECK (
(
status = 'pending'
AND decision IS NULL
AND terminal_reason IS NULL
AND resolved_at_ms IS NULL
AND resolver_kind IS NULL
AND resolver_id IS NULL
AND consumed_at_ms IS NULL
AND consumed_by IS NULL
)
OR (
status = 'allowed'
AND decision IN ('allow-once', 'allow-always')
AND terminal_reason = 'user'
AND resolved_at_ms IS NOT NULL
AND resolver_kind IS NOT NULL
)
OR (
status = 'denied'
AND decision = 'deny'
AND terminal_reason IN ('user', 'malformed-verdict', 'no-route', 'storage-corrupt')
AND resolved_at_ms IS NOT NULL
AND resolver_kind IS NOT NULL
AND consumed_at_ms IS NULL
AND consumed_by IS NULL
)
OR (
status = 'expired'
AND decision = 'deny'
AND terminal_reason = 'timeout'
AND resolved_at_ms IS NOT NULL
AND resolver_kind IS NOT NULL
AND consumed_at_ms IS NULL
AND consumed_by IS NULL
)
OR (
status = 'cancelled'
AND decision = 'deny'
AND terminal_reason IN ('run-aborted', 'gateway-restart')
AND resolved_at_ms IS NOT NULL
AND resolver_kind IS NOT NULL
AND consumed_at_ms IS NULL
AND consumed_by IS NULL
)
),
CHECK (
(consumed_at_ms IS NULL AND consumed_by IS NULL)
OR (
status = 'allowed'
AND decision = 'allow-once'
AND consumed_at_ms IS NOT NULL
AND consumed_by IS NOT NULL
)
)
) STRICT;
CREATE INDEX IF NOT EXISTS idx_operator_approvals_status_expiry
ON operator_approvals(status, expires_at_ms, approval_id);
CREATE UNIQUE INDEX IF NOT EXISTS idx_operator_approvals_resolution_ref
ON operator_approvals(resolution_ref);
CREATE INDEX IF NOT EXISTS idx_operator_approvals_source_session_created
ON operator_approvals(source_session_key, created_at_ms DESC, approval_id);
CREATE INDEX IF NOT EXISTS idx_operator_approvals_resolved
ON operator_approvals(resolved_at_ms, approval_id)
WHERE resolved_at_ms IS NOT NULL;
CREATE INDEX IF NOT EXISTS idx_operator_approvals_runtime_pending
ON operator_approvals(runtime_epoch, approval_id)
WHERE status = 'pending';
`;

View file

@ -0,0 +1,119 @@
import assert from "node:assert/strict";
import { randomUUID } from "node:crypto";
import type { QaGatewayChild } from "../../../../extensions/qa-lab/src/gateway-child.js";
import type { AuditRunInspectResult } from "../../../../packages/gateway-protocol/src/index.js";
import {
connectHotReloadClient,
waitForHotReloadFact,
type HotReloadConnection,
} from "./gateway-config-hot-reload-fixtures.js";
export async function runIdentityGatewayTurn(gateway: QaGatewayChild, label: string) {
const accepted = (await gateway.call(
"agent",
{
sessionKey: `agent:qa:identity-hot-${randomUUID()}`,
message: `Reply exactly: ${label}`,
deliver: false,
idempotencyKey: randomUUID(),
},
{ expectFinal: false },
)) as { status: string; runId: string };
assert.equal(accepted.status, "accepted");
assert.equal(typeof accepted.runId, "string");
const terminal = (await gateway.call(
"agent.wait",
{ runId: accepted.runId, timeoutMs: 60_000 },
{ timeoutMs: 65_000 },
)) as { status: string };
assert.equal(terminal.status, "ok", `${label} must complete through the real Gateway`);
return accepted.runId;
}
export async function patchExecutionIdentity(
gateway: QaGatewayChild,
connection: HotReloadConnection,
values: { enabled?: boolean; executionIdentity: boolean },
) {
const pid = gateway.pid;
const bootId = connection.bootId;
const snapshot = await connection.client.request<{ hash: string }>("config.get", {});
// config.patch acknowledges the runtime application, not merely the file write.
const applied = await connection.client.request<{
sentinel: { payload: { stats: { requiresRestart: boolean } } };
}>("config.patch", {
baseHash: snapshot.hash,
raw: JSON.stringify({ logging: { audit: values } }),
});
assert.equal(applied.sentinel.payload.stats.requiresRestart, false);
assert.equal((await connection.client.request<{ pid: number }>("system.info", {})).pid, pid);
assert.equal(connection.hellos, 1, "hot audit changes must retain the connected socket");
assert.equal(connection.closes, 0, "hot audit changes must not disconnect operators");
const fresh = await connectHotReloadClient(gateway);
try {
assert.equal(fresh.bootId, bootId, "hot audit changes must retain the same Gateway boot");
} finally {
await fresh.client.stopAndWait({ timeoutMs: 2_000 });
}
}
export async function waitForIdentityAuditFence(gateway: QaGatewayChild, runId: string) {
return waitForHotReloadFact(`identity and terminal activity for ${runId}`, async () => {
const result = (await gateway.call("audit.run.inspect", { runId })) as AuditRunInspectResult;
const activity = (await gateway.call("audit.activity.list", { runId })) as {
events: Array<{ action: string }>;
};
return result.identity.state === "present" &&
activity.events.some((event) => event.action === "agent.run.finished")
? result.identity.context
: undefined;
});
}
export async function proveHotExecutionIdentity(gateway: QaGatewayChild) {
const connection = await connectHotReloadClient(gateway);
const inspect = (runId: string) =>
connection.client.request<AuditRunInspectResult>("audit.run.inspect", { runId });
const present = (runId: string) =>
waitForHotReloadFact(`persisted identity for ${runId}`, async () => {
const result = await inspect(runId);
return result.identity.state === "present" ? result.identity.context : undefined;
});
try {
const beforeEnable = await runIdentityGatewayTurn(gateway, "HOT-DEFAULT-OFF");
await patchExecutionIdentity(gateway, connection, { executionIdentity: true });
const enabled = await runIdentityGatewayTurn(gateway, "HOT-ENABLED");
const retained = await present(enabled);
assert.equal(retained.runId, enabled);
await patchExecutionIdentity(gateway, connection, { executionIdentity: false });
const whileDisabled = await runIdentityGatewayTurn(gateway, "HOT-DISABLED");
assert.deepEqual(await present(enabled), retained, "disable must retain immutable evidence");
await patchExecutionIdentity(gateway, connection, { executionIdentity: true });
const reenabled = await runIdentityGatewayTurn(gateway, "HOT-REENABLED");
const next = await present(reenabled);
assert.notEqual(next.executionId, retained.executionId);
assert.notEqual(next.contextId, retained.contextId);
// The later positive write settles the same audit FIFO. Earlier disabled
// admissions must remain without identity after both enabling transitions.
for (const runId of [beforeEnable, whileDisabled]) {
const result = await inspect(runId);
assert.equal(result.identity.state, "unsupported", "enabling must not backfill a run");
assert.deepEqual(result.decisionDisplays, []);
}
assert.deepEqual(await present(enabled), retained, "reenable must not rewrite an old context");
return {
bootId: connection.bootId,
pid: gateway.pid,
beforeEnable,
enabled,
whileDisabled,
reenabled,
noBackfill: true,
retainedContextUnchanged: true,
};
} finally {
await connection.client.stopAndWait({ timeoutMs: 2_000 });
}
}

View file

@ -19,6 +19,13 @@ import { startQaMockOpenAiServer } from "../../../../extensions/qa-lab/src/provi
import type { AuditRunInspectResult } from "../../../../packages/gateway-protocol/src/index.js";
import { formatErrorMessage } from "../../../../src/infra/errors.js";
import { stopQaGatewayFixture } from "../../../helpers/qa-gateway-cleanup.js";
import {
proveHotExecutionIdentity,
patchExecutionIdentity,
runIdentityGatewayTurn,
waitForIdentityAuditFence,
} from "./agent-run-identity-hot-reload.js";
import { connectHotReloadClient } from "./gateway-config-hot-reload-fixtures.js";
import { createQaScriptEvidenceWriter, type QaScriptEvidenceStatus } from "./script-evidence.js";
const SCENARIO_ID = "agent-run-identity-inspection";
@ -85,24 +92,6 @@ async function assertUntrustedProxyHeadersRejected(
});
}
async function updateExecutionIdentityConfig(
configPath: string,
values: { enabled?: boolean; executionIdentity: boolean },
) {
const raw = await fs.readFile(configPath, "utf8");
const config = parseJson(raw || "{}", "QA Gateway config") as Record<string, unknown>;
const logging =
config.logging && typeof config.logging === "object"
? (config.logging as Record<string, unknown>)
: {};
const audit =
logging.audit && typeof logging.audit === "object"
? (logging.audit as Record<string, unknown>)
: {};
config.logging = { ...logging, audit: { ...audit, ...values } };
await fs.writeFile(configPath, `${JSON.stringify(config, null, 2)}\n`, "utf8");
}
function parseOptions(argv: readonly string[]): ProducerOptions {
const readValue = (name: string) => {
const index = argv.indexOf(name);
@ -400,7 +389,12 @@ async function runProof(options: ProducerOptions): Promise<string> {
try {
gateway = await gatewayOwner.start({
repoRoot: options.repoRoot,
useRepoCli: true,
command: {
executablePath: process.execPath,
argsPrefix: [path.join(options.repoRoot, "dist/index.js")],
cwd: options.repoRoot,
usePackagedPlugins: true,
},
providerBaseUrl: `${mock.baseUrl}/v1`,
providerMode: "mock-openai",
transportBaseUrl: "http://127.0.0.1",
@ -425,8 +419,8 @@ async function runProof(options: ProducerOptions): Promise<string> {
if (inspectExecutionIdentityStorage(gateway).tablePresent) {
throw new Error("existing-install restart unexpectedly created execution identity storage");
}
await gateway.restartAfterStateMutation(async ({ configPath }) => {
await updateExecutionIdentityConfig(configPath, { executionIdentity: true });
const hotReload = await proveHotExecutionIdentity(gateway);
await gateway.restartAfterStateMutation(async () => {
await runLocalTurn(gateway!, "Reply exactly: IDENTITY-INSPECTION-OK");
});
const runId = findLocalRunId(gateway);
@ -593,22 +587,42 @@ async function runProof(options: ProducerOptions): Promise<string> {
}
}
const retainedBeforeGlobalDisable = inspectExecutionIdentityStorage(gateway).rowCount;
await gateway.restartAfterStateMutation(async ({ configPath }) => {
await updateExecutionIdentityConfig(configPath, {
const globalDisableConnection = await connectHotReloadClient(gateway);
let globalDisabledRunId: string;
try {
await patchExecutionIdentity(gateway, globalDisableConnection, {
enabled: false,
executionIdentity: true,
});
await runLocalTurn(gateway!, "Reply exactly: IDENTITY-DISABLED-GLOBAL");
});
if (inspectExecutionIdentityStorage(gateway).rowCount !== retainedBeforeGlobalDisable) {
throw new Error("global audit disable unexpectedly retained a new execution context");
globalDisabledRunId = await runIdentityGatewayTurn(gateway, "IDENTITY-DISABLED-GLOBAL");
const afterGlobalDisable = parseJson(
await gateway.runCli(["audit", "--run", runId, "--explain", "--json"]),
"global-disabled retained inspection",
) as AuditRunInspectResult;
if (normalizedContextJson(afterGlobalDisable) !== beforeContext) {
throw new Error("global audit disable hid or changed retained identity evidence");
}
await patchExecutionIdentity(gateway, globalDisableConnection, {
enabled: true,
executionIdentity: true,
});
const fenceRunId = await runIdentityGatewayTurn(gateway, "IDENTITY-GLOBAL-REENABLED");
await waitForIdentityAuditFence(gateway, fenceRunId);
} finally {
await globalDisableConnection.client.stopAndWait({ timeoutMs: 2_000 });
}
const afterGlobalDisable = parseJson(
await gateway.runCli(["audit", "--run", runId, "--explain", "--json"]),
"global-disabled retained inspection",
) as AuditRunInspectResult;
if (normalizedContextJson(afterGlobalDisable) !== beforeContext) {
throw new Error("global audit disable hid or changed retained identity evidence");
// A later identity and terminal event have crossed the same writer FIFO;
// agent.wait alone does not settle queued audit persistence.
if (inspectExecutionIdentityStorage(gateway).rowCount !== retainedBeforeGlobalDisable + 1) {
throw new Error(
"global audit disable unexpectedly retained or backfilled an execution context",
);
}
const disabledActivity = (await gateway.call("audit.activity.list", {
runId: globalDisabledRunId,
})) as { events: unknown[] };
if (disabledActivity.events.length !== 0) {
throw new Error("hot global audit disable recorded activity for a subsequent run");
}
const snapshotPath = path.join(options.artifactBase, SNAPSHOT_FILE);
@ -635,6 +649,7 @@ async function runProof(options: ProducerOptions): Promise<string> {
contextSha256: sha256(beforeContext),
byteEquivalentAfterRestart: true,
byteEquivalentPersistedReadback: true,
hotReload,
optIn: {
explicitEnablement: true,
freshInstallDisabled: true,

View file

@ -0,0 +1,448 @@
// Shared receipt evidence helpers and admitted webhook proof for the real QA Gateway.
import { randomUUID } from "node:crypto";
import fs from "node:fs/promises";
import path from "node:path";
import { DatabaseSync } from "node:sqlite";
import { setTimeout as delay } from "node:timers/promises";
import { isRecord } from "@openclaw/normalization-core/record-coerce";
import type { QaGatewayChild } from "../../../../extensions/qa-lab/src/gateway-child.js";
import { validateExecutionIdentityContextV1 } from "../../../../packages/gateway-protocol/src/audit-run-validators.js";
import { lazyCompile } from "../../../../packages/gateway-protocol/src/protocol-validator.js";
import {
AuditRunInspectResultSchema,
type AuditRunInspectResult,
type ExecutionIdentityContextV1,
} from "../../../../packages/gateway-protocol/src/schema/audit-run.js";
import { formatErrorMessage } from "../../../../src/infra/errors.js";
export type ExactOwnerRow = {
context_id: string;
execution_id: string;
run_id: string;
status: string;
};
type OwnerDisplayProducer = "cron-lifecycle" | "task-lifecycle" | "flow-lifecycle";
type ContextRow = {
context_id: string;
execution_id: string;
run_id: string;
context_json: string;
};
export function hasSqliteColumns(
db: DatabaseSync,
table: string,
columns: readonly string[],
): boolean {
const exists = db
.prepare("SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = ?")
.get(table);
if (!exists) {
return false;
}
const present = new Set(
(db.prepare(`PRAGMA table_info(${table})`).all() as Array<{ name: string }>).map(
(row) => row.name,
),
);
return columns.every((column) => present.has(column));
}
function parseJson(raw: string, label: string): unknown {
try {
return JSON.parse(raw);
} catch (error: unknown) {
throw new Error(`${label} was not JSON: ${formatErrorMessage(error)}`, { cause: error });
}
}
const validateAuditInspection = lazyCompile(AuditRunInspectResultSchema);
export function parseAuditInspection(raw: string, label: string): AuditRunInspectResult {
const value = parseJson(raw, label);
if (!validateAuditInspection(value)) {
throw new Error(`${label} did not match the audit inspection contract`);
}
return value;
}
export function stateDatabasePath(gateway: QaGatewayChild): string {
const stateDir = gateway.runtimeEnv.OPENCLAW_STATE_DIR;
if (!stateDir) {
throw new Error("QA Gateway did not expose its isolated state directory");
}
return path.join(stateDir, "state", "openclaw.sqlite");
}
export function countExecutionContexts(gateway: QaGatewayChild): number {
const db = new DatabaseSync(stateDatabasePath(gateway), { readOnly: true });
try {
if (!hasSqliteColumns(db, "execution_identity_contexts", ["context_id"])) {
return 0;
}
const row = db.prepare("SELECT COUNT(*) AS count FROM execution_identity_contexts").get() as {
count: number;
};
return row.count;
} finally {
db.close();
}
}
function readExecutionContexts(gateway: QaGatewayChild): ContextRow[] {
const db = new DatabaseSync(stateDatabasePath(gateway), { readOnly: true });
try {
if (!hasSqliteColumns(db, "execution_identity_contexts", ["context_json"])) {
return [];
}
return db
.prepare(
"SELECT context_id, execution_id, run_id, context_json FROM execution_identity_contexts ORDER BY execution_id",
)
.all() as ContextRow[];
} finally {
db.close();
}
}
function requirePrivateSentinelsAbsent(
text: string,
sentinels: readonly string[],
surface: string,
) {
if (sentinels.some((sentinel) => text.includes(sentinel))) {
throw new Error(`${surface} leaked a private webhook or prompt sentinel`);
}
}
function requireWebhookContext(context: ExecutionIdentityContextV1) {
if (
context.ingress.kind !== "webhook" ||
context.ingress.boundary !== "gateway.hooks.agent" ||
context.ingress.state !== "present" ||
!/^hmac-sha256:v1:[a-f0-9]{32}:[a-f0-9]{64}$/u.test(context.ingress.sourceRef ?? "")
) {
throw new Error("admitted mapped webhook omitted its pseudonymized ingress source");
}
if (
JSON.stringify(context.invoker) !== JSON.stringify({ state: "absent" }) ||
context.coverageState !== "unattributed" ||
!context.missingEvidence.includes("invoker.principal")
) {
throw new Error("mapped webhook source or shared authentication became invoker evidence");
}
}
function requireWebhookIdentity(result: AuditRunInspectResult, row: ContextRow) {
if (result.identity.state !== "present") {
throw new Error("mapped webhook inspection did not retain its exact persisted context");
}
const context = result.identity.context;
if (
JSON.stringify(context) !== row.context_json ||
context.contextId !== row.context_id ||
context.executionId !== row.execution_id ||
context.runId !== row.run_id
) {
throw new Error("mapped webhook inspection did not retain its exact persisted context");
}
requireWebhookContext(context);
const admission = result.decisionDisplays.find(
(display) =>
display.provenance.state === "verified" && display.provenance.producer === "run-admission",
);
if (
admission?.enforcement.coverageState !== "unattributed" ||
admission.decision.outcome !== "not-applicable"
) {
throw new Error("mapped webhook source or shared authentication became invoker evidence");
}
return context;
}
export function readCliOwnerRows(
gateway: QaGatewayChild,
runId: string,
): { task: ExactOwnerRow } | undefined {
const db = new DatabaseSync(stateDatabasePath(gateway), { readOnly: true });
try {
if (
!hasSqliteColumns(db, "execution_identity_contexts", ["context_id", "execution_id"]) ||
!hasSqliteColumns(db, "execution_owner_lifecycle_bindings", [
"owner_kind",
"owner_id",
"context_id",
"execution_id",
]) ||
!hasSqliteColumns(db, "task_runs", ["task_id"])
) {
return undefined;
}
const task = db
.prepare(
`SELECT binding.context_id, binding.execution_id, context.run_id, task.status
FROM task_runs AS task
JOIN execution_owner_lifecycle_bindings AS binding
ON binding.owner_kind = 'task' AND binding.owner_id = task.task_id
JOIN execution_identity_contexts AS context
ON context.context_id = binding.context_id
AND context.execution_id = binding.execution_id
WHERE task.runtime = 'cli' AND task.run_id = ? AND task.ended_at IS NOT NULL
LIMIT 1`,
)
.get(runId) as ExactOwnerRow | undefined;
return task ? { task } : undefined;
} finally {
db.close();
}
}
export async function waitFor<T>(label: string, read: () => T | undefined): Promise<T> {
const deadline = Date.now() + 30_000;
while (Date.now() < deadline) {
const value = read();
if (value !== undefined) {
return value;
}
await delay(50);
}
throw new Error(`timed out waiting for ${label}`);
}
export function requireOwnerDisplay(result: AuditRunInspectResult, producer: OwnerDisplayProducer) {
const receipt = result.decisionDisplays.find(
(candidate) =>
candidate.provenance.state === "verified" && candidate.provenance.producer === producer,
);
if (
!receipt ||
receipt.enforcement.coverageState !== "attribution-only" ||
receipt.decision.outcome !== "not-applicable"
) {
throw new Error(`inspection omitted exact attribution-only ${producer} display`);
}
return receipt;
}
export async function inspectExecution(params: {
gateway: QaGatewayChild;
executionId: string;
producers: OwnerDisplayProducer[];
privateSentinels: string[];
}) {
const jsonRaw = await params.gateway.runCli([
"audit",
"--execution",
params.executionId,
"--explain",
"--json",
]);
const json = parseAuditInspection(jsonRaw, "owner lifecycle inspection");
for (const producer of params.producers) {
requireOwnerDisplay(json, producer);
}
requirePrivateSentinelsAbsent(jsonRaw, params.privateSentinels, "JSON inspection");
const human = await params.gateway.runCli([
"audit",
"--execution",
params.executionId,
"--explain",
]);
requirePrivateSentinelsAbsent(human, params.privateSentinels, "human inspection");
for (const producer of params.producers) {
if (!human.includes(`Display producer: ${producer}`)) {
throw new Error(`human inspection omitted ${producer}`);
}
}
return { json, jsonRaw, human };
}
export function createMappedWebhookProof(hookToken: string) {
const mappingId = `PRIVATE-MAPPING-${randomUUID()}`;
const createRequest = () => ({
requestId: `PRIVATE-REQUEST-${randomUUID()}`,
body: `PRIVATE-WEBHOOK-BODY-${randomUUID()}`,
});
const webhookRequests = [createRequest(), createRequest()];
const restartRequest = createRequest();
const webhookSentinels = [
mappingId,
...[...webhookRequests, restartRequest].flatMap(({ requestId, body }) => [requestId, body]),
];
const admitRequest = async (
gateway: QaGatewayChild,
request: ReturnType<typeof createRequest>,
) => {
const before = new Set(readExecutionContexts(gateway).map((row) => row.execution_id));
const response = await fetch(`${gateway.baseUrl}/hooks/admitted`, {
method: "POST",
headers: {
Authorization: `Bearer ${hookToken}`,
"Content-Type": "application/json",
"Idempotency-Key": request.requestId,
"X-Request-Id": request.requestId,
},
body: JSON.stringify({ value: request.body }),
});
const accepted = parseJson(await response.text(), "mapped webhook admission");
if (
response.status !== 200 ||
!isRecord(accepted) ||
accepted.ok !== true ||
typeof accepted.runId !== "string" ||
!accepted.runId
) {
throw new Error(`mapped webhook admission returned HTTP ${response.status}`);
}
// The HTTP dispatch id differs from the inner admitted run id. One POST
// at a time binds the new context without guessing from either id.
const row = await waitFor("one admitted webhook context", () => {
const added = readExecutionContexts(gateway).filter(
(candidate) => !before.has(candidate.execution_id),
);
if (added.length > 1) {
throw new Error("one mapped webhook allocated multiple execution contexts");
}
return added[0];
});
const persistedContext = parseJson(row.context_json, "persisted webhook context");
if (!validateExecutionIdentityContextV1(persistedContext)) {
throw new Error("persisted webhook context did not match the execution identity contract");
}
requireWebhookContext(persistedContext);
// Hooks invoke the isolated runner directly; only scheduled cron owns task rows.
// Observe the admitted runner's terminal event before inspection or replacement.
const completed = (await gateway.call(
"agent.wait",
{ runId: row.run_id, timeoutMs: 30_000 },
{ timeoutMs: 35_000 },
)) as { runId: string; status: string };
if (completed.runId !== row.run_id || completed.status !== "ok") {
throw new Error(`admitted mapped webhook did not complete successfully: ${completed.status}`);
}
const inspection = await inspectExecution({
gateway,
executionId: row.execution_id,
producers: [],
privateSentinels: webhookSentinels,
});
const context = requireWebhookIdentity(inspection.json, row);
for (const text of [
"Invoker [absent]",
"Ingress [present]",
"webhook at gateway.hooks.agent",
context.ingress.sourceRef!,
]) {
if (!inspection.human.includes(text)) {
throw new Error("human inspection omitted mapped webhook identity evidence");
}
}
return { row, context, inspection };
};
return {
mapping: {
id: mappingId,
match: { path: "admitted" },
action: "agent" as const,
messageTemplate: "{{payload.value}}: reply WEBHOOK-DONE",
deliver: false,
wakeMode: "next-heartbeat" as const,
},
async admit(gateway: QaGatewayChild) {
const webhookProofs: Array<Awaited<ReturnType<typeof admitRequest>>> = [];
for (const request of webhookRequests) {
webhookProofs.push(await admitRequest(gateway, request));
}
const [firstWebhook, secondWebhook] = webhookProofs;
if (
!firstWebhook ||
!secondWebhook ||
firstWebhook.context.ingress.sourceRef !== secondWebhook.context.ingress.sourceRef ||
firstWebhook.row.context_id === secondWebhook.row.context_id
) {
throw new Error(
"one mapping did not retain a stable source across distinct webhook requests",
);
}
requirePrivateSentinelsAbsent(
JSON.stringify(readExecutionContexts(gateway)),
webhookSentinels,
"persisted execution contexts",
);
return {
contexts: webhookProofs.map(({ context }) => ({
contextId: context.contextId,
executionId: context.executionId,
ingress: context.ingress,
invoker: context.invoker,
coverageState: context.coverageState,
})),
inspections: webhookProofs.map(({ inspection }) => inspection.json),
async verifyAfterRestart(replacementGateway: QaGatewayChild) {
const restarted = await admitRequest(replacementGateway, restartRequest);
if (
webhookProofs.some(
({ row, context }) =>
context.ingress.sourceRef !== restarted.context.ingress.sourceRef ||
row.context_id === restarted.row.context_id ||
row.execution_id === restarted.row.execution_id,
)
) {
throw new Error(
"new webhook admission changed its mapping source across Gateway replacement",
);
}
const webhooksAfter = [];
const persistedAfter = readExecutionContexts(replacementGateway);
for (const proof of webhookProofs) {
const row = persistedAfter.find(
(candidate) => candidate.execution_id === proof.row.execution_id,
);
if (!row || row.context_json !== proof.row.context_json) {
throw new Error("persisted webhook context changed across Gateway replacement");
}
const inspection = await inspectExecution({
gateway: replacementGateway,
executionId: row.execution_id,
producers: [],
privateSentinels: webhookSentinels,
});
requireWebhookIdentity(inspection.json, row);
if (inspection.human !== proof.inspection.human) {
throw new Error("human webhook inspection changed across Gateway replacement");
}
webhooksAfter.push(inspection.json);
}
requirePrivateSentinelsAbsent(
JSON.stringify(persistedAfter),
webhookSentinels,
"persisted execution contexts after restart",
);
const auditLogLines = (
await Promise.all(
["gateway.stdout.log", "gateway.stderr.log"].map((file) =>
fs.readFile(path.join(replacementGateway.tempRoot, file), "utf8"),
),
)
)
.join("\n")
.split(/\r?\n/u)
.filter((line) => /\baudit\b|execution identity|execution decision/iu.test(line));
requirePrivateSentinelsAbsent(auditLogLines.join("\n"), webhookSentinels, "audit logs");
return {
inspections: webhooksAfter,
admittedContext: {
contextId: restarted.context.contextId,
executionId: restarted.context.executionId,
ingress: restarted.context.ingress,
invoker: restarted.context.invoker,
coverageState: restarted.context.coverageState,
},
auditLogLinesChecked: auditLogLines.length,
};
},
};
},
};
}

View file

@ -14,9 +14,20 @@ import {
type QaGatewayChild,
} from "../../../../extensions/qa-lab/src/gateway-child.js";
import { startQaMockOpenAiServer } from "../../../../extensions/qa-lab/src/providers/mock-openai/server.js";
import type { AuditRunInspectResult } from "../../../../packages/gateway-protocol/src/index.js";
import { formatErrorMessage } from "../../../../src/infra/errors.js";
import { stopQaGatewayFixture } from "../../../helpers/qa-gateway-cleanup.js";
import {
countExecutionContexts,
createMappedWebhookProof,
hasSqliteColumns,
inspectExecution,
parseAuditInspection,
readCliOwnerRows,
requireOwnerDisplay,
stateDatabasePath,
waitFor,
type ExactOwnerRow,
} from "./autonomous-task-lifecycle-receipts.fixtures.js";
import { createQaScriptEvidenceWriter, type QaScriptEvidenceStatus } from "./script-evidence.js";
const SCENARIO_ID = "autonomous-task-lifecycle-receipts";
@ -30,29 +41,6 @@ type ProofResult = {
durationMs: number;
status: QaScriptEvidenceStatus;
};
type ExactOwnerRow = {
context_id: string;
execution_id: string;
run_id: string;
status: string;
};
type OwnerDisplayProducer = "cron-lifecycle" | "task-lifecycle" | "flow-lifecycle";
function hasSqliteColumns(db: DatabaseSync, table: string, columns: readonly string[]): boolean {
const exists = db
.prepare("SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = ?")
.get(table);
if (!exists) {
return false;
}
const present = new Set(
(db.prepare(`PRAGMA table_info(${table})`).all() as Array<{ name: string }>).map(
(row) => row.name,
),
);
return columns.every((column) => present.has(column));
}
function parseOptions(argv: readonly string[]): ProducerOptions {
const readValue = (name: string) => {
const index = argv.indexOf(name);
@ -68,41 +56,10 @@ function parseOptions(argv: readonly string[]): ProducerOptions {
};
}
function parseJson<T>(raw: string, label: string): T {
try {
return JSON.parse(raw) as T;
} catch (error) {
throw new Error(`${label} was not JSON: ${formatErrorMessage(error)}`);
}
}
function sha256(value: string): string {
return createHash("sha256").update(value).digest("hex");
}
function stateDatabasePath(gateway: QaGatewayChild): string {
const stateDir = gateway.runtimeEnv.OPENCLAW_STATE_DIR;
if (!stateDir) {
throw new Error("QA Gateway did not expose its isolated state directory");
}
return path.join(stateDir, "state", "openclaw.sqlite");
}
function countExecutionContexts(gateway: QaGatewayChild): number {
const db = new DatabaseSync(stateDatabasePath(gateway), { readOnly: true });
try {
if (!hasSqliteColumns(db, "execution_identity_contexts", ["context_id"])) {
return 0;
}
const row = db.prepare("SELECT COUNT(*) AS count FROM execution_identity_contexts").get() as {
count: number;
};
return row.count;
} finally {
db.close();
}
}
function readCronOwnerRows(
gateway: QaGatewayChild,
jobId: string,
@ -183,114 +140,21 @@ function readCronOwnerBindingDiagnostic(gateway: QaGatewayChild, jobId: string):
}
}
function readCliOwnerRows(
gateway: QaGatewayChild,
runId: string,
): { task: ExactOwnerRow } | undefined {
const db = new DatabaseSync(stateDatabasePath(gateway), { readOnly: true });
try {
if (
!hasSqliteColumns(db, "execution_identity_contexts", ["context_id", "execution_id"]) ||
!hasSqliteColumns(db, "execution_owner_lifecycle_bindings", [
"owner_kind",
"owner_id",
"context_id",
"execution_id",
]) ||
!hasSqliteColumns(db, "task_runs", ["task_id"])
) {
return undefined;
}
const task = db
.prepare(
`SELECT binding.context_id, binding.execution_id, context.run_id, task.status
FROM task_runs AS task
JOIN execution_owner_lifecycle_bindings AS binding
ON binding.owner_kind = 'task' AND binding.owner_id = task.task_id
JOIN execution_identity_contexts AS context
ON context.context_id = binding.context_id
AND context.execution_id = binding.execution_id
WHERE task.runtime = 'cli' AND task.run_id = ? AND task.ended_at IS NOT NULL
LIMIT 1`,
)
.get(runId) as ExactOwnerRow | undefined;
return task ? { task } : undefined;
} finally {
db.close();
}
}
async function waitFor<T>(label: string, read: () => T | undefined): Promise<T> {
const deadline = Date.now() + 30_000;
while (Date.now() < deadline) {
const value = read();
if (value !== undefined) {
return value;
}
await delay(50);
}
throw new Error(`timed out waiting for ${label}`);
}
function requireOwnerDisplay(result: AuditRunInspectResult, producer: OwnerDisplayProducer) {
const receipt = result.decisionDisplays.find(
(candidate) =>
candidate.provenance.state === "verified" && candidate.provenance.producer === producer,
);
if (
!receipt ||
receipt.enforcement.coverageState !== "attribution-only" ||
receipt.decision.outcome !== "not-applicable"
) {
throw new Error(`inspection omitted exact attribution-only ${producer} display`);
}
return receipt;
}
async function inspectExecution(params: {
gateway: QaGatewayChild;
executionId: string;
producers: OwnerDisplayProducer[];
privateSentinels: string[];
}) {
const jsonRaw = await params.gateway.runCli([
"audit",
"--execution",
params.executionId,
"--explain",
"--json",
]);
const json = parseJson<AuditRunInspectResult>(jsonRaw, "owner lifecycle inspection");
for (const producer of params.producers) {
requireOwnerDisplay(json, producer);
}
for (const sentinel of params.privateSentinels) {
if (jsonRaw.includes(sentinel)) {
throw new Error(`owner receipt leaked private sentinel ${sentinel}`);
}
}
const human = await params.gateway.runCli([
"audit",
"--execution",
params.executionId,
"--explain",
]);
for (const producer of params.producers) {
if (!human.includes(`Display producer: ${producer}`)) {
throw new Error(`human inspection omitted ${producer}`);
}
}
return { json, jsonRaw, human };
}
async function runProof(options: ProducerOptions): Promise<string> {
const mock = await startQaMockOpenAiServer();
const gatewayOwner = createQaGatewayChild();
let gateway: QaGatewayChild | undefined;
const webhook = createMappedWebhookProof(HOOK_TOKEN);
try {
gateway = await gatewayOwner.start({
repoRoot: options.repoRoot,
useRepoCli: true,
command: {
executablePath: process.execPath,
argsPrefix: [path.join(options.repoRoot, "dist/index.js")],
cwd: options.repoRoot,
usePackagedPlugins: true,
},
providerBaseUrl: `${mock.baseUrl}/v1`,
providerMode: "mock-openai",
transportBaseUrl: "http://127.0.0.1",
@ -305,6 +169,7 @@ async function runProof(options: ProducerOptions): Promise<string> {
enabled: true,
token: HOOK_TOKEN,
mappings: [
webhook.mapping,
{
id: "qa-suppressed-source",
match: { path: "suppressed" },
@ -341,6 +206,8 @@ async function runProof(options: ProducerOptions): Promise<string> {
throw new Error("pre-admission mapping suppression allocated execution identity");
}
const webhookProof = await webhook.admit(gateway);
const cronSentinel = `PRIVATE-CRON-${randomUUID()}`;
const cronJob = (await gateway.call("cron.add", {
name: "QA autonomous receipt",
@ -375,7 +242,7 @@ async function runProof(options: ProducerOptions): Promise<string> {
producers: ["cron-lifecycle", "task-lifecycle"],
privateSentinels: [cronSentinel],
});
const cronCursorPage = parseJson<AuditRunInspectResult>(
const cronCursorPage = parseAuditInspection(
await gateway.runCli([
"audit",
"--execution",
@ -421,6 +288,7 @@ async function runProof(options: ProducerOptions): Promise<string> {
const beforeRestart = JSON.stringify({
cron: cronInspection.json,
task: taskInspection.json,
webhooks: webhookProof.inspections,
});
await gateway.restartAfterStateMutation(async () => {});
const cronAfter = await inspectExecution({
@ -435,7 +303,12 @@ async function runProof(options: ProducerOptions): Promise<string> {
producers: ["task-lifecycle"],
privateSentinels: [taskSentinel],
});
const afterRestart = JSON.stringify({ cron: cronAfter.json, task: taskAfter.json });
const webhooksAfter = await webhookProof.verifyAfterRestart(gateway);
const afterRestart = JSON.stringify({
cron: cronAfter.json,
task: taskAfter.json,
webhooks: webhooksAfter.inspections,
});
if (afterRestart !== beforeRestart) {
throw new Error("owner lifecycle JSON changed across Gateway replacement");
}
@ -468,6 +341,8 @@ async function runProof(options: ProducerOptions): Promise<string> {
`${JSON.stringify(
{
suppression: { httpStatus: 204, identityAllocation: 0 },
webhooks: webhookProof.contexts,
webhookAdmittedAfterRestart: webhooksAfter.admittedContext,
cron: {
contextId: cronRows.cron.context_id,
executionId: cronRows.cron.execution_id,
@ -483,7 +358,12 @@ async function runProof(options: ProducerOptions): Promise<string> {
cursorCompatibility: { cronPrefixAccepted: true },
genericDuplicateAbsent: true,
byteEquivalentAfterRestart: true,
privacy: { cronPromptAbsent: true, taskPromptAbsent: true },
privacy: {
cronPromptAbsent: true,
taskPromptAbsent: true,
webhookMappingRequestAndBodyAbsent: true,
auditLogLinesChecked: webhooksAfter.auditLogLinesChecked,
},
resultSha256: sha256(afterRestart),
},
null,
@ -554,7 +434,7 @@ if (import.meta.url === pathToFileURL(process.argv[1] ?? "").href) {
.then((exitCode) => {
process.exitCode = exitCode;
})
.catch((error) => {
.catch((error: unknown) => {
console.error(formatErrorMessage(error));
process.exitCode = 1;
});

View file

@ -5,6 +5,34 @@ import { normalizeCredentialPayloadForKind } from "../qa/convex-credential-broke
const BUZZ_DRIVER_PRIVATE_KEY = "01".repeat(32);
const BUZZ_SUT_PRIVATE_KEY = "02".repeat(32);
const BUZZ_DRIVER_NSEC = "nsec1qyqszqgpqyqszqgpqyqszqgpqyqszqgpqyqszqgpqyqszqgpqyqstywftw";
const TELEGRAM_PRIMARY_ARCHIVE = "YQ==";
const TELEGRAM_GUEST_ARCHIVE = "Yg==";
function buildTelegramTestUserbotPayload() {
return {
schemaVersion: 1,
environment: "test",
groupId: "-1001",
forumGroupId: "-1002",
forumTopicId: 42,
sutToken: "test-token",
sutUsername: "test_bot",
sutBotId: "700000001",
testerUserId: "700000002",
tdlibArchiveBase64: TELEGRAM_PRIMARY_ARCHIVE,
tdlibArchiveSha256: "a".repeat(64),
tdlibVersion: "1.8.67",
participants: [
{
alias: "guest",
testerUserId: "700000003",
tdlibArchiveBase64: TELEGRAM_GUEST_ARCHIVE,
tdlibArchiveSha256: "b".repeat(64),
tdlibVersion: "1.8.67",
},
],
};
}
describe("QA Convex credential payload validation", () => {
it("normalizes Buzz credential payloads", () => {
@ -195,6 +223,51 @@ describe("QA Convex credential payload validation", () => {
});
});
it("retains a validated Telegram forum topic and distinct participant sessions", () => {
const normalized = normalizeCredentialPayloadForKind(
"telegram-test-userbot",
buildTelegramTestUserbotPayload(),
);
expect({
forumGroupId: normalized.forumGroupId,
forumTopicId: normalized.forumTopicId,
participants: normalized.participants,
}).toEqual({
forumGroupId: "-1002",
forumTopicId: 42,
participants: [
{
alias: "guest",
testerUserId: "700000003",
tdlibArchiveBase64: TELEGRAM_GUEST_ARCHIVE,
tdlibArchiveSha256: "b".repeat(64),
tdlibVersion: "1.8.67",
},
],
});
});
it("rejects invalid Telegram forum selectors and duplicate participant authority", () => {
expect(() =>
normalizeCredentialPayloadForKind("telegram-test-userbot", {
...buildTelegramTestUserbotPayload(),
forumTopicId: 0,
}),
).toThrow(/invalid forumTopicId/u);
expect(() =>
normalizeCredentialPayloadForKind("telegram-test-userbot", {
...buildTelegramTestUserbotPayload(),
participants: [
{
...buildTelegramTestUserbotPayload().participants[0],
testerUserId: "700000002",
},
],
}),
).toThrow(/distinct participant identities/u);
});
it.each([
["environment", { environment: "production" }],
["bot identity", { sutBotId: "bot" }],

View file

@ -0,0 +1,273 @@
// Qualification of the package Codex harness public audit inspection boundary.
import { spawnSync } from "node:child_process";
import { mkdirSync, readFileSync, writeFileSync } from "node:fs";
import path from "node:path";
import { DatabaseSync } from "node:sqlite";
import { afterEach, describe, expect, it } from "vitest";
import type { AuditRunInspectResult } from "../../packages/gateway-protocol/src/schema/audit-run.js";
import { inspectCodexAudit } from "../../scripts/e2e/lib/codex-npm-plugin-live/audit-inspection.mjs";
import { useAutoCleanupTempDirTracker } from "../helpers/temp-dir.js";
const CODEX_NPM_PLUGIN_LIVE_ASSERTIONS_SCRIPT =
"scripts/e2e/lib/codex-npm-plugin-live/assertions.mjs";
const auditTempDirs = useAutoCleanupTempDirTracker(afterEach);
function nodeOptionsWithoutExperimentalWarnings(): string {
return [process.env.NODE_OPTIONS, "--disable-warning=ExperimentalWarning"]
.filter(Boolean)
.join(" ");
}
function writeJson(filePath: string, value: unknown) {
mkdirSync(path.dirname(filePath), { recursive: true });
writeFileSync(filePath, `${JSON.stringify(value, null, 2)}\n`, "utf8");
}
function runCodexNpmPluginLiveConfigure(root: string, auditIdentity = "0") {
return spawnSync(process.execPath, [CODEX_NPM_PLUGIN_LIVE_ASSERTIONS_SCRIPT, "configure"], {
encoding: "utf8",
env: {
...process.env,
HOME: path.join(root, "home"),
NODE_OPTIONS: nodeOptionsWithoutExperimentalWarnings(),
OPENCLAW_CONFIG_PATH: path.join(root, "state", "openclaw.json"),
OPENCLAW_STATE_DIR: path.join(root, "state"),
OPENCLAW_CODEX_NPM_PLUGIN_AUDIT_IDENTITY: auditIdentity,
},
});
}
function codexAuditFixture() {
const domainRef = `hmac-sha256:v1:${"a".repeat(32)}:${"b".repeat(64)}`;
const selectors = [1, 2, 3].map((index) => ({
runId: `codex-run-${index}`,
executionId: `codex-execution-${index}`,
contextId: `codex-context-${index}`,
}));
const replies = new Map<string, AuditRunInspectResult>();
for (const selector of selectors) {
const result: AuditRunInspectResult = {
schemaVersion: 1,
run: { runId: selector.runId, executionId: selector.executionId, status: "known" },
identity: {
state: "present",
context: {
schemaVersion: 1,
...selector,
createdAt: 100,
trustDomain: {
kind: "gateway-cell",
domainRef,
state: "present",
},
invoker: { state: "absent" },
ingress: { kind: "local-cli", boundary: "agent-command.local", state: "present" },
agentPrincipal: {
kind: "agent",
domainRef,
principalRef: "main",
},
agentDefinition: { definitionRef: "main", state: "present" },
runtimeInstance: {
kind: "plugin-harness",
runtimeRef: `hmac-sha256:v1:${"a".repeat(32)}:${"c".repeat(64)}`,
state: "present",
},
applicableGrants: [],
assurance: [
{
kind: "runtime-binding",
evidenceRef: `hmac-sha256:v1:${"a".repeat(32)}:${"d".repeat(64)}`,
strength: "boundary-verified",
},
],
coverageState: "unattributed",
missingEvidence: ["invoker.principal"],
},
},
decisionDisplays: [
{
schemaVersion: 1,
selectorId: `${selector.contextId}:admission`,
occurredAt: 100,
action: { family: "run", operation: "admission" },
decision: {
outcome: "not-applicable",
reasonCode: "run_admission_identity_not_evaluated",
},
enforcement: {
coverageState: "unattributed",
grantCount: 0,
policyCount: 0,
contextFieldsUsed: [],
},
provenance: { state: "verified", producer: "run-admission" },
missingEvidence: ["invoker.principal"],
remediation: [],
},
],
coverage: { state: "unattributed", missingEvidence: ["invoker.principal"] },
};
replies.set(`--run ${selector.runId} --explain --limit 50`, structuredClone(result));
replies.set(`--execution ${selector.executionId} --explain --limit 100`, result);
}
return { selectors, replies };
}
describe("Codex package audit qualification", () => {
it.each([
"missing-context",
"wrong-run",
"wrong-execution",
"wrong-context",
"fallback-runtime",
"raw-runtime-reference",
"invented-invoker",
"missing-admission",
"invented-enforcement",
"private-reply",
"raw-receipts",
"truncated-decisions",
"missing-turn",
])("rejects %s from the public audit inspector", (failure) => {
const fixture = codexAuditFixture();
const exact = fixture.replies.get("--execution codex-execution-1 --explain --limit 100");
if (!exact || exact.identity.state !== "present") {
throw new Error("expected exact fixture identity");
}
switch (failure) {
case "missing-context":
exact.identity = {
state: "unsupported",
reasonCode: "identity_context_unavailable",
missingEvidence: ["identity.context"],
remediation: [],
};
break;
case "wrong-run":
exact.identity.context.runId = "another-run";
break;
case "wrong-execution":
exact.identity.context.executionId = "another-execution";
break;
case "wrong-context":
exact.identity.context.contextId = "another-context";
break;
case "fallback-runtime":
exact.identity.context.runtimeInstance.kind = "embedded";
break;
case "raw-runtime-reference":
exact.identity.context.runtimeInstance.runtimeRef = "hmac-sha256:v1:private-runtime";
break;
case "invented-invoker":
exact.identity.context.invoker = {
state: "present",
principal: { kind: "person", principalRef: "someone", domainRef: "domain" },
};
break;
case "missing-admission":
exact.decisionDisplays = [];
break;
case "invented-enforcement":
exact.decisionDisplays[0]!.enforcement.coverageState = "enforced";
break;
case "private-reply":
exact.decisionDisplays[0]!.action.summary = "PRIVATE_REPLY_MARKER";
break;
case "raw-receipts":
Object.assign(exact, { decisions: [] });
break;
case "truncated-decisions":
exact.nextDecisionCursor = "a:1:1";
break;
case "missing-turn":
fixture.selectors.pop();
break;
}
expect(() =>
inspectCodexAudit({
selectors: fixture.selectors,
expectedExecutions: 3,
privateValues: ["PRIVATE_REPLY_MARKER"],
query: (args: string[]) => fixture.replies.get(args.join(" ")),
}),
).toThrow();
});
});
describe("Codex audit package CLI boundary", () => {
it("enables audit identity only in the opted-in package fixture and queries the public CLI", () => {
const root = auditTempDirs.make("openclaw-codex-audit-");
const defaultConfig = runCodexNpmPluginLiveConfigure(root);
expect(defaultConfig.status, defaultConfig.stderr).toBe(0);
expect(
JSON.parse(readFileSync(path.join(root, "state", "openclaw.json"), "utf8")),
).not.toHaveProperty("logging.audit.executionIdentity");
const configured = runCodexNpmPluginLiveConfigure(root, "1");
expect(configured.status, configured.stderr).toBe(0);
expect(
JSON.parse(readFileSync(path.join(root, "state", "openclaw.json"), "utf8")),
).toMatchObject({
logging: { audit: { enabled: true, executionIdentity: true } },
gateway: { mode: "local", bind: "loopback", auth: { mode: "token" } },
});
const fixture = codexAuditFixture();
const databasePath = path.join(root, "state", "state", "openclaw.sqlite");
mkdirSync(path.dirname(databasePath), { recursive: true });
const database = new DatabaseSync(databasePath);
try {
database.exec(
"CREATE TABLE execution_identity_contexts (run_id TEXT, execution_id TEXT, context_id TEXT, created_at INTEGER)",
);
const insert = database.prepare(
"INSERT INTO execution_identity_contexts VALUES (?, ?, ?, ?)",
);
fixture.selectors.forEach((selector, index) =>
insert.run(selector.runId, selector.executionId, selector.contextId, index),
);
} finally {
database.close();
}
writeJson(path.join(root, "replies.json"), Object.fromEntries(fixture.replies));
const cliPath = path.join(root, "openclaw-fixture.cjs");
writeFileSync(
cliPath,
`const fs = require("node:fs");
const path = require("node:path");
const [command, ...args] = process.argv.slice(2);
if (command !== "audit" || args.pop() !== "--json") process.exit(2);
const replies = JSON.parse(fs.readFileSync(path.join(__dirname, "replies.json"), "utf8"));
const reply = replies[args.join(" ")];
if (!reply) process.exit(3);
fs.appendFileSync(path.join(__dirname, "queries.jsonl"), JSON.stringify(args) + "\\n");
process.stdout.write(JSON.stringify(reply));\n`,
);
const result = spawnSync(
process.execPath,
[CODEX_NPM_PLUGIN_LIVE_ASSERTIONS_SCRIPT, "assert-audit", "PRIVATE_REPLY_MARKER"],
{
encoding: "utf8",
env: {
...process.env,
HOME: path.join(root, "home"),
OPENCLAW_STATE_DIR: path.join(root, "state"),
OPENCLAW_E2E_CLI_BIN: cliPath,
NODE_OPTIONS: nodeOptionsWithoutExperimentalWarnings(),
},
},
);
expect(result.status, result.stderr).toBe(0);
expect(result.stdout).toContain('codex_audit_identity: {"executionCount":3}');
expect(
readFileSync(path.join(root, "queries.jsonl"), "utf8")
.trim()
.split("\n")
.map((line) => JSON.parse(line)),
).toEqual(
fixture.selectors.flatMap((selector) => [
["--run", selector.runId, "--explain", "--limit", "50"],
["--execution", selector.executionId, "--explain", "--limit", "100"],
]),
);
});
});

View file

@ -22,6 +22,7 @@ type Scenario = {
sourcePlugin?: boolean;
helpFailure?: "exit" | "timeout";
failProbe?: number;
dirtyState?: boolean;
};
function sha256(file: string): string {
@ -82,6 +83,10 @@ function runScenario(scenario: Scenario = {}) {
const bundled = scenario.bundled ?? channel === "telegram";
mkdirSync(bin);
mkdirSync(home);
mkdirSync(join(home, ".openclaw"));
if (scenario.dirtyState) {
writeFileSync(join(home, ".openclaw", "existing-state"), "unrelated fixture state");
}
mkdirSync(packageRoot);
writeFileSync(eventsPath, "");
if (bundled) {
@ -124,6 +129,8 @@ if (help) {
} else if (args[0] === "channels" && args[1] === "add" && env.BUNDLED === "0") {
if (current && !fs.existsSync(dependencyPath())) fail("external channel needs consent first");
if (!current) installChannelDependency();
} else if (args[0] === "identity") {
console.log("execution-fixture");
}
function dependencyPath() {
const dep = { telegram: "grammy", discord: "discord-api-types", slack: "@slack/bolt" }[env.OPENCLAW_NPM_ONBOARD_CHANNEL];
@ -143,6 +150,13 @@ function installChannelDependency() {
if [ "$1" = scripts/e2e/lib/npm-onboard-channel-agent/assertions.mjs ]; then
exec "$REAL_NODE" "$FIXTURE_CLI" assertion "\${@:2}"
fi
if [ "$1" = scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs ]; then
if [ "$2" = clean-home ]; then exec "$REAL_NODE" "$@"; fi
if [ "$2" = run-id ]; then printf '%s' admitted-run-fixture; exit 0; fi
# Projection/storage assertions have their own real SQLite support tests.
[ "$2" = verify ] || exit 0
exec "$REAL_NODE" "$FIXTURE_CLI" identity "\${@:2}"
fi
exec "$REAL_NODE" "$@"
`,
{ mode: 0o755 },
@ -156,10 +170,17 @@ exec "$REAL_NODE" "$@"
throw new Error("npm onboarding container program not found");
}
const testState = `
OPENCLAW_TEST_STATE_HOME="$HOME"
export OPENCLAW_STATE_DIR="$HOME/.openclaw"
export OPENCLAW_CONFIG_PATH="$OPENCLAW_STATE_DIR/openclaw.json"
openclaw_e2e_install_package() { mkdir -p "$HOME/.openclaw"; }
openclaw_e2e_package_root() { printf '%s' "$PACKAGE_ROOT"; }
openclaw_e2e_package_entrypoint() { printf '%s/openclaw.mjs' "$PACKAGE_ROOT"; }
openclaw_e2e_start_mock_openai() { :; }
openclaw_e2e_wait_mock_openai() { :; }
openclaw_e2e_start_gateway() { "$FIXTURE_CLI" gateway-start; printf '%s' fixture-gateway; }
openclaw_e2e_wait_gateway_ready() { :; }
openclaw_e2e_stop_process() { if [ -n "$1" ]; then "$FIXTURE_CLI" gateway-stop; fi; }
`;
const registryEnv = scenario.registry ? registryFixture(root, scenario) : {};
if (scenario.corruptRegistry) {
@ -209,6 +230,40 @@ openclaw_e2e_wait_mock_openai() { :; }
}
describe("npm onboarding fixture consent", () => {
it("inspects the admitted local turn with the installed CLI across Gateway restart", () => {
const { result, events, detail } = runScenario();
expect(result.status, detail).toBe(0);
const optIn = events.findIndex(
(args) => args.join(" ") === "config set logging.audit.executionIdentity true",
);
const turn = events.findIndex((args) => args[0] === "agent");
expect(optIn).toBeGreaterThanOrEqual(0);
expect(optIn).toBeLessThan(turn);
expect(events.filter((args) => args[0] === "agent")).toHaveLength(1);
expect(
events
.slice(turn + 1)
.filter((args) =>
["gateway-start", "gateway-stop", "audit"].some((name) => name === args[0]),
)
.map((args) => args.slice(0, 3)),
).toEqual([
["gateway-start"],
["audit", "--run", "admitted-run-fixture"],
["gateway-stop"],
["gateway-start"],
["audit", "--execution", "execution-fixture"],
["gateway-stop"],
]);
});
it("rejects inherited state before installing or admitting a turn", () => {
const { result, events, detail } = runScenario({ dirtyState: true });
expect(result.status).not.toBe(0);
expect(detail).toContain("package proof inherited existing state");
expect(events).toEqual([]);
});
it.each([false, true])("selects the reviewed Codex source with registry=%s", (registry) => {
const { result, events, installs, detail } = runScenario({ registry });
expect(result.status, detail).toBe(0);

View file

@ -0,0 +1,169 @@
import { spawnSync } from "node:child_process";
import { existsSync, mkdirSync, readdirSync, writeFileSync } from "node:fs";
import path from "node:path";
import { DatabaseSync } from "node:sqlite";
import { afterEach, describe, expect, it } from "vitest";
import {
assertIdentityProjection,
readIdentityRows,
} from "../../scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs";
import { useAutoCleanupTempDirTracker } from "../helpers/temp-dir.js";
const tempDirs = useAutoCleanupTempDirTracker(afterEach);
const helper = path.resolve("scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs");
const domainRef = `hmac-sha256:v1:${"a".repeat(32)}:${"b".repeat(64)}`;
const privateMarker = "PRIVATE_PACKAGE_AUDIT_FIXTURE";
function projection() {
return {
run: { runId: "admitted-run-1", executionId: "execution-1" },
identity: {
state: "present",
context: {
schemaVersion: 1,
contextId: "context-1",
executionId: "execution-1",
runId: "admitted-run-1",
createdAt: 123,
trustDomain: { kind: "gateway-cell", domainRef, state: "present" },
invoker: { state: "absent" },
ingress: { kind: "local-cli", boundary: "agent-command.local", state: "present" },
agentPrincipal: { kind: "agent", domainRef, principalRef: "main" },
agentDefinition: { definitionRef: "main", state: "present" },
runtimeInstance: { runtimeRef: domainRef, kind: "gateway", state: "present" },
applicableGrants: [],
assurance: [],
coverageState: "unattributed",
missingEvidence: ["invoker.principal"],
},
},
decisionDisplays: [
{
provenance: { state: "verified", producer: "run-admission" },
decision: { outcome: "not-applicable", reasonCode: "run_admission_identity_not_evaluated" },
enforcement: { coverageState: "unattributed" },
},
],
};
}
describe("installed-package execution identity proof", () => {
it("does not initialize missing audit state during inspection", () => {
const state = tempDirs.make("openclaw-package-audit-empty-");
expect(readIdentityRows(state)).toEqual([]);
expect(readdirSync(state)).toEqual([]);
expect(existsSync(path.join(state, "state/openclaw.sqlite"))).toBe(false);
});
it("accepts the persisted run id when it differs from the caller session id", () => {
const result = projection();
expect(
assertIdentityProjection(result, JSON.stringify(result.identity.context), [privateMarker]),
).toBe("execution-1");
});
it.each([
[
"missing identity",
(result: ReturnType<typeof projection>) => {
result.identity.state = "unknown";
},
],
[
"wrong run",
(result: ReturnType<typeof projection>) => {
result.run.runId = "other-run";
},
],
[
"fabricated invoker",
(result: ReturnType<typeof projection>) => {
result.identity.context.invoker.state = "present";
},
],
[
"raw runtime",
(result: ReturnType<typeof projection>) => {
result.identity.context.runtimeInstance.runtimeRef = "raw-runtime";
},
],
[
"false enforcement",
(result: ReturnType<typeof projection>) => {
for (const display of result.decisionDisplays) {
display.enforcement.coverageState = "enforced";
}
},
],
[
"lost admission",
(result: ReturnType<typeof projection>) => {
result.decisionDisplays = [];
},
],
] as const)("rejects %s", (_label, mutate) => {
const result = projection();
mutate(result);
expect(() =>
assertIdentityProjection(result, JSON.stringify(result.identity.context), []),
).toThrow();
});
it.each(["export", "storage", "receipt"])("rejects a private canary in %s", (where) => {
const result = projection();
const stored = JSON.stringify(result.identity.context);
const exported =
where === "export"
? { ...result, prompt: privateMarker }
: where === "receipt"
? { ...result, decisions: [{ body: privateMarker }] }
: result;
expect(() =>
assertIdentityProjection(exported, where === "storage" ? stored + privateMarker : stored, [
privateMarker,
]),
).toThrow("execution identity exposed private fixture data");
});
it("rejects identity replacement and duplicate rows after the Gateway restart", () => {
const home = tempDirs.make("openclaw-package-audit-restart-");
const stateDir = path.join(home, ".openclaw");
mkdirSync(path.join(stateDir, "state"), { recursive: true });
const config = path.join(stateDir, "openclaw.json");
writeFileSync(config, "{}");
const beforePath = path.join(home, "before.json");
const afterPath = path.join(home, "after.json");
const before = projection();
const db = new DatabaseSync(path.join(stateDir, "state/openclaw.sqlite"));
const verify = () =>
spawnSync(process.execPath, [helper, "verify", afterPath, beforePath], {
encoding: "utf8",
env: {
HOME: home,
OPENCLAW_HOME: home,
OPENCLAW_TEST_STATE_HOME: home,
OPENCLAW_STATE_DIR: stateDir,
OPENCLAW_CONFIG_PATH: config,
},
});
try {
db.exec("CREATE TABLE execution_identity_contexts (context_json TEXT NOT NULL)");
const insert = db.prepare("INSERT INTO execution_identity_contexts VALUES (?)");
insert.run(JSON.stringify(before.identity.context));
writeFileSync(beforePath, JSON.stringify(before));
writeFileSync(afterPath, JSON.stringify(before));
expect(verify().status).toBe(0);
const after = projection();
after.identity.context.runtimeInstance.runtimeRef = domainRef.replace(/b/gu, "c");
db.prepare("UPDATE execution_identity_contexts SET context_json = ?").run(
JSON.stringify(after.identity.context),
);
writeFileSync(afterPath, JSON.stringify(after));
expect(verify().stderr).toContain("CLI context differs from persisted bytes");
insert.run(JSON.stringify(after.identity.context));
expect(verify().stderr).toContain("expected exactly one admitted execution identity");
} finally {
db.close();
}
});
});

View file

@ -227,6 +227,7 @@ export const databaseWorkerCoreTestFiles = [
"src/audit/audit-event-store.message.test.ts",
"src/audit/audit-event-writer.test.ts",
"src/audit/audit-event-writer.worker.test.ts",
"src/audit/execution-decision-cursors.test.ts",
"src/audit/execution-decision-work.test.ts",
"src/audit/execution-identity-context.test.ts",
"src/agents/tools/sessions-access.test.ts",