diff --git a/.agents/skills/telegram-e2e-userbot/features/runtime-reference.md b/.agents/skills/telegram-e2e-userbot/features/runtime-reference.md index 4495e1874d6e..d3ade85f7623 100644 --- a/.agents/skills/telegram-e2e-userbot/features/runtime-reference.md +++ b/.agents/skills/telegram-e2e-userbot/features/runtime-reference.md @@ -76,6 +76,33 @@ Keep one TDLib client per restored state directory. Run custom TDLib inspection before the recorder starts or after it exits, under the same live lease. Bot API inspection can use the scenario `command` action while recording. +## QA Lab participant identity fixtures + +The QA Lab Telegram adapter accepts optional fields on one Convex-leased Test +Server credential: `forumGroupId`, positive numeric `forumTopicId`, and +`participants`. Each additional participant supplies a unique lowercase `alias`, +`testerUserId`, `tdlibArchiveBase64`, `tdlibArchiveSha256`, and `tdlibVersion`. +The pool owner provisions these independently authorized users under the same +lease. The SUT bot and every participant must already belong to the selected +group and forum, and the topic must exist. No credential is acquired by merely +listing scenarios or running deterministic support tests. + +For mixed-user flows, use `senderId: primary` and the additional aliases. A +single-user fixture binds its first scenario sender label to its leased user; +changing that label cannot impersonate a second person. `conversation.kind: +direct` sends to the SUT DM. A group/channel conversation uses `groupId`; adding +a positive numeric `threadId` selects that forum topic in `forumGroupId` (or the +existing group when it is itself a forum). Each native chat/topic belongs to one +logical conversation until transport reset. Replies require a receipt observed +by the sending participant in that chat; TDLib message IDs cannot cross accounts. + +Flow preparation exposes in-memory `telegramIdentityFixture.participantAliases` +and `forumTopicId`. Set `execution.config.requireParticipantIdentityFixture: +true` to require the complete mixed-user/forum fixture before sending. It also +retains the existing `readTelegramMessages()` observer for native topic evidence. +These inputs enable real Telegram identity proof; deterministic adapter tests do +not claim that live transport or audit inspection has run. + ## Backends | Backend | Use | diff --git a/.agents/skills/telegram-e2e-userbot/scripts/telegram-test-credential.mjs b/.agents/skills/telegram-e2e-userbot/scripts/telegram-test-credential.mjs index b3d0e6368c06..24c7ebb8fa0b 100755 --- a/.agents/skills/telegram-e2e-userbot/scripts/telegram-test-credential.mjs +++ b/.agents/skills/telegram-e2e-userbot/scripts/telegram-test-credential.mjs @@ -50,7 +50,7 @@ export function parseTelegramTestCredential(value) { } const tdlibArchiveBase64 = requireString(payload, "tdlibArchiveBase64"); decodeBase64(tdlibArchiveBase64); - return { + const credential = { schemaVersion: 1, environment: "test", groupId: requireIntegerString(payload, "groupId", /^-?\d+$/u), @@ -62,6 +62,49 @@ export function parseTelegramTestCredential(value) { tdlibArchiveSha256, tdlibVersion: requireString(payload, "tdlibVersion"), }; + if (payload.forumGroupId !== undefined) { + credential.forumGroupId = requireIntegerString(payload, "forumGroupId", /^-\d+$/u); + } + if (payload.forumTopicId !== undefined) { + if (!Number.isSafeInteger(payload.forumTopicId) || payload.forumTopicId <= 0) { + throw new Error("Telegram QA forumTopicId must be a positive integer."); + } + credential.forumTopicId = payload.forumTopicId; + } + if (payload.participants !== undefined) { + if (!Array.isArray(payload.participants)) { + throw new Error("Telegram QA participants must be an array."); + } + const identities = new Set([credential.testerUserId]); + const aliases = new Set(["primary"]); + credential.participants = payload.participants.map((value) => { + const participant = requireObject(value, "Telegram QA participant"); + const alias = requireString(participant, "alias"); + if (!/^[a-z][a-z0-9-]*$/u.test(alias) || aliases.has(alias)) { + throw new Error("Telegram QA participants require distinct lowercase aliases."); + } + const parsed = parseTelegramTestCredential({ + ...credential, + testerUserId: participant.testerUserId, + tdlibArchiveBase64: participant.tdlibArchiveBase64, + tdlibArchiveSha256: participant.tdlibArchiveSha256, + tdlibVersion: participant.tdlibVersion, + }); + if (identities.has(parsed.testerUserId)) { + throw new Error("Telegram QA mixed participants require distinct leased user identities."); + } + aliases.add(alias); + identities.add(parsed.testerUserId); + return { + alias, + testerUserId: parsed.testerUserId, + tdlibArchiveBase64: parsed.tdlibArchiveBase64, + tdlibArchiveSha256: parsed.tdlibArchiveSha256, + tdlibVersion: parsed.tdlibVersion, + }; + }); + } + return credential; } function normalizeArchiveEntry(entry) { diff --git a/.agents/skills/telegram-e2e-userbot/scripts/telegram-test-credential.test.mjs b/.agents/skills/telegram-e2e-userbot/scripts/telegram-test-credential.test.mjs index b7860da0ee16..877dc88930de 100644 --- a/.agents/skills/telegram-e2e-userbot/scripts/telegram-test-credential.test.mjs +++ b/.agents/skills/telegram-e2e-userbot/scripts/telegram-test-credential.test.mjs @@ -218,3 +218,52 @@ test("failed broker release retains only the private handle for same-owner recov fs.rmSync(fixture, { recursive: true, force: true }); } }); + +test("validates optional forum and distinct participant fixtures without echoing private fields", () => { + const base = { + schemaVersion: 1, + environment: "test", + groupId: "-1001", + sutToken: "synthetic-token", + sutUsername: "sut_bot", + sutBotId: "200", + testerUserId: "100", + tdlibArchiveBase64: "YQ==", + tdlibArchiveSha256: "a".repeat(64), + tdlibVersion: "1.8.67", + }; + const second = { + alias: "second", + testerUserId: "101", + tdlibArchiveBase64: "Yg==", + tdlibArchiveSha256: "b".repeat(64), + tdlibVersion: "1.8.67", + }; + assert.deepEqual( + parseTelegramTestCredential({ + ...base, + forumGroupId: "-1002", + forumTopicId: 42, + participants: [second], + }).participants, + [second], + ); + for (const patch of [ + { participants: [second, second] }, + { participants: [{ ...second, testerUserId: "100" }] }, + { participants: [{ ...second, alias: "primary" }] }, + { participants: [{ ...second, tdlibArchiveBase64: "private-invalid-value" }] }, + { participants: [{ ...second, tdlibArchiveSha256: "private-invalid-value" }] }, + { forumGroupId: "1002" }, + { forumTopicId: 0 }, + ]) { + assert.throws( + () => parseTelegramTestCredential({ ...base, ...patch }), + (error) => { + assert.equal(error.message.includes("private-invalid-value"), false); + assert.equal(error.message.includes(base.sutToken), false); + return true; + }, + ); + } +}); diff --git a/.agents/skills/telegram-e2e-userbot/scripts/user-driver.py b/.agents/skills/telegram-e2e-userbot/scripts/user-driver.py index b3710fbfcc8f..c31288c43786 100755 --- a/.agents/skills/telegram-e2e-userbot/scripts/user-driver.py +++ b/.agents/skills/telegram-e2e-userbot/scripts/user-driver.py @@ -523,6 +523,19 @@ class UserDriver: elif state == "authorizationStateWaitPassword": password = getattr(args, "password", "") or prompt_secret("Telegram 2FA password: ") self.client.send({"@type": "checkAuthenticationPassword", "password": password}) + elif state == "authorizationStateWaitRegistration": + first_name = getattr(args, "first_name", "").strip() + if not first_name: + raise DriverError( + "New Telegram Test Server users require login --first-name." + ) + self.client.send( + { + "@type": "registerUser", + "first_name": first_name, + "last_name": getattr(args, "last_name", "").strip(), + } + ) elif state == "authorizationStateReady": return True elif state in {"authorizationStateClosing", "authorizationStateClosed", "authorizationStateLoggingOut"}: @@ -1112,6 +1125,200 @@ def cleanup_owned_group(driver, manifest_path): return {"ok": True, **record} +def private_forum_identity(driver): + if driver.config.get("testDc") is True: + raise DriverError("Private production forum setup requires Telegram production.") + me = driver.client.request({"@type": "getMe"}) + if str(me["id"]) != str(driver.config.get("testerUserId")): + raise DriverError("Private production forum setup requires the leased QA user.") + sut = resolve_sut(driver.config, driver.bot_config) + if not sut["id"] or not sut["username"]: + raise DriverError("Private production forum setup requires the SUT bot identity.") + return { + "testerUserId": str(me["id"]), + "sutBotId": str(sut["id"]), + "sutUsername": sut["username"], + } + + +def public_private_forum_record(record): + return { + key: record[key] + for key in ("ok", "status", "groupId", "forumTopicId", "title", "topicTitle") + if key in record + } + + +def prepare_private_forum(driver, manifest_path): + identity = private_forum_identity(driver) + if manifest_path.exists(): + raise DriverError( + "This lease already has a private forum record; clean it before another creation." + ) + record = { + **identity, + "status": "creating", + "title": f"OpenClaw private QA {secrets.token_hex(6)}", + "topicTitle": f"Identity proof {secrets.token_hex(4)}", + } + write_json_private(manifest_path, record) + try: + basic = driver.client.request( + { + "@type": "createNewBasicGroupChat", + "user_ids": [], + "title": record["title"], + "message_auto_delete_time": 0, + } + ) + record.update(status="basic-group-created", basicGroupId=str(basic["chat_id"])) + write_json_private(manifest_path, record) + upgraded = driver.client.request( + { + "@type": "upgradeBasicGroupChatToSupergroupChat", + "chat_id": int(record["basicGroupId"]), + } + ) + if ( + upgraded["type"].get("@type") != "chatTypeSupergroup" + or upgraded["type"].get("is_channel") is True + ): + raise DriverError("Telegram did not upgrade the private fixture to a supergroup.") + record.update( + status="supergroup-created", + groupId=str(upgraded["id"]), + supergroupId=str(upgraded["type"]["supergroup_id"]), + ) + write_json_private(manifest_path, record) + driver.client.request( + { + "@type": "setChatMemberStatus", + "chat_id": int(record["groupId"]), + "member_id": { + "@type": "messageSenderUser", + "user_id": int(identity["sutBotId"]), + }, + "status": { + "@type": "chatMemberStatusMember", + "member_until_date": 0, + }, + } + ) + record["status"] = "bot-added" + write_json_private(manifest_path, record) + driver.client.request( + { + "@type": "toggleSupergroupIsForum", + "supergroup_id": int(record["supergroupId"]), + "is_forum": True, + "has_forum_tabs": True, + } + ) + record["status"] = "forum-enabled" + write_json_private(manifest_path, record) + topic = driver.client.request( + { + "@type": "createForumTopic", + "chat_id": int(record["groupId"]), + "name": record["topicTitle"], + "is_name_implicit": False, + "icon": { + "@type": "forumTopicIcon", + "color": 0x6FB9F0, + "custom_emoji_id": 0, + }, + } + ) + record.update(status="topic-created", forumTopicId=int(topic["forum_topic_id"])) + write_json_private(manifest_path, record) + invite = driver.client.request( + { + "@type": "createChatInviteLink", + "chat_id": int(record["groupId"]), + "name": "OpenClaw private qualification", + "expiration_date": 0, + "member_limit": 1, + "creates_join_request": False, + } + ) + record.update(status="ready", inviteLink=invite["invite_link"], ok=True) + write_json_private(manifest_path, record) + return public_private_forum_record(record) + except (DriverError, KeyError, TypeError, ValueError): + record["status"] = "setup-failed" + write_json_private(manifest_path, record) + raise + + +def cleanup_private_forum(driver, manifest_path): + record = read_json(manifest_path) + if not record: + return {"ok": True, "status": "not-created"} + identity = private_forum_identity(driver) + if any(record.get(key) != identity[key] for key in ("testerUserId", "sutBotId")): + raise DriverError("Private forum cleanup record belongs to a different identity.") + if record.get("status") == "deleted": + return public_private_forum_record(record) + group_id = record.get("groupId") or record.get("basicGroupId") + if not group_id: + record.pop("inviteLink", None) + record.update(status="not-created", ok=True) + write_json_private(manifest_path, record) + return public_private_forum_record(record) + chat = driver.client.request({"@type": "getChat", "chat_id": int(group_id)}) + membership = driver.client.request( + { + "@type": "getChatMember", + "chat_id": int(group_id), + "member_id": { + "@type": "messageSenderUser", + "user_id": int(identity["testerUserId"]), + }, + } + ) + if ( + chat["type"].get("@type") not in {"chatTypeBasicGroup", "chatTypeSupergroup"} + or chat["type"].get("is_channel") is True + or membership["status"].get("@type") != "chatMemberStatusCreator" + ): + raise DriverError("The leased QA user cannot delete the private forum for all members.") + # The can_be_deleted projection can lag immediately after creation. The + # creator-owned delete is authoritative; retry only its observed propagation error. + for attempt in range(5): + try: + deletion = driver.client.request( + {"@type": "deleteChat", "chat_id": int(group_id)} + ) + break + except TdRequestError as error: + if "The chat can't be deleted" not in str(error) or attempt == 4: + raise + time.sleep(1) + record.pop("inviteLink", None) + record.update(status="deleted", deletion=deletion, ok=True) + write_json_private(manifest_path, record) + return public_private_forum_record(record) + + +def command_private_forum(args): + if not os.environ.get("TELEGRAM_USER_DRIVER_STATE_DIR"): + raise DriverError("Private forum commands require runner-owned leased credential state.") + manifest = STATE_DIR / "owned-private-forum.json" + if args.command == "cleanup-private-forum" and not manifest.exists(): + print_result({"ok": True, "status": "not-created"}, args.json, args.output) + return + config, bot_config = load_config() + driver = UserDriver(config, bot_config) + if not driver.authorize(args, need_ready=False): + raise DriverError("The leased QA user is not authorized.") + result = ( + prepare_private_forum(driver, manifest) + if args.command == "prepare-private-forum" + else cleanup_private_forum(driver, manifest) + ) + print_result(result, args.json, args.output) + + def command_test_group(args): if not os.environ.get("TELEGRAM_USER_DRIVER_STATE_DIR"): raise DriverError("Test group commands require runner-owned leased credential state.") @@ -1315,6 +1522,8 @@ def serve_message(message, users): "senderId": normalized.get("senderId"), "senderUsername": normalized.get("senderUsername"), "replyToMessageId": normalized.get("replyToMessageId"), + "threadId": normalized.get("threadId"), + "forumTopicId": (message.get("topic_id") or {}).get("forum_topic_id"), "timestamp": int(message.get("date") or 0) * 1000, "contentType": normalized["contentType"], "text": normalized["text"], @@ -1425,6 +1634,11 @@ def command_serve(args): chat_id = driver.resolve_chat(args.chat) tester = driver.client.request({"@type": "getMe"}) driver.check_group_write_access(chat_id, tester["id"]) + selected_chats = {chat_id} + for target in getattr(args, "observe_chat", []): + selected = driver.resolve_chat(target) + driver.check_group_write_access(selected, tester["id"]) + selected_chats.add(selected) write_ndjson( { "type": "ready", @@ -1435,7 +1649,7 @@ def command_serve(args): known_messages = {} while True: update = driver.client.next_update(timeout=0.1) - if update and update_chat_id(update) == chat_id: + if update and update_chat_id(update) in selected_chats: event = serve_update(update, driver.client, known_messages) if event: write_ndjson({"type": "update", "update": event}) @@ -1451,13 +1665,27 @@ def command_serve(args): request_id = request.get("id") if not isinstance(request_id, str) or not request_id: raise DriverError("serve command requires a string id") - if request.get("method") != "send": + method = request.get("method") + if method == "cleanup-private-forum": + result = cleanup_private_forum( + driver, STATE_DIR / "owned-private-forum.json" + ) + write_ndjson({"type": "response", "id": request_id, "result": result}) + continue + if method != "send": raise DriverError("serve command method must be send") text = request.get("text") if not isinstance(text, str) or not text: raise DriverError("serve send command requires text") reply_to = request.get("replyToMessageId") - sent = driver.send_text(chat_id, text, reply_to=reply_to) + target = driver.resolve_chat(request["chatId"]) if "chatId" in request else chat_id + if target not in selected_chats: + driver.check_group_write_access(target, tester["id"]) + selected_chats.add(target) + topic = request.get("forumTopicId") + if topic is not None and (type(topic) is not int or topic <= 0): + raise DriverError("serve forumTopicId must be a positive integer") + sent = driver.send_text(target, text, reply_to=reply_to, forum_topic_id=topic) normalized = serve_message(sent, driver.client.users) known_messages[(normalized["chatId"], normalized["messageId"])] = normalized write_ndjson({"type": "response", "id": request_id, "result": normalized}) @@ -1511,6 +1739,8 @@ def main(): login.add_argument("--phone", default="") login.add_argument("--code", default="") login.add_argument("--password", default="") + login.add_argument("--first-name", default="") + login.add_argument("--last-name", default="") login.set_defaults(func=command_login) status = sub.add_parser("status") @@ -1530,6 +1760,11 @@ def main(): group.add_argument("--chat", default="") group.set_defaults(func=command_test_group) + for name in ("prepare-private-forum", "cleanup-private-forum"): + private_forum = sub.add_parser(name) + add_common(private_forum) + private_forum.set_defaults(func=command_private_forum) + confirm_qr = sub.add_parser("confirm-qr") add_common(confirm_qr) confirm_qr.add_argument("--link", required=True) @@ -1583,6 +1818,7 @@ def main(): serve = sub.add_parser("serve") serve.add_argument("--chat", default="") + serve.add_argument("--observe-chat", action="append", default=[]) serve.add_argument("--timeout-ms", type=int, default=120000) serve.set_defaults(func=command_serve) diff --git a/.agents/skills/telegram-e2e-userbot/scripts/user-driver.test.py b/.agents/skills/telegram-e2e-userbot/scripts/user-driver.test.py index d6530d1a234c..a297b72ec1c8 100644 --- a/.agents/skills/telegram-e2e-userbot/scripts/user-driver.test.py +++ b/.agents/skills/telegram-e2e-userbot/scripts/user-driver.test.py @@ -42,6 +42,59 @@ def native_message(content, chat_id=-1001, message_id=42 << 20, sender_id=101): } +class AuthorizationTest(unittest.TestCase): + def test_registers_a_new_test_user_with_explicit_names(self): + class Client: + def __init__(self): + self.requests = [] + self.updates = [ + { + "@type": "updateAuthorizationState", + "authorization_state": { + "@type": "authorizationStateWaitRegistration", + }, + }, + { + "@type": "updateAuthorizationState", + "authorization_state": {"@type": "authorizationStateReady"}, + }, + ] + + def execute(self, payload): + self.requests.append(payload) + + def send(self, payload): + self.requests.append(payload) + + def receive(self, _timeout): + return self.updates.pop(0) + + instance = driver.UserDriver.__new__(driver.UserDriver) + instance.client = Client() + instance.printed_qr_link = None + instance.config = {} + instance.bot_config = {} + instance.authorize( + SimpleNamespace( + timeout_ms=1000, + phone="", + qr=False, + code="", + password="", + first_name="OpenClaw", + last_name="Guest", + ), + ) + self.assertIn( + { + "@type": "registerUser", + "first_name": "OpenClaw", + "last_name": "Guest", + }, + instance.client.requests, + ) + + class OwnedGroupTest(unittest.TestCase): def fixture(self): class Client: @@ -101,6 +154,110 @@ class OwnedGroupTest(unittest.TestCase): driver.prepare_owned_group(instance, Path(root) / "group.json") self.assertEqual(instance.client.requests, []) + def test_private_production_forum_is_owned_and_deleted(self): + instance = self.fixture() + instance.config["testDc"] = False + + def request(payload, timeout=20): + instance.client.requests.append(payload) + kind = payload["@type"] + if kind == "getMe": + return {"id": 123, "username": "leased_primary"} + if kind == "createNewBasicGroupChat": + self.assertEqual(payload["user_ids"], []) + return {"chat_id": -2042, "failed_to_add_members": {"failed_to_add_members": []}} + if kind == "upgradeBasicGroupChatToSupergroupChat": + self.assertEqual(payload["chat_id"], -2042) + return {"id": -1002042, "type": {"@type": "chatTypeSupergroup", "supergroup_id": 2042, "is_channel": False}} + if kind == "setChatMemberStatus": + self.assertEqual(payload["chat_id"], -1002042) + self.assertEqual(payload["member_id"]["user_id"], 42) + self.assertEqual(payload["status"]["@type"], "chatMemberStatusMember") + self.assertEqual(payload["status"]["member_until_date"], 0) + return {"@type": "ok"} + if kind == "toggleSupergroupIsForum": + self.assertEqual(payload, {"@type": kind, "supergroup_id": 2042, "is_forum": True, "has_forum_tabs": True}) + return {"@type": "ok"} + if kind == "createForumTopic": + self.assertEqual(payload["chat_id"], -1002042) + return {"forum_topic_id": 777, "name": payload["name"]} + if kind == "createChatInviteLink": + return {"invite_link": "https://example.invalid/private-invite"} + if kind == "getChat": + return { + "id": -1002042, + "type": {"@type": "chatTypeSupergroup", "is_channel": False}, + # The server can lag this projection immediately after creation. + "can_be_deleted_for_all_users": False, + } + if kind == "getChatMember": + return {"status": {"@type": "chatMemberStatusCreator"}} + if kind == "deleteChat": + return {"@type": "ok"} + raise AssertionError(kind) + + instance.client.request = request + with tempfile.TemporaryDirectory() as root: + manifest = Path(root) / "private-forum.json" + prepared = driver.prepare_private_forum(instance, manifest) + self.assertEqual(prepared["groupId"], "-1002042") + self.assertEqual(prepared["forumTopicId"], 777) + self.assertNotIn("inviteLink", prepared) + self.assertIn("inviteLink", driver.read_json(manifest)) + + cleaned = driver.cleanup_private_forum(instance, manifest) + self.assertEqual(cleaned["status"], "deleted") + self.assertNotIn("inviteLink", driver.read_json(manifest)) + + self.assertEqual( + [request["@type"] for request in instance.client.requests], + [ + "getMe", + "createNewBasicGroupChat", + "upgradeBasicGroupChatToSupergroupChat", + "setChatMemberStatus", + "toggleSupergroupIsForum", + "createForumTopic", + "createChatInviteLink", + "getMe", + "getChat", + "getChatMember", + "deleteChat", + ], + ) + + def test_private_forum_cleanup_deletes_an_interrupted_basic_group(self): + instance = self.fixture() + instance.config["testDc"] = False + instance.client.members = {123} + request = instance.client.request + + def request_as_creator(payload, timeout=20): + if payload["@type"] == "getChatMember": + instance.client.requests.append(payload) + return {"status": {"@type": "chatMemberStatusCreator"}} + return request(payload, timeout) + + instance.client.request = request_as_creator + with tempfile.TemporaryDirectory() as root: + manifest = Path(root) / "private-forum.json" + driver.write_json_private( + manifest, + { + "basicGroupId": "-2042", + "status": "basic-group-created", + "sutBotId": "42", + "testerUserId": "123", + }, + ) + cleaned = driver.cleanup_private_forum(instance, manifest) + self.assertEqual(cleaned["status"], "deleted") + + self.assertEqual( + [request["@type"] for request in instance.client.requests], + ["getMe", "getChat", "getChatMember", "deleteChat"], + ) + def test_existing_group_membership_is_repaired_and_only_added_bot_leaves(self): instance = self.fixture() instance.client.members = {123, 777} @@ -762,6 +919,45 @@ class RichObservationTest(unittest.TestCase): self.assertEqual(known[(-2002, 42 << 20)]["senderId"], 202) +class ServeTargetTest(unittest.TestCase): + def test_serves_dm_group_and_forum_without_losing_observed_topic(self): + client = observation_client() + client.request.side_effect = None + client.request.return_value = {"id": 303} + client.next_update = lambda timeout: None + def send(chat_id, text, reply_to=None, forum_topic_id=None): + message = native_message({"@type": "messageText", "text": {"text": text, "entities": []}}, chat_id=chat_id, sender_id=303) + if forum_topic_id is not None: + message["topic_id"] = {"@type": "messageTopicForum", "forum_topic_id": forum_topic_id} + return message + instance = SimpleNamespace(client=client, authorize=lambda *_: None, + resolve_chat=lambda value: int(value), check_group_write_access=Mock(return_value=True), + send_text=Mock(side_effect=send)) + commands = [ + {"id": "1", "method": "send", "text": "dm", "chatId": "200"}, + {"id": "2", "method": "send", "text": "group"}, + {"id": "3", "method": "send", "text": "forum", "chatId": "-2002", "forumTopicId": 42}, + {"id": "4", "method": "cleanup-private-forum"}, + {"id": "5", "method": "send", "text": "invalid", "forumTopicId": True}, + ] + events = [] + with patch.object(driver, "load_config", return_value=({}, {})), \ + patch.object(driver, "UserDriver", return_value=instance), \ + patch.object(driver, "cleanup_private_forum", return_value={"ok": True, "status": "deleted"}) as cleanup, \ + patch.object(driver, "write_ndjson", side_effect=events.append), \ + patch.object(driver.sys, "stdin", io.StringIO("\n".join(driver.json.dumps(value) for value in commands))), \ + patch.object(driver.select, "select", side_effect=lambda *_: ([driver.sys.stdin], [], [])): + driver.command_serve(SimpleNamespace(chat="-1001", observe_chat=["-2002"], timeout_ms=1000)) + sent = [event["result"] for event in events if "result" in event and "chatId" in event["result"]] + self.assertEqual([value["chatId"] for value in sent], [200, -1001, -2002]) + self.assertEqual([value["senderId"] for value in sent], [303, 303, 303]) + self.assertEqual(sent[-1]["forumTopicId"], 42) + self.assertIn("positive integer", events[-1]["error"]) + self.assertEqual(instance.send_text.call_count, 3) + cleanup.assert_called_once_with(instance, driver.STATE_DIR / "owned-private-forum.json") + self.assertEqual([call.args[0] for call in instance.check_group_write_access.call_args_list], [-1001, -2002, 200]) + + class GroupWriteAccessTest(unittest.TestCase): def make_driver(self, status, default=True, boosts=None, kind="chatTypeSupergroup", active=True): instance = driver.UserDriver.__new__(driver.UserDriver) diff --git a/config/assertion-safety-baseline.txt b/config/assertion-safety-baseline.txt index 13f5681b2e4f..95120d40e453 100644 --- a/config/assertion-safety-baseline.txt +++ b/config/assertion-safety-baseline.txt @@ -953,7 +953,6 @@ extensions/qa-lab/src/live-transports/slack/slack-live.config.ts 2 extensions/qa-lab/src/live-transports/slack/slack-live.message-observations.ts 1 extensions/qa-lab/src/live-transports/slack/slack-live.observations.ts 7 extensions/qa-lab/src/live-transports/slack/slack-live.scenario-implementations.ts 2 -extensions/qa-lab/src/live-transports/telegram/adapter.runtime.ts 1 extensions/qa-lab/src/live-transports/telegram/telegram-api.runtime.ts 1 extensions/qa-lab/src/live-transports/whatsapp/adapter.runtime.ts 2 extensions/qa-lab/src/live-transports/whatsapp/scenario-environment.ts 3 diff --git a/extensions/qa-lab/src/live-transports/telegram/adapter.runtime.test.ts b/extensions/qa-lab/src/live-transports/telegram/adapter.runtime.test.ts index c3bfc068e53f..d2f0ebfb8d70 100644 --- a/extensions/qa-lab/src/live-transports/telegram/adapter.runtime.test.ts +++ b/extensions/qa-lab/src/live-transports/telegram/adapter.runtime.test.ts @@ -14,11 +14,15 @@ const mocks = vi.hoisted(() => ({ loadTelegramUserbotSkillRuntime: vi.fn(), proxyClose: vi.fn(), proxyDrainUpdates: vi.fn(), + readTelegramPrivateProductionDescriptor: vi.fn(), + requestTelegramPrivateAppTurn: vi.fn(), + resolveTelegramPrivateProductionBot: vi.fn(), restoreCredential: vi.fn(), shouldRetainQaGatewayCredentialLease: vi.fn(), startApiProxy: vi.fn(), userbotAssertHealthy: vi.fn(), userbotClose: vi.fn(), + userbotCleanupPrivateForum: vi.fn(), userbotSend: vi.fn(), userbotStart: vi.fn(), })); @@ -41,10 +45,17 @@ vi.mock("./userbot-driver.runtime.js", () => ({ TelegramUserbotDriver: { start: mocks.userbotStart }, })); +vi.mock("./private-production.runtime.js", () => ({ + readTelegramPrivateProductionDescriptor: mocks.readTelegramPrivateProductionDescriptor, + requestTelegramPrivateAppTurn: mocks.requestTelegramPrivateAppTurn, + resolveTelegramPrivateProductionBot: mocks.resolveTelegramPrivateProductionBot, +})); + vi.mock("./userbot-skill.runtime.js", () => ({ loadTelegramUserbotSkillRuntime: mocks.loadTelegramUserbotSkillRuntime, })); +import { readQaScenarioById } from "../../scenario-catalog.js"; import { createTelegramQaTransportAdapter } from "./adapter.runtime.js"; const credential = { @@ -60,12 +71,28 @@ const credential = { tdlibVersion: "1.8.67", } as const; -async function prepareMessageReader( +const identityCredential = { + ...credential, + forumGroupId: "-100456", + forumTopicId: 42, + participants: [ + { + alias: "second", + testerUserId: "101", + tdlibArchiveBase64: "Yg==", + tdlibArchiveSha256: "b".repeat(64), + tdlibVersion: "1.8.67", + }, + ], +}; + +async function prepareFlow( adapter: Awaited>, + config: Record = {}, ) { const stateRoot = mocks.createStateRoot.mock.results.at(-1)?.value; - const prepared = await adapter.prepareFlow?.({ - config: {}, + return await adapter.prepareFlow?.({ + config, scenarioId: "telegram-entities", scenarioTitle: "Telegram native entities", gateway: { @@ -79,6 +106,12 @@ async function prepareMessageReader( outputDir: stateRoot, timeoutMs: 30_000, }); +} + +async function prepareMessageReader( + adapter: Awaited>, +) { + const prepared = await prepareFlow(adapter); const read = prepared?.readTelegramMessages; if (typeof read !== "function") { throw new Error("Telegram flow did not expose native message observations"); @@ -91,6 +124,7 @@ describe("Telegram QA transport adapter", () => { vi.clearAllMocks(); const stateRoot = fs.mkdtempSync(path.join(os.tmpdir(), "telegram-adapter-test-")); mocks.createStateRoot.mockReturnValue(stateRoot); + mocks.readTelegramPrivateProductionDescriptor.mockReturnValue(undefined); mocks.acquireQaCredentialLease.mockResolvedValue({ payload: credential, source: "convex", @@ -119,6 +153,7 @@ describe("Telegram QA transport adapter", () => { assertHealthy: mocks.userbotAssertHealthy, chatId: -100123, close: mocks.userbotClose, + cleanupPrivateForum: mocks.userbotCleanupPrivateForum, send: mocks.userbotSend, }); mocks.proxyDrainUpdates.mockResolvedValue(undefined); @@ -133,10 +168,11 @@ describe("Telegram QA transport adapter", () => { assertHealthy: mocks.userbotAssertHealthy, chatId: 200, close: mocks.userbotClose, + cleanupPrivateForum: mocks.userbotCleanupPrivateForum, send: mocks.userbotSend, }; }); - mocks.userbotSend.mockResolvedValueOnce({ messageId: 10 }); + mocks.userbotSend.mockResolvedValueOnce({ messageId: 10, senderId: 100, chatId: 200 }); const addInboundMessage = vi.fn().mockResolvedValue({ id: "in-1" }); const addOutboundMessage = vi.fn().mockResolvedValue({ id: "out-1" }); const adapter = await createTelegramQaTransportAdapter({ @@ -277,9 +313,9 @@ describe("Telegram QA transport adapter", () => { mocks.userbotSend .mockImplementationOnce(async () => { await onUpdate?.(preview); - return { messageId: 10 }; + return { messageId: 10, senderId: 100, chatId: -100123 }; }) - .mockResolvedValueOnce({ messageId: 12 }); + .mockResolvedValueOnce({ messageId: 12, senderId: 100, chatId: -100123 }); const addInboundMessage = vi.fn().mockResolvedValue({ id: "in-1" }); const addOutboundMessage = vi.fn().mockResolvedValue({ id: "out-1" }); const editMessage = vi.fn(); @@ -295,6 +331,7 @@ describe("Telegram QA transport adapter", () => { }); expect(mocks.userbotSend).toHaveBeenCalledWith({ text: "@sut_bot reply exactly: QA-MARKER", + chatId: "-100123", replyToMessageId: undefined, }); expect(addInboundMessage).toHaveBeenCalledWith( @@ -321,6 +358,7 @@ describe("Telegram QA transport adapter", () => { }); expect(mocks.userbotSend).toHaveBeenLastCalledWith({ text: "follow-up", + chatId: "-100123", replyToMessageId: 11, }); const edited = { @@ -399,6 +437,307 @@ describe("Telegram QA transport adapter", () => { } }); + it("keeps DM, group, forum, and mixed leased participants distinct", async () => { + mocks.acquireQaCredentialLease.mockResolvedValueOnce({ + payload: identityCredential, + heartbeat: mocks.leaseHeartbeat, + release: mocks.leaseRelease, + }); + const updates: Array<(update: unknown) => Promise> = []; + const sends = [100, 101].map((senderId) => + vi.fn(async (input) => ({ + messageId: 10, + senderId, + chatId: Number(input.chatId), + forumTopicId: input.forumTopicId, + })), + ); + mocks.userbotStart.mockImplementation(async (params) => { + const index = updates.length; + updates.push(params.onUpdate); + return { + assertHealthy: mocks.userbotAssertHealthy, + chatId: -100123, + close: mocks.userbotClose, + send: sends[index], + }; + }); + let nextId = 0; + const addInboundMessage = vi.fn().mockImplementation(async () => ({ id: `in-${++nextId}` })); + const addOutboundMessage = vi.fn().mockImplementation(async () => ({ id: `out-${++nextId}` })); + const editMessage = vi.fn(); + const adapter = await createTelegramQaTransportAdapter({ + adapterOptions: {}, + messages: { addInboundMessage, addOutboundMessage, editMessage }, + } as never); + try { + const scenario = readQaScenarioById("telegram-participant-identity-inspection"); + await expect(prepareFlow(adapter, scenario.execution.config)).resolves.toMatchObject({ + telegramIdentityFixture: { participantAliases: ["primary", "second"], forumTopicId: 42 }, + }); + expect(mocks.userbotStart.mock.calls.map(([input]) => input.expectedUserId)).toEqual([ + "100", + "101", + ]); + expect(mocks.userbotStart).toHaveBeenNthCalledWith( + 1, + expect.objectContaining({ observeChatIds: ["-100456"] }), + ); + expect(adapter.createGatewayConfig({ baseUrl: "http://127.0.0.1:1234" })).toMatchObject({ + channels: { + telegram: { + accounts: { + sut: { + allowFrom: ["100", "101"], + groups: { "-100456": { allowFrom: ["100", "101"] } }, + }, + }, + }, + }, + }); + await adapter.sendInbound({ + conversation: { id: "dm", kind: "direct" }, + senderId: "primary", + text: "dm", + }); + await adapter.sendInbound({ + conversation: { id: "room", kind: "group" }, + senderId: "primary", + text: "group", + }); + await adapter.sendInbound({ + conversation: { id: "room", kind: "group" }, + senderId: "second", + text: "mixed", + }); + await adapter.sendInbound({ + conversation: { id: "forum", kind: "group" }, + threadId: "42", + senderId: "second", + text: "forum", + }); + expect(sends.map((send) => send.mock.calls.map(([input]) => input.chatId))).toEqual([ + ["200", "-100123"], + ["-100123", "-100456"], + ]); + expect(sends[1]).toHaveBeenLastCalledWith({ + chatId: "-100456", + forumTopicId: 42, + text: "forum", + replyToMessageId: undefined, + }); + expect(addInboundMessage.mock.calls.map(([input]) => input.senderId)).toEqual([ + "100", + "100", + "101", + "101", + ]); + const update = { + kind: "message", + messageId: 11, + senderId: 200, + text: "reply", + timestamp: 1000, + entities: [], + }; + // Deliver a late DM reply after two other native conversations have sent. + const [primaryUpdate, secondUpdate] = updates; + if (!primaryUpdate || !secondUpdate) { + throw new Error("Expected both leased participant observers to start."); + } + await primaryUpdate({ ...update, chatId: 200 }); + await primaryUpdate({ ...update, chatId: -100123 }); + await secondUpdate({ ...update, chatId: -100123 }); + await primaryUpdate({ ...update, chatId: -100456, forumTopicId: 42 }); + await primaryUpdate({ ...update, chatId: -100456, forumTopicId: 43 }); + expect(addOutboundMessage.mock.calls.map(([input]) => input.to)).toEqual([ + "dm:dm", + "group:room", + "thread:/v1/group/forum/42", + ]); + await primaryUpdate({ ...update, kind: "edit", chatId: 200, text: "dm edit" }); + expect(editMessage).toHaveBeenCalledWith( + expect.objectContaining({ messageId: "out-5", text: "dm edit" }), + ); + await expect( + adapter.sendInbound({ + conversation: { id: "other-room", kind: "group" }, + senderId: "primary", + text: "remap", + }), + ).rejects.toThrow("another logical conversation"); + await expect( + adapter.sendInbound({ + conversation: { id: "room", kind: "group" }, + senderId: "missing", + text: "impersonate", + }), + ).rejects.toThrow("named leased participant"); + await expect( + adapter.sendInbound({ + conversation: { id: "room", kind: "group" }, + senderId: "second", + replyToId: "out-6", + text: "wrong observer", + }), + ).rejects.toThrow("observed by this participant"); + expect(adapter.buildAgentDelivery({ target: "group:forum", threadId: "42" })).toEqual({ + channel: "telegram", + to: "-100456", + replyChannel: "telegram", + replyTo: "-100456", + threadId: "42", + }); + } finally { + await adapter.cleanup?.(); + await adapter.cleanupAfterGatewayStop?.(); + } + expect(mocks.userbotClose).toHaveBeenCalledTimes(2); + expect(mocks.leaseRelease).toHaveBeenCalledOnce(); + }); + + it("uses local Telegram.app participants without acquiring a shared credential", async () => { + const descriptor = { + file: "/private/descriptor.json", + mode: "private-production-local-apps", + forumGroupId: "-100456", + forumTopicId: 42, + topicTitle: "Private identity proof", + participants: [ + { alias: "primary", host: "mainframe", userId: "100" }, + { alias: "second", host: "macbook", userId: "101" }, + ], + }; + mocks.readTelegramPrivateProductionDescriptor.mockReturnValueOnce(descriptor); + mocks.resolveTelegramPrivateProductionBot.mockResolvedValue({ + id: "200", + token: "private-token", + username: "qa_bot", + }); + mocks.requestTelegramPrivateAppTurn.mockResolvedValue({ replyText: "synthetic-marker" }); + const addInboundMessage = vi.fn().mockResolvedValue({ id: "in-hitl" }); + const addOutboundMessage = vi.fn().mockResolvedValue({ id: "out-hitl" }); + const adapter = await createTelegramQaTransportAdapter({ + adapterOptions: { credentialFile: "/private/descriptor.json" }, + messages: { addInboundMessage, addOutboundMessage }, + } as never); + try { + expect(mocks.acquireQaCredentialLease).not.toHaveBeenCalled(); + expect(mocks.startApiProxy).not.toHaveBeenCalled(); + expect(adapter.createGatewayConfig({ baseUrl: "http://127.0.0.1:1234" })).toMatchObject({ + channels: { + telegram: { + accounts: { + sut: { + allowFrom: ["100", "101"], + groups: { "-100456": { allowFrom: ["100", "101"] } }, + }, + }, + }, + }, + }); + expect( + adapter.createGatewayConfig({ baseUrl: "http://127.0.0.1:1234" }).channels?.telegram + ?.accounts?.sut, + ).not.toHaveProperty("apiRoot"); + + await adapter.sendInbound({ + conversation: { id: "forum", kind: "group" }, + senderId: "second", + threadId: "42", + text: "@openclaw Reply exactly: synthetic-marker", + }); + expect(mocks.requestTelegramPrivateAppTurn).toHaveBeenCalledWith({ + descriptor, + destination: "forum-topic", + participant: descriptor.participants[1], + text: "@qa_bot Reply exactly: synthetic-marker", + }); + expect(addInboundMessage).toHaveBeenCalledWith(expect.objectContaining({ senderId: "101" })); + expect(addOutboundMessage).toHaveBeenCalledWith( + expect.objectContaining({ + replyToId: "in-hitl", + text: "synthetic-marker", + to: "thread:/v1/group/forum/42", + }), + ); + expect(mocks.userbotSend).not.toHaveBeenCalled(); + } finally { + await adapter.cleanup?.(); + await adapter.cleanupAfterGatewayStop?.(); + } + expect(mocks.userbotStart).not.toHaveBeenCalled(); + expect(mocks.leaseRelease).not.toHaveBeenCalled(); + }); + + it.each(["participants", "forumGroupId", "forumTopicId"] as const)( + "fails the cataloged identity scenario clearly when the lease lacks %s", + async (missing) => { + mocks.acquireQaCredentialLease.mockResolvedValueOnce({ + payload: { ...identityCredential, [missing]: undefined }, + heartbeat: mocks.leaseHeartbeat, + release: mocks.leaseRelease, + }); + const adapter = await createTelegramQaTransportAdapter({ + adapterOptions: {}, + messages: {}, + } as never); + try { + const scenario = readQaScenarioById("telegram-participant-identity-inspection"); + await expect(prepareFlow(adapter, scenario.execution.config)).rejects.toThrow( + "requires distinct participants, forumGroupId, and forumTopicId", + ); + expect(mocks.userbotSend).not.toHaveBeenCalled(); + } finally { + await adapter.cleanup?.(); + await adapter.cleanupAfterGatewayStop?.(); + } + expect(mocks.leaseRelease).toHaveBeenCalledOnce(); + }, + ); + + it("rejects an unconfirmed participant receipt without recording synthetic identity", async () => { + mocks.userbotSend.mockResolvedValueOnce({ messageId: 10, senderId: 999, chatId: -100123 }); + const addInboundMessage = vi.fn(); + const adapter = await createTelegramQaTransportAdapter({ + adapterOptions: {}, + messages: { addInboundMessage }, + } as never); + try { + await expect( + adapter.sendInbound({ + conversation: { id: "room", kind: "group" }, + senderId: "primary", + text: "identity", + }), + ).rejects.toThrow("send receipt"); + expect(addInboundMessage).not.toHaveBeenCalled(); + } finally { + await adapter.cleanup?.(); + await adapter.cleanupAfterGatewayStop?.(); + } + }); + + it("retains participant recovery state and the lease when observer shutdown is unconfirmed", async () => { + const adapter = await createTelegramQaTransportAdapter({ + adapterOptions: {}, + messages: {}, + } as never); + const root = mocks.createStateRoot.mock.results.at(-1)?.value; + mocks.userbotClose.mockRejectedValueOnce(new Error("exit unconfirmed")); + try { + await expect(adapter.cleanup?.()).rejects.toThrow("retained private state"); + await expect(adapter.cleanupAfterGatewayStop?.()).rejects.toThrow( + "retained Telegram credential", + ); + expect(fs.existsSync(root)).toBe(true); + expect(mocks.leaseRelease).not.toHaveBeenCalled(); + expect(mocks.heartbeatStop).toHaveBeenCalledOnce(); + } finally { + fs.rmSync(root, { recursive: true, force: true }); + } + }); + it("releases the lease when userbot startup fails", async () => { const stateRoot = fs.mkdtempSync(path.join(os.tmpdir(), "telegram-adapter-failure-test-")); mocks.createStateRoot.mockReturnValueOnce(stateRoot); diff --git a/extensions/qa-lab/src/live-transports/telegram/adapter.runtime.ts b/extensions/qa-lab/src/live-transports/telegram/adapter.runtime.ts index 43515b300f74..8f64352629ca 100644 --- a/extensions/qa-lab/src/live-transports/telegram/adapter.runtime.ts +++ b/extensions/qa-lab/src/live-transports/telegram/adapter.runtime.ts @@ -9,6 +9,11 @@ import { acquireQaCredentialLease, startQaCredentialLeaseHeartbeat, } from "../shared/credential-lease.runtime.js"; +import { + readTelegramPrivateProductionDescriptor, + requestTelegramPrivateAppTurn, + resolveTelegramPrivateProductionBot, +} from "./private-production.runtime.js"; import { buildTelegramQaConfig, waitForTelegramChannelRunning } from "./telegram-api.runtime.js"; import { TelegramUserbotDriver, type TelegramUserbotUpdate } from "./userbot-driver.runtime.js"; import { @@ -19,6 +24,28 @@ import { type AdapterFactory = NonNullable; type FactoryContext = Parameters[0]; type AdapterDefinition = Awaited>; +type TelegramUserbotSkillRuntime = Awaited>; + +type TelegramRuntimeCredential = Pick< + TelegramTestCredential, + | "forumGroupId" + | "forumTopicId" + | "groupId" + | "participants" + | "sutBotId" + | "sutToken" + | "sutUsername" + | "testerUserId" +> & { + environment: "production" | "test"; +}; + +type TelegramRuntimeParticipant = { + alias: string; + credential?: TelegramRuntimeCredential; + mode: "hitl" | "userbot"; + testerUserId: string; +}; const TELEGRAM_QA_DIAGNOSTIC_COUNT_LIMIT = 9_999; @@ -73,38 +100,83 @@ export async function createTelegramQaTransportAdapter( context: FactoryContext, ): Promise { const options = context.adapterOptions ?? {}; - const skillRuntime = await loadTelegramUserbotSkillRuntime({ repoRoot: options.repoRoot }); - const credentialLease = await acquireQaCredentialLease({ - kind: "telegram-test-userbot", - source: options.credentialSource || "convex", - role: options.credentialRole, - resolveEnvPayload: () => { - throw new Error("Telegram live QA requires a Convex-leased Test Server userbot."); - }, - parsePayload: (payload) => skillRuntime.parseCredential(payload), - }); - try { - assertQaGatewayCredentialLeaseQuarantine(credentialLease); - } catch (error) { - await credentialLease.release(); - throw error; - } - const heartbeat = startQaCredentialLeaseHeartbeat(credentialLease); + const { buildQaTarget, parseQaTarget } = await import("openclaw/plugin-sdk/qa-channel-protocol"); + const privateDescriptor = readTelegramPrivateProductionDescriptor(options.credentialFile); + const privateProduction = privateDescriptor !== undefined; + const skillRuntime = privateProduction + ? undefined + : await loadTelegramUserbotSkillRuntime({ repoRoot: options.repoRoot }); + const leasedRuntime = privateProduction + ? undefined + : await (async () => { + const credentialLease = await acquireQaCredentialLease({ + kind: "telegram-test-userbot", + source: options.credentialSource || "convex", + role: options.credentialRole, + resolveEnvPayload: () => { + throw new Error("Telegram live QA requires a Convex-leased userbot."); + }, + parsePayload: (payload) => skillRuntime!.parseCredential(payload), + }); + try { + assertQaGatewayCredentialLeaseQuarantine(credentialLease); + } catch (error) { + await credentialLease.release(); + throw error; + } + return { + credential: credentialLease.payload, + credentialLease, + heartbeat: startQaCredentialLeaseHeartbeat(credentialLease), + }; + })(); + const privateBot = privateProduction + ? await resolveTelegramPrivateProductionBot(process.env) + : undefined; + const credential: TelegramRuntimeCredential = privateDescriptor + ? { + environment: "production", + forumGroupId: privateDescriptor.forumGroupId, + forumTopicId: privateDescriptor.forumTopicId, + groupId: privateDescriptor.forumGroupId, + participants: undefined, + sutBotId: privateBot!.id, + sutToken: privateBot!.token, + sutUsername: privateBot!.username, + testerUserId: privateDescriptor.participants[0]!.userId, + } + : leasedRuntime!.credential; const leaseHealth = { - assertHealthy: () => heartbeat.throwIfFailed(), - whenUnhealthy: heartbeat.whenFailed, + assertHealthy: () => leasedRuntime?.heartbeat.throwIfFailed(), + whenUnhealthy: leasedRuntime?.heartbeat.whenFailed ?? new Promise(() => {}), }; let leaseReleased = false; const releaseCredentialLease = async () => { - if (leaseReleased) { + if (leaseReleased || !leasedRuntime) { return; } - await releaseTelegramCredential({ heartbeat, release: () => credentialLease.release() }); + await releaseTelegramCredential({ + heartbeat: leasedRuntime.heartbeat, + release: () => leasedRuntime.credentialLease.release(), + }); leaseReleased = true; }; let stateRoot: string | undefined; - let apiProxy: Awaited> | undefined; + let apiProxy: Awaited> | undefined; let userbot: TelegramUserbotDriver | undefined; + let participants: TelegramRuntimeParticipant[] = []; + const drivers: TelegramUserbotDriver[] = []; + const participantRoots: string[] = []; + let primaryAlias: string | undefined; + let participantCleanupUncertain = false; + const closeParticipants = async () => { + const results = await Promise.allSettled(drivers.map((driver) => driver.close())); + const errors = results.flatMap((result) => + result.status === "rejected" ? [result.reason] : [], + ); + participantCleanupUncertain ||= errors.length > 0; + return errors; + }; const observerState: TelegramQaObserverState = { filteredCount: 0, matchedCount: 0, @@ -113,19 +185,37 @@ export async function createTelegramQaTransportAdapter( }; const accountId = options.sutAccountId?.trim() || "sut"; const directMessageOnly = options.transportPolicy?.directMessageOnly === true; - const agentDeliveryTarget = directMessageOnly - ? credentialLease.payload.testerUserId - : credentialLease.payload.groupId; - let nativeChatId = Number(credentialLease.payload.groupId); - let logicalConversationId = credentialLease.payload.groupId; - let logicalConversationKind: "channel" | "direct" | "group" = "channel"; - const nativeMessageIds = new Map(); - const busMessages = new Map(); + type Route = { + id: string; + kind: "channel" | "direct" | "group"; + threadId?: string; + bound?: true; + }; + const routes = new Map(); + const nativeKey = (observer: number, chatId: number, messageId: number) => + `${observer}:${chatId}:${messageId}`; + const routeKey = (observer: number, chatId: number, topic?: number) => + `${observer}:${chatId}:${topic ?? ""}`; + const resetRoutes = () => { + const initialChatId = Number(directMessageOnly ? credential.sutBotId : credential.groupId); + routes.clear(); + routes.set(routeKey(0, initialChatId), { + id: credential.groupId, + kind: directMessageOnly ? "direct" : "channel", + }); + }; + const nativeMessageIds = new Map< + string, + { observer: number; chatId: number; messageId: number } + >(); + const busMessages = new Map(); let sendsInFlight = 0; - let deferredReplies: TelegramUserbotUpdate[] = []; + let deferredReplies: Array<{ update: TelegramUserbotUpdate; observer: number }> = []; + let localMessageId = 1; - const publishUpdate = async (update: TelegramUserbotUpdate) => { - const existing = busMessages.get(update.messageId); + const publishUpdate = async (update: TelegramUserbotUpdate, observer: number) => { + const key = nativeKey(observer, update.chatId, update.messageId); + const existing = busMessages.get(key); if (update.kind === "edit" && existing) { await context.messages.editMessage({ accountId, @@ -136,67 +226,145 @@ export async function createTelegramQaTransportAdapter( existing.update = update; return; } + const route = routes.get(routeKey(observer, update.chatId, update.forumTopicId)); + if (!route) { + return; + } const outbound = await context.messages.addOutboundMessage({ accountId, - to: `${logicalConversationKind === "direct" ? "dm" : logicalConversationKind}:${logicalConversationId}`, + to: buildQaTarget({ + chatType: route.kind, + conversationId: route.id, + threadId: route.threadId, + }), senderId: String(update.senderId), senderName: update.senderUsername, text: update.text, timestamp: update.timestamp, - replyToId: update.replyToMessageId ? busMessages.get(update.replyToMessageId)?.id : undefined, + replyToId: update.replyToMessageId + ? busMessages.get(nativeKey(observer, update.chatId, update.replyToMessageId))?.id + : undefined, }); - nativeMessageIds.set(outbound.id, update.messageId); - busMessages.set(update.messageId, { id: outbound.id, update }); + nativeMessageIds.set(outbound.id, { + observer, + chatId: update.chatId, + messageId: update.messageId, + }); + busMessages.set(key, { id: outbound.id, update }); }; - const observeUpdate = async (update: TelegramUserbotUpdate) => { + const observeUpdate = async (update: TelegramUserbotUpdate, observer: number) => { observerState.updateCount += 1; observerState.relevantUpdateKinds.add(update.kind); if ( - update.chatId !== nativeChatId || - update.senderId !== Number(credentialLease.payload.sutBotId) + !routes.has(routeKey(observer, update.chatId, update.forumTopicId)) || + update.senderId !== Number(credential.sutBotId) ) { observerState.filteredCount += 1; return; } observerState.matchedCount += 1; - if (sendsInFlight > 0 && update.replyToMessageId && !busMessages.has(update.replyToMessageId)) { - deferredReplies.push(update); + if ( + sendsInFlight > 0 && + update.replyToMessageId && + !busMessages.has(nativeKey(observer, update.chatId, update.replyToMessageId)) + ) { + deferredReplies.push({ update, observer }); return; } - await publishUpdate(update); + await publishUpdate(update, observer); }; try { - stateRoot = skillRuntime.createStateRoot(); - const restored = skillRuntime.restoreCredential(credentialLease.payload, stateRoot); - apiProxy = await skillRuntime.startApiProxy(leaseHealth); - await apiProxy.drainUpdates(restored.sutToken); - userbot = await TelegramUserbotDriver.start({ - chatId: directMessageOnly ? `@${credentialLease.payload.sutUsername}` : restored.groupId, - driverEnv: restored.driverEnv, - leaseHealth, - userDriverPath: skillRuntime.userDriverPath, - onUpdate: observeUpdate, - }); - nativeChatId = userbot.chatId; + participants = privateProduction + ? privateDescriptor!.participants.map((participant) => ({ + alias: participant.alias, + mode: "hitl" as const, + testerUserId: participant.userId, + })) + : [ + { + alias: "primary", + credential, + mode: "userbot", + testerUserId: credential.testerUserId, + }, + ...(credential.participants ?? []).map((participant) => ({ + alias: participant.alias, + credential: { ...credential, ...participant, participants: undefined }, + mode: "userbot" as const, + testerUserId: participant.testerUserId, + })), + ]; + resetRoutes(); + if (!privateProduction) { + stateRoot = skillRuntime!.createStateRoot(); + const restored = skillRuntime!.restoreCredential( + // SAFETY: The leased credential passed the Telegram test-credential parser above. + credential as TelegramTestCredential, + stateRoot, + ); + apiProxy = await skillRuntime!.startApiProxy(leaseHealth); + await apiProxy.drainUpdates(credential.sutToken); + userbot = await TelegramUserbotDriver.start({ + chatId: directMessageOnly ? `@${credential.sutUsername}` : restored.groupId, + ...(credential.forumGroupId ? { observeChatIds: [credential.forumGroupId] } : {}), + expectedUserId: credential.testerUserId, + driverEnv: restored.driverEnv, + leaseHealth, + userDriverPath: skillRuntime!.userDriverPath, + onUpdate: (update) => observeUpdate(update, 0), + }); + drivers.push(userbot); + for (const [offset, participant] of participants.slice(1).entries()) { + if (!participant.credential) { + continue; + } + const root = skillRuntime!.createStateRoot(); + participantRoots.push(root); + const restoredParticipant = skillRuntime!.restoreCredential( + // SAFETY: Userbot participants are derived from the parsed Telegram test credential. + participant.credential as TelegramTestCredential, + root, + ); + drivers.push( + await TelegramUserbotDriver.start({ + chatId: directMessageOnly ? `@${credential.sutUsername}` : restored.groupId, + expectedUserId: participant.testerUserId, + driverEnv: restoredParticipant.driverEnv, + leaseHealth, + userDriverPath: skillRuntime!.userDriverPath, + onUpdate: (update) => observeUpdate(update, offset + 1), + }), + ); + } + } } catch (error) { const cleanupErrors: unknown[] = []; - try { - await userbot?.close(); - } catch (cleanupError) { - cleanupErrors.push(cleanupError); - } + cleanupErrors.push(...(await closeParticipants())); try { await apiProxy?.close(); } catch (cleanupError) { cleanupErrors.push(cleanupError); } - if (stateRoot) { - fs.rmSync(stateRoot, { recursive: true, force: true }); + if (!participantCleanupUncertain) { + for (const root of participantRoots) { + fs.rmSync(root, { recursive: true, force: true }); + } + if (stateRoot) { + fs.rmSync(stateRoot, { recursive: true, force: true }); + } } try { - await releaseCredentialLease(); + if (participantCleanupUncertain && leasedRuntime) { + try { + await leasedRuntime.credentialLease.heartbeat(); + } finally { + await leasedRuntime.heartbeat.stop(); + } + } else { + await releaseCredentialLease(); + } } catch (cleanupError) { cleanupErrors.push(cleanupError); } @@ -208,10 +376,9 @@ export async function createTelegramQaTransportAdapter( throw error; } - if (!userbot || !apiProxy || !stateRoot) { + if (!privateProduction && (!userbot || !apiProxy || !stateRoot)) { throw new Error("Telegram userbot runtime did not start."); } - const activeUserbot = userbot; const activeApiProxy = apiProxy; const activeStateRoot = stateRoot; let observerStopped = false; @@ -223,32 +390,166 @@ export async function createTelegramQaTransportAdapter( requiredPluginIds: ["telegram"], supportedActions: [], assertTransportHealthy() { - activeUserbot.assertHealthy(); - heartbeat.throwIfFailed(); + for (const driver of drivers) { + driver.assertHealthy(); + } + leasedRuntime?.heartbeat.throwIfFailed(); }, describeTransportState: () => describeTelegramQaObserverState(observerState), async sendInbound(input) { - heartbeat.throwIfFailed(); - logicalConversationId = input.conversation.id; - logicalConversationKind = input.conversation.kind; - const text = renderTelegramQaInboundText(input, credentialLease.payload.sutUsername); - const nativeReplyToId = input.replyToId ? nativeMessageIds.get(input.replyToId) : undefined; + leasedRuntime?.heartbeat.throwIfFailed(); + let observer = participants.findIndex( + (participant) => + participant.alias === input.senderId || participant.testerUserId === input.senderId, + ); + if (observer < 0) { + if (participants.length > 1) { + throw new Error( + "Telegram QA sender requires primary or a named leased participant alias.", + ); + } + primaryAlias ??= input.senderId; + if (!input.senderId || input.senderId !== primaryAlias) { + throw new Error( + "Telegram QA sender requires a named leased participant; labels cannot impersonate another user.", + ); + } + observer = 0; + } + const participant = participants[observer]; + if (!participant) { + throw new Error("Telegram QA participant is unavailable."); + } + if (directMessageOnly && input.conversation.kind !== "direct") { + throw new Error("Telegram QA direct-message-only policy rejects group sends."); + } + const forumTopicId = input.threadId === undefined ? undefined : Number(input.threadId); + if ( + forumTopicId !== undefined && + (!Number.isSafeInteger(forumTopicId) || + forumTopicId <= 0 || + input.conversation.kind === "direct") + ) { + throw new Error( + "Telegram QA forum sends require a positive numeric topic and a group conversation.", + ); + } + const chatId = + input.conversation.kind === "direct" + ? Number(credential.sutBotId) + : Number( + forumTopicId ? (credential.forumGroupId ?? credential.groupId) : credential.groupId, + ); + // Shared rooms use one observer. TDLib IDs and local-app attestations share that route. + const routeObserver = input.conversation.kind === "direct" ? observer : 0; + const routeId = routeKey(routeObserver, chatId, forumTopicId); + const previous = routes.get(routeId); + if ( + previous?.bound && + (previous.id !== input.conversation.id || previous.kind !== input.conversation.kind) + ) { + throw new Error( + "Telegram QA chat/topic already belongs to another logical conversation; reset transport before reusing it.", + ); + } + routes.set(routeId, { ...input.conversation, threadId: input.threadId, bound: true }); + const reply = input.replyToId ? nativeMessageIds.get(input.replyToId) : undefined; + if (input.replyToId && (!reply || reply.observer !== observer || reply.chatId !== chatId)) { + throw new Error( + "Telegram QA reply requires a message observed by this participant in this chat.", + ); + } + const text = renderTelegramQaInboundText(input, credential.sutUsername); sendsInFlight += 1; try { - const sent = await activeUserbot.send({ text, replyToMessageId: nativeReplyToId }); + let sent: TelegramUserbotUpdate; + let messageObserver = observer; + let appProofReply: string | undefined; + if (participant.mode === "hitl") { + const appParticipant = privateDescriptor?.participants[observer]; + if (!privateDescriptor || !appParticipant) { + throw new Error("Telegram local-app participant is unavailable."); + } + if ( + input.conversation.kind === "group" && + (forumTopicId !== privateDescriptor.forumTopicId || + chatId !== Number(privateDescriptor.forumGroupId)) + ) { + throw new Error("Telegram local-app forum send targets the wrong private topic."); + } + messageObserver = routeObserver; + const proof = await requestTelegramPrivateAppTurn({ + descriptor: privateDescriptor, + destination: input.conversation.kind === "direct" ? "bot-dm" : "forum-topic", + participant: appParticipant, + text, + }); + appProofReply = proof.replyText; + sent = { + kind: "message", + chatId, + ...(forumTopicId === undefined ? {} : { forumTopicId }), + messageId: localMessageId++, + senderId: Number(participant.testerUserId), + timestamp: Date.now(), + text, + entities: [], + }; + } else { + const driver = drivers[observer]; + if (!driver) { + throw new Error("Telegram QA participant driver is unavailable."); + } + sent = await driver.send({ + text, + chatId: String(chatId), + ...(forumTopicId === undefined ? {} : { forumTopicId }), + replyToMessageId: reply?.messageId, + }); + } + if ( + String(sent.senderId) !== participant.testerUserId || + sent.chatId !== chatId || + (forumTopicId !== undefined && sent.forumTopicId !== forumTopicId) + ) { + throw new Error( + "Telegram send receipt does not match the configured participant and requested chat/topic.", + ); + } const message = await context.messages.addInboundMessage({ ...input, accountId, - senderId: credentialLease.payload.testerUserId, + senderId: participant.testerUserId, }); - nativeMessageIds.set(message.id, sent.messageId); - busMessages.set(sent.messageId, { id: message.id }); - const readyReplies = deferredReplies.filter( - (update) => update.replyToMessageId && busMessages.has(update.replyToMessageId), - ); - deferredReplies = deferredReplies.filter((update) => !readyReplies.includes(update)); - for (const update of readyReplies) { - await publishUpdate(update); + nativeMessageIds.set(message.id, { + observer: messageObserver, + chatId, + messageId: sent.messageId, + }); + busMessages.set(nativeKey(messageObserver, chatId, sent.messageId), { id: message.id }); + if (appProofReply !== undefined) { + await observeUpdate( + { + kind: "message", + chatId, + ...(forumTopicId === undefined ? {} : { forumTopicId }), + messageId: localMessageId++, + replyToMessageId: sent.messageId, + senderId: Number(credential.sutBotId), + senderUsername: credential.sutUsername, + timestamp: Date.now(), + text: appProofReply, + entities: [], + }, + messageObserver, + ); + } + // A shared-room reply can quote an ID in another user's private TDLib sequence. + // Preserve the reply without claiming a cross-account quote relationship. + const readyReplies = deferredReplies; + deferredReplies = []; + for (const entry of readyReplies) { + await publishUpdate(entry.update, entry.observer); } return message; } finally { @@ -256,8 +557,8 @@ export async function createTelegramQaTransportAdapter( } }, resetTransport: () => { - logicalConversationId = credentialLease.payload.groupId; - logicalConversationKind = "channel"; + resetRoutes(); + primaryAlias = undefined; nativeMessageIds.clear(); busMessages.clear(); deferredReplies = []; @@ -266,11 +567,25 @@ export async function createTelegramQaTransportAdapter( observerState.matchedCount = 0; observerState.relevantUpdateKinds.clear(); }, - async prepareFlow() { + async prepareFlow({ config }) { + if ( + config.requireParticipantIdentityFixture === true && + (participants.length < 2 || !credential.forumGroupId || !credential.forumTopicId) + ) { + throw new Error( + "Telegram participant identity proof requires distinct participants, forumGroupId, and forumTopicId.", + ); + } return { + telegramIdentityFixture: { + participantAliases: participants.map((participant) => participant.alias), + forumTopicId: credential.forumTopicId, + }, readTelegramMessages: () => { - activeUserbot.assertHealthy(); - heartbeat.throwIfFailed(); + for (const driver of drivers) { + driver.assertHealthy(); + } + leasedRuntime?.heartbeat.throwIfFailed(); // Share the existing message lifetime; readers cannot mutate a later snapshot. return [...busMessages.values()].flatMap(({ update }) => update ? [structuredClone(update)] : [], @@ -279,12 +594,18 @@ export async function createTelegramQaTransportAdapter( }; }, createGatewayConfig: () => + // SAFETY: The builder accepts an empty base and supplies every QA-owned config section. buildTelegramQaConfig({} as OpenClawConfig, { - apiRoot: activeApiProxy.apiRoot, + apiRoot: activeApiProxy?.apiRoot, directMessageOnly, - groupId: credentialLease.payload.groupId, - sutToken: credentialLease.payload.sutToken, - testerUserId: credentialLease.payload.testerUserId, + enableDirectMessages: true, + additionalTesterUserIds: participants + .slice(1) + .map((participant) => participant.testerUserId), + forumGroupId: credential.forumGroupId, + groupId: credential.groupId, + sutToken: credential.sutToken, + testerUserId: credential.testerUserId, sutAccountId: accountId, }), waitReady: async ({ gateway, timeoutMs, pollIntervalMs }) => @@ -292,30 +613,51 @@ export async function createTelegramQaTransportAdapter( timeoutMs, pollMs: pollIntervalMs, }), - buildAgentDelivery: () => ({ - channel: "telegram", - to: agentDeliveryTarget, - replyChannel: "telegram", - replyTo: agentDeliveryTarget, - }), + buildAgentDelivery: ({ target, threadId }) => { + const parsed = parseQaTarget(target); + const topic = threadId ?? parsed?.threadId; + const to = + parsed?.chatType === "direct" || directMessageOnly + ? credential.testerUserId + : topic + ? (credential.forumGroupId ?? credential.groupId) + : credential.groupId; + return { + channel: "telegram", + to, + replyChannel: "telegram", + replyTo: to, + ...(topic ? { threadId: topic } : {}), + }; + }, async handleAction() { throw new Error("Telegram live QA adapter does not implement transport actions"); }, - createReportNotes: () => ["Runs through the Telegram Test Server userbot adapter."], + createReportNotes: () => [ + privateProduction + ? "Runs through two operator-local Telegram.app participants; private UI attestations back the native send and reply observations." + : "Runs through the Telegram Test Server userbot adapter.", + ], async cleanup() { if (observerStopped) { return; } observerStopped = true; - try { - await activeUserbot.close(); - } finally { - fs.rmSync(activeStateRoot, { recursive: true, force: true }); + const errors: unknown[] = []; + errors.push(...(await closeParticipants())); + if (errors.length) { + throw new AggregateError( + errors, + "Telegram participant cleanup is unconfirmed; retained private state and lease.", + ); + } + for (const root of [...(activeStateRoot ? [activeStateRoot] : []), ...participantRoots]) { + fs.rmSync(root, { recursive: true, force: true }); } }, async cleanupAfterGatewayStop() { const cleanupErrors: unknown[] = []; - if (!apiProxyClosed) { + if (!apiProxyClosed && activeApiProxy) { try { await activeApiProxy.close(); apiProxyClosed = true; @@ -323,19 +665,22 @@ export async function createTelegramQaTransportAdapter( cleanupErrors.push(error); } } - if (await shouldRetainQaGatewayCredentialLease()) { + if ( + leasedRuntime && + (participantCleanupUncertain || (await shouldRetainQaGatewayCredentialLease())) + ) { try { - await credentialLease.heartbeat(); + await leasedRuntime.credentialLease.heartbeat(); } catch (error) { cleanupErrors.push(error); } try { - await heartbeat.stop(); + await leasedRuntime.heartbeat.stop(); } catch (error) { cleanupErrors.push(error); } throw new Error( - "retained Telegram credential lease for two hours because isolated SUT quiescence was not proven", + "retained Telegram credential lease for two hours because participant or isolated SUT quiescence was not proven", cleanupErrors.length > 0 ? { cause: new AggregateError(cleanupErrors) } : undefined, ); } diff --git a/extensions/qa-lab/src/live-transports/telegram/cli.runtime.ts b/extensions/qa-lab/src/live-transports/telegram/cli.runtime.ts index d1460ca05e8d..f5ec2d7fc75c 100644 --- a/extensions/qa-lab/src/live-transports/telegram/cli.runtime.ts +++ b/extensions/qa-lab/src/live-transports/telegram/cli.runtime.ts @@ -179,6 +179,7 @@ export async function runQaTelegramSuite(opts: TelegramQaSuiteOptions) { ], adapterOptions: { repoRoot: runOptions.repoRoot, + ...(runOptions.credentialFile ? { credentialFile: runOptions.credentialFile } : {}), ...(runOptions.credentialRole ? { credentialRole: runOptions.credentialRole } : {}), ...(runOptions.credentialSource ? { credentialSource: runOptions.credentialSource } : {}), ...(runOptions.sutAccountId ? { sutAccountId: runOptions.sutAccountId } : {}), diff --git a/extensions/qa-lab/src/live-transports/telegram/cli.ts b/extensions/qa-lab/src/live-transports/telegram/cli.ts index d61b3c09db96..3f9333f46e99 100644 --- a/extensions/qa-lab/src/live-transports/telegram/cli.ts +++ b/extensions/qa-lab/src/live-transports/telegram/cli.ts @@ -27,7 +27,8 @@ export const telegramQaCliRegistration: LiveTransportQaCliRegistration = roleDescription: "Credential role for convex auth: maintainer or ci (default: ci in CI, maintainer otherwise)", }, - description: "Run Telegram Test Server QA with a Convex-leased real-user driver", + credentialFileHelp: "Private qualification-mode descriptor (contains no credentials)", + description: "Run Telegram Test Server QA with a Convex-leased user or private production apps", listScenariosHelp: "Print available Telegram scenario ids and exit", outputDirHelp: "Telegram QA artifact directory", profileHelp: "Taxonomy profile for Telegram scenario selection (default: release)", diff --git a/extensions/qa-lab/src/live-transports/telegram/participant-identity-flow.test.ts b/extensions/qa-lab/src/live-transports/telegram/participant-identity-flow.test.ts new file mode 100644 index 000000000000..f6ae0c2defd6 --- /dev/null +++ b/extensions/qa-lab/src/live-transports/telegram/participant-identity-flow.test.ts @@ -0,0 +1,321 @@ +import { describe, expect, it, vi } from "vitest"; +import { createQaBusState } from "../../bus-state.js"; +import { readQaScenarioById, readQaScenarioPack } from "../../scenario-catalog.js"; +import { runLoadedScenarioFlow } from "../../scenario-flow-runner.test-support.js"; +import { selectQaFlowSuiteScenarios } from "../../suite-planning.js"; +import { resolveTelegramQaScenarioIds } from "./scenario-selection.js"; + +const scenarioId = "telegram-participant-identity-inspection"; +const primaryUserId = "710000001"; +const additionalUserId = "710000002"; +const botId = 710000003; +const forumGroupId = -100710000004; +const forumTopicId = 42; +const hmac = (digit: string) => `hmac-sha256:v1:${"a".repeat(32)}:${digit.repeat(64)}`; + +type Fault = + | "wrong-transport" + | "wrong-provider" + | "missing-fixture" + | "missing-participant" + | "duplicate-alias" + | "missing-topic" + | "wrong-topic" + | "extra-run" + | "unknown-person" + | "room-principal" + | "wrong-run" + | "raw-principal" + | "missing-assurance" + | "raw-room" + | "prompt-leak" + | "oversized-context" + | "verified-generic" + | "wrong-cli-execution" + | "missing-human" + | "same-person" + | "changed-primary" + | "reused-context" + | "restart-drift" + | "restart-leak"; + +function runIdentityFlow(fault?: Fault) { + const state = createQaBusState(); + const admittedRuns = ["previous-run"]; + const turns: Array> = []; + const nativeReplies: Array<{ + text: string; + chatId: number; + senderId: number; + forumTopicId?: number; + }> = []; + let restarted = false; + const inspect = (selector: { runId?: string; executionId?: string }) => { + const runId = selector.runId ?? selector.executionId?.replace("execution-", ""); + const index = admittedRuns.indexOf(runId ?? "") - 1; + const turn = turns[index]; + if (!turn || !runId) { + throw new Error("identity proof must inspect only a newly admitted transport turn"); + } + const additional = turn.senderId === additionalUserId; + const principalRef = + fault === "raw-principal" + ? turn.senderId + : hmac( + fault === "same-person" + ? "b" + : fault === "changed-primary" && index === 1 + ? "e" + : additional + ? "c" + : "b", + ); + return { + run: { runId, executionId: `execution-${runId}` }, + identity: { + state: "present", + context: { + contextId: + fault === "reused-context" + ? "context-first" + : restarted && fault === "restart-drift" + ? "context-replacement" + : `context-${runId}`, + executionId: `execution-${runId}`, + runId: fault === "wrong-run" ? "unrelated-run" : runId, + invoker: { + state: fault === "unknown-person" ? "unknown" : "present", + principal: { + kind: fault === "room-principal" ? "service" : "person", + principalRef, + domainRef: hmac("a"), + }, + }, + // Channel admission supplies no rawSourceRef; do not invent one in proof support. + ingress: { kind: "channel", state: "present" }, + runtimeInstance: { runtimeRef: hmac("e") }, + assurance: + fault === "missing-assurance" + ? [] + : [ + { + kind: "channel-admission", + strength: "boundary-verified", + evidenceRef: hmac("d"), + }, + ], + applicableGrants: [], + ...(fault === "oversized-context" ? { padding: "x".repeat(16384) } : {}), + }, + }, + decisionDisplays: [ + { + action: { family: "decision", operation: "record" }, + provenance: { state: fault === "verified-generic" ? "verified" : "unverified" }, + }, + ], + ...(fault === "raw-room" || (restarted && fault === "restart-leak") + ? { leakedReference: String(botId) } + : {}), + ...(fault === "prompt-leak" ? { leakedText: turn.text } : {}), + }; + }; + const call = vi.fn( + async ( + method: string, + selector: { runId?: string; executionId?: string; decisionLimit: number }, + ) => { + expect(method).toBe("audit.run.inspect"); + // audit.run.inspect has a closed request schema, distinct from audit CLI --limit. + expect(Object.keys(selector).toSorted()).toEqual([ + "decisionLimit", + selector.runId === undefined ? "executionId" : "runId", + ]); + expect(selector.decisionLimit).toBe(100); + return inspect(selector); + }, + ); + const restart = vi.fn(async (mutate: () => Promise) => { + await mutate(); + restarted = true; + }); + const runQaCli = vi.fn(async (_env: unknown, args: string[], options?: { json?: boolean }) => { + expect(args[0]).toBe("audit"); + if (args.includes("--kind")) { + return { events: admittedRuns.map((runId) => ({ runId })) }; + } + expect(args.slice(0, 2)).toEqual(["audit", "--execution"]); + const executionId = args[2]; + if (!executionId) { + throw new Error("identity CLI proof requires an exact execution selector"); + } + const inspection = inspect({ executionId }); + if (!options?.json) { + return fault === "missing-human" + ? "Identity unavailable" + : `Invoker [present] ${inspection.identity.context.invoker.principal.principalRef}\nDecisions`; + } + return fault === "wrong-cli-execution" + ? { ...inspection, run: { ...inspection.run, executionId: "foreign-execution" } } + : inspection; + }); + return { + call, + restart, + runQaCli, + turns, + result: runLoadedScenarioFlow(scenarioId, { + state, + api: { + env: { + providerMode: fault === "wrong-provider" ? "live-frontier" : "mock-openai", + gateway: { call, restartAfterStateMutation: restart }, + }, + // Only the adapter's prepared, noncredential surface is supplied here. + telegramIdentityFixture: + fault === "missing-fixture" + ? undefined + : { + participantAliases: + fault === "missing-participant" + ? ["primary"] + : ["primary", fault === "duplicate-alias" ? "primary" : "guest"], + forumTopicId: fault === "missing-topic" ? undefined : forumTopicId, + }, + readTelegramMessages: () => nativeReplies, + transport: { + id: fault === "wrong-transport" ? "qa-channel" : "telegram", + reset: async () => state.reset(), + sendInbound: async (input: Parameters[0]) => { + expect(["primary", "guest"]).toContain(input.senderId); + const inbound = state.addInboundMessage({ + ...input, + senderId: input.senderId === "primary" ? primaryUserId : additionalUserId, + }); + turns.push(inbound); + return inbound; + }, + waitForOutbound: async (input: { + conversation: { id: string; kind: string }; + threadId?: string; + textIncludes: string; + }) => { + const inbound = turns.at(-1); + if (!inbound) { + throw new Error("identity proof must send a transport turn before inspecting it"); + } + expect(input.conversation).toEqual(inbound.conversation); + expect(input.threadId).toBe(inbound.threadId); + const forum = inbound.conversation.kind === "group"; + expect(inbound.threadId).toBe(forum ? String(forumTopicId) : undefined); + expect(inbound.text.startsWith("@openclaw ")).toBe(forum); + expect(inbound.text).toContain(`Reply exactly: ${input.textIncludes}`); + admittedRuns.push(`run-${turns.length}`); + if (fault === "extra-run") { + admittedRuns.push("unrelated-new-run"); + } + nativeReplies.push({ + text: input.textIncludes, + chatId: forum ? forumGroupId : botId, + senderId: botId, + ...(forum + ? { forumTopicId: fault === "wrong-topic" ? forumTopicId + 1 : forumTopicId } + : {}), + }); + return state.addOutboundMessage({ + accountId: "sut", + to: `${forum ? "group" : "dm"}:${inbound.conversation.id}`, + threadId: inbound.threadId, + text: input.textIncludes, + }); + }, + }, + runQaCli, + }, + }), + }; +} + +describe("Telegram participant identity executable flow", () => { + it("catalogs live participant identity qualification with the fixture gate", () => { + const scenarios = readQaScenarioPack().scenarios.filter( + (scenario) => + scenario.execution.kind === "flow" && + scenario.execution.channels?.includes("telegram") && + scenario.execution.config?.requiredChannelDriver === "live" && + scenario.execution.config.requireParticipantIdentityFixture === true, + ); + expect(scenarios.map((scenario) => scenario.id)).toContain(scenarioId); + expect( + resolveTelegramQaScenarioIds({ providerMode: "mock-openai", scenarioIds: [scenarioId] }), + ).toEqual([scenarioId]); + const scenario = readQaScenarioById(scenarioId); + expect(scenario.execution).toMatchObject({ suiteIsolation: "isolated", retryCount: 0 }); + expect(scenario.gatewayConfigPatch).toMatchObject({ + logging: { audit: { enabled: true, executionIdentity: true } }, + }); + expect( + selectQaFlowSuiteScenarios({ + scenarios: [scenario], + channel: "telegram", + channelDriver: "crabline", + providerMode: "mock-openai", + primaryModel: "mock-openai/fixture", + }), + ).toEqual([]); + }); + + it("executes both participant aliases through DM/forum, exact RPC/CLI inspection, and one restart", async () => { + const proof = runIdentityFlow(); + await expect(proof.result).resolves.toMatchObject({ status: "pass" }); + expect(proof.turns.map((turn) => [turn.senderId, turn.conversation.kind])).toEqual([ + [primaryUserId, "direct"], + [primaryUserId, "group"], + [additionalUserId, "group"], + ]); + expect(proof.restart).toHaveBeenCalledOnce(); + expect(proof.call.mock.calls.map(([, selector]) => selector)).toEqual([ + { runId: "run-1", decisionLimit: 100 }, + { executionId: "execution-run-1", decisionLimit: 100 }, + { runId: "run-2", decisionLimit: 100 }, + { executionId: "execution-run-2", decisionLimit: 100 }, + { runId: "run-3", decisionLimit: 100 }, + { executionId: "execution-run-3", decisionLimit: 100 }, + { executionId: "execution-run-1", decisionLimit: 100 }, + { executionId: "execution-run-2", decisionLimit: 100 }, + { executionId: "execution-run-3", decisionLimit: 100 }, + ]); + expect( + proof.runQaCli.mock.calls.filter(([, args]) => args.includes("--execution")), + ).toHaveLength(12); + }); + + it.each([ + ["wrong-transport", "requires the live Telegram adapter"], + ["wrong-provider", "requires the live Telegram adapter"], + ["missing-fixture", "requires distinct participant aliases"], + ["missing-participant", "requires distinct participant aliases"], + ["duplicate-alias", "requires distinct participant aliases"], + ["missing-topic", "requires distinct participant aliases"], + ["wrong-topic", "requested DM or leased forum topic"], + ["extra-run", "exactly one newly admitted run"], + ["unknown-person", "retain the admitted person"], + ["room-principal", "retain the admitted person"], + ["wrong-run", "retain the admitted person"], + ["raw-principal", "bounded redacted identity"], + ["missing-assurance", "bounded redacted identity"], + ["raw-room", "bounded redacted identity"], + ["prompt-leak", "bounded redacted identity"], + ["oversized-context", "bounded redacted identity"], + ["verified-generic", "bounded redacted identity"], + ["wrong-cli-execution", "must agree with run discovery"], + ["missing-human", "must agree with run discovery"], + ["same-person", "distinguish the additional participant"], + ["changed-primary", "preserve the same primary person"], + ["reused-context", "three distinct execution contexts"], + ["restart-drift", "changed or exposed private references after restart"], + ["restart-leak", "changed or exposed private references after restart"], + ] satisfies Array<[Fault, string]>)("rejects %s evidence", async (fault, message) => { + await expect(runIdentityFlow(fault).result).rejects.toThrow(message); + }); +}); diff --git a/extensions/qa-lab/src/live-transports/telegram/private-production.runtime.test.ts b/extensions/qa-lab/src/live-transports/telegram/private-production.runtime.test.ts new file mode 100644 index 000000000000..5da197082b51 --- /dev/null +++ b/extensions/qa-lab/src/live-transports/telegram/private-production.runtime.test.ts @@ -0,0 +1,142 @@ +import fs from "node:fs"; +import os from "node:os"; +import path from "node:path"; +import { afterEach, describe, expect, it, vi } from "vitest"; +import { + readTelegramPrivateProductionDescriptor, + requestTelegramPrivateAppTurn, + resolveTelegramPrivateProductionBot, +} from "./private-production.runtime.js"; + +const fetchWithSsrFGuardMock = vi.hoisted(() => vi.fn()); +vi.mock("openclaw/plugin-sdk/ssrf-runtime", () => ({ + fetchWithSsrFGuard: fetchWithSsrFGuardMock, +})); + +const roots: string[] = []; + +function writeDescriptor() { + const root = fs.mkdtempSync(path.join(os.tmpdir(), "telegram-private-apps-test-")); + roots.push(root); + const file = path.join(root, "descriptor.json"); + fs.writeFileSync( + file, + JSON.stringify({ + mode: "private-production-local-apps", + forumGroupId: "-100456", + forumTopicId: 42, + topicTitle: "Private proof", + participants: [ + { alias: "primary", host: "mainframe", userId: "100" }, + { alias: "second", host: "macbook", userId: "101" }, + ], + }), + { mode: 0o600 }, + ); + return file; +} + +afterEach(() => { + fetchWithSsrFGuardMock.mockReset(); + for (const root of roots.splice(0)) { + fs.rmSync(root, { force: true, recursive: true }); + } +}); + +describe("Telegram private production local-app proof", () => { + it("parses two distinct operator-local participants without credential material", () => { + const file = writeDescriptor(); + + expect(readTelegramPrivateProductionDescriptor(file)).toEqual({ + file, + mode: "private-production-local-apps", + forumGroupId: "-100456", + forumTopicId: 42, + topicTitle: "Private proof", + participants: [ + { alias: "primary", host: "mainframe", userId: "100" }, + { alias: "second", host: "macbook", userId: "101" }, + ], + }); + }); + + it("resolves the bot through the guarded network boundary and releases it", async () => { + const release = vi.fn(); + fetchWithSsrFGuardMock.mockResolvedValue({ + response: new Response( + JSON.stringify({ + ok: true, + result: { id: 700000001, username: "qa_bot" }, + }), + { status: 200 }, + ), + release, + }); + + await expect( + resolveTelegramPrivateProductionBot({ TELEGRAM_BOT_TOKEN: "test-token" }), + ).resolves.toEqual({ + id: "700000001", + token: "test-token", + username: "qa_bot", + }); + expect(fetchWithSsrFGuardMock).toHaveBeenCalledWith({ + url: "https://api.telegram.org/bottest-token/getMe", + init: { method: "POST" }, + timeoutMs: 30_000, + maxRedirects: 0, + auditContext: "qa-lab-telegram-private-production-bot-api", + }); + expect(release).toHaveBeenCalledOnce(); + }); + + it("accepts a native UI acknowledgement without placing raw participant IDs in the handoff", async () => { + const file = writeDescriptor(); + const descriptor = readTelegramPrivateProductionDescriptor(file)!; + const pending = requestTelegramPrivateAppTurn({ + descriptor, + destination: "forum-topic", + participant: descriptor.participants[1]!, + text: "@qa_bot Reply exactly: marker", + }); + const proofRoot = `${file}.app-proof`; + let requestPath: string | undefined; + for (let attempts = 0; attempts < 100 && !requestPath; attempts += 1) { + requestPath = fs.existsSync(proofRoot) + ? fs + .readdirSync(proofRoot) + .map((entry) => path.join(proofRoot, entry)) + .find((entry) => entry.endsWith(".request.json")) + : undefined; + if (!requestPath) { + await new Promise((resolve) => { + setTimeout(resolve, 10); + }); + } + } + expect(requestPath).toBeDefined(); + const requestText = fs.readFileSync(requestPath!, "utf8"); + expect(requestText).not.toContain('"100"'); + expect(requestText).not.toContain('"101"'); + const request = JSON.parse(requestText) as { + destination: string; + text: string; + token: string; + }; + fs.writeFileSync( + path.join(proofRoot, `${request.token}.ack.json`), + JSON.stringify({ + schemaVersion: 1, + token: request.token, + sentText: request.text, + replyText: "marker", + replyObservedIn: request.destination, + replyToRequestedMessage: true, + }), + { mode: 0o600 }, + ); + + await expect(pending).resolves.toEqual({ replyText: "marker" }); + expect(fs.readdirSync(proofRoot)).toEqual([]); + }); +}); diff --git a/extensions/qa-lab/src/live-transports/telegram/private-production.runtime.ts b/extensions/qa-lab/src/live-transports/telegram/private-production.runtime.ts new file mode 100644 index 000000000000..7ce923d8edf8 --- /dev/null +++ b/extensions/qa-lab/src/live-transports/telegram/private-production.runtime.ts @@ -0,0 +1,226 @@ +import { randomUUID } from "node:crypto"; +import fs from "node:fs"; +import fsp from "node:fs/promises"; +import path from "node:path"; +import { fetchWithSsrFGuard } from "openclaw/plugin-sdk/ssrf-runtime"; +import { isRecord } from "openclaw/plugin-sdk/string-coerce-runtime"; + +export type TelegramPrivateAppParticipant = { + alias: string; + host: string; + userId: string; +}; + +export type TelegramPrivateProductionDescriptor = { + file: string; + forumGroupId: string; + forumTopicId: number; + mode: "private-production-local-apps"; + participants: TelegramPrivateAppParticipant[]; + topicTitle: string; +}; + +type TelegramBotIdentity = { + id: string; + token: string; + username: string; +}; + +function requireString(value: Record, key: string) { + const item = value[key]; + if (typeof item !== "string" || !item.trim()) { + throw new Error(`Telegram private production descriptor has invalid ${key}.`); + } + return item.trim(); +} + +function parseParticipant(value: unknown): TelegramPrivateAppParticipant { + if (!isRecord(value)) { + throw new Error("Telegram private production participant is not an object."); + } + const participant = { + alias: requireString(value, "alias"), + host: requireString(value, "host"), + userId: requireString(value, "userId"), + }; + if (!/^\d+$/u.test(participant.userId)) { + throw new Error("Telegram private production participant has invalid userId."); + } + return participant; +} + +export function readTelegramPrivateProductionDescriptor( + file: string | undefined, +): TelegramPrivateProductionDescriptor | undefined { + if (!file?.trim()) { + return undefined; + } + const stat = fs.lstatSync(file); + if (!stat.isFile() || stat.isSymbolicLink()) { + throw new Error("Telegram private production descriptor must be a regular file."); + } + const value: unknown = JSON.parse(fs.readFileSync(file, "utf8")); + if (!isRecord(value) || value.mode !== "private-production-local-apps") { + throw new Error("Telegram credential descriptor must select private-production-local-apps."); + } + const participants = Array.isArray(value.participants) + ? value.participants.map(parseParticipant) + : []; + if ( + participants.length < 2 || + participants[0]?.alias !== "primary" || + new Set(participants.map((participant) => participant.alias)).size !== participants.length || + new Set(participants.map((participant) => participant.userId)).size !== participants.length + ) { + throw new Error( + "Telegram private production descriptor requires primary plus a distinct local-app participant.", + ); + } + const forumGroupId = requireString(value, "forumGroupId"); + if (!/^-\d+$/u.test(forumGroupId)) { + throw new Error("Telegram private production descriptor has invalid forumGroupId."); + } + const forumTopicId = value.forumTopicId; + if ( + typeof forumTopicId !== "number" || + !Number.isSafeInteger(forumTopicId) || + forumTopicId <= 0 + ) { + throw new Error("Telegram private production descriptor has invalid forumTopicId."); + } + return { + file, + forumGroupId, + forumTopicId, + mode: value.mode, + participants, + topicTitle: requireString(value, "topicTitle"), + }; +} + +async function botApi(bot: TelegramBotIdentity, method: string) { + try { + const guarded = await fetchWithSsrFGuard({ + url: `https://api.telegram.org/bot${bot.token}/${method}`, + init: { method: "POST" }, + timeoutMs: 30_000, + maxRedirects: 0, + auditContext: "qa-lab-telegram-private-production-bot-api", + }); + try { + const value: unknown = await guarded.response.json(); + if (!guarded.response.ok || !isRecord(value) || value.ok !== true) { + throw new Error("request rejected"); + } + return value.result; + } finally { + await guarded.release(); + } + } catch { + throw new Error(`Telegram private production Bot API ${method} failed.`); + } +} + +export async function resolveTelegramPrivateProductionBot(env: NodeJS.ProcessEnv) { + const token = env.TELEGRAM_BOT_TOKEN?.trim(); + if (!token) { + throw new Error("Telegram private production proof requires TELEGRAM_BOT_TOKEN."); + } + const placeholder = { id: "", token, username: "" }; + const result = await botApi(placeholder, "getMe"); + if (!isRecord(result) || typeof result.id !== "number" || typeof result.username !== "string") { + throw new Error("Telegram private production bot identity is invalid."); + } + return { id: String(result.id), token, username: result.username }; +} + +function appProofRoot(descriptorFile: string) { + return `${descriptorFile}.app-proof`; +} + +async function readAcknowledgement(params: { + acknowledgementPath: string; + destination: "bot-dm" | "forum-topic"; + sentText: string; + token: string; +}) { + const deadline = Date.now() + 10 * 60_000; + while (Date.now() < deadline) { + try { + const stat = await fsp.lstat(params.acknowledgementPath); + if (!stat.isFile() || stat.isSymbolicLink()) { + throw new Error("Telegram.app proof acknowledgement must be a regular file."); + } + const value: unknown = JSON.parse(await fsp.readFile(params.acknowledgementPath, "utf8")); + if ( + !isRecord(value) || + value.schemaVersion !== 1 || + value.token !== params.token || + value.sentText !== params.sentText || + value.replyObservedIn !== params.destination || + value.replyToRequestedMessage !== true || + typeof value.replyText !== "string" || + !value.replyText.trim() + ) { + throw new Error("Telegram.app proof acknowledgement is invalid."); + } + return value.replyText; + } catch (error) { + // SAFETY: Node filesystem rejections carry errno codes; only ENOENT is retryable here. + if ((error as NodeJS.ErrnoException).code !== "ENOENT") { + throw error; + } + } + await new Promise((resolve) => { + setTimeout(resolve, 250); + }); + } + throw new Error("Timed out waiting for Telegram.app send and native reply proof."); +} + +export async function requestTelegramPrivateAppTurn(params: { + descriptor: TelegramPrivateProductionDescriptor; + destination: "bot-dm" | "forum-topic"; + participant: TelegramPrivateAppParticipant; + text: string; +}) { + const root = appProofRoot(params.descriptor.file); + await fsp.mkdir(root, { recursive: true, mode: 0o700 }); + await fsp.chmod(root, 0o700); + const token = randomUUID(); + const requestPath = path.join(root, `${token}.request.json`); + const acknowledgementPath = path.join(root, `${token}.ack.json`); + await fsp.writeFile( + requestPath, + `${JSON.stringify( + { + schemaVersion: 1, + token, + participant: { alias: params.participant.alias, host: params.participant.host }, + destination: params.destination, + ...(params.destination === "forum-topic" + ? { topicTitle: params.descriptor.topicTitle } + : {}), + text: params.text, + }, + null, + 2, + )}\n`, + { flag: "wx", mode: 0o600 }, + ); + process.stdout.write(`TELEGRAM_PRIVATE_APP_SEND_REQUIRED ${token}\n`); + try { + const replyText = await readAcknowledgement({ + acknowledgementPath, + destination: params.destination, + sentText: params.text, + token, + }); + return { replyText }; + } finally { + await Promise.all([ + fsp.rm(requestPath, { force: true }), + fsp.rm(acknowledgementPath, { force: true }), + ]); + } +} diff --git a/extensions/qa-lab/src/live-transports/telegram/run-options.runtime.test.ts b/extensions/qa-lab/src/live-transports/telegram/run-options.runtime.test.ts index 0df80f6bb230..cba0fcc49dc8 100644 --- a/extensions/qa-lab/src/live-transports/telegram/run-options.runtime.test.ts +++ b/extensions/qa-lab/src/live-transports/telegram/run-options.runtime.test.ts @@ -25,4 +25,12 @@ describe("resolveTelegramQaRunOptions", () => { "supports only --credential-source convex", ); }); + + it("preserves the private production credential descriptor", () => { + expect( + resolveTelegramQaRunOptions({ + credentialFile: "/private/telegram-production.json", + }), + ).toMatchObject({ credentialFile: "/private/telegram-production.json" }); + }); }); diff --git a/extensions/qa-lab/src/live-transports/telegram/run-options.runtime.ts b/extensions/qa-lab/src/live-transports/telegram/run-options.runtime.ts index af67f399dd6f..8444e43d35ee 100644 --- a/extensions/qa-lab/src/live-transports/telegram/run-options.runtime.ts +++ b/extensions/qa-lab/src/live-transports/telegram/run-options.runtime.ts @@ -39,6 +39,7 @@ export function resolveTelegramQaRunOptions( scenarioIds: opts.scenarioIds, listScenarios: opts.listScenarios, sutAccountId: opts.sutAccountId, + credentialFile: opts.credentialFile, credentialSource, credentialRole: opts.credentialRole?.trim(), }; diff --git a/extensions/qa-lab/src/live-transports/telegram/telegram-api.runtime.test.ts b/extensions/qa-lab/src/live-transports/telegram/telegram-api.runtime.test.ts index ee5bd4a68a8e..262308cede80 100644 --- a/extensions/qa-lab/src/live-transports/telegram/telegram-api.runtime.test.ts +++ b/extensions/qa-lab/src/live-transports/telegram/telegram-api.runtime.test.ts @@ -55,6 +55,26 @@ describe("Telegram QA API boundary", () => { }); }); + it("omits apiRoot for a production Bot API qualification", () => { + const config = buildTelegramQaConfig( + {}, + { + groupId: "-10042", + sutAccountId: "sut", + sutToken: "secret-token", + testerUserId: "100", + additionalTesterUserIds: ["101"], + enableDirectMessages: true, + }, + ); + + expect(config.channels?.telegram?.accounts?.sut).not.toHaveProperty("apiRoot"); + expect(config.channels?.telegram?.accounts?.sut).toMatchObject({ + allowFrom: ["100", "101"], + groups: { "-10042": { allowFrom: ["100", "101"] } }, + }); + }); + it("waits for the selected Telegram account to connect", async () => { const call = vi .fn() diff --git a/extensions/qa-lab/src/live-transports/telegram/telegram-api.runtime.ts b/extensions/qa-lab/src/live-transports/telegram/telegram-api.runtime.ts index 9d0e78c56d22..d190cd17c10b 100644 --- a/extensions/qa-lab/src/live-transports/telegram/telegram-api.runtime.ts +++ b/extensions/qa-lab/src/live-transports/telegram/telegram-api.runtime.ts @@ -23,14 +23,18 @@ const TELEGRAM_QA_DEFAULT_READY_TIMEOUT_MS = 45_000; export function buildTelegramQaConfig( baseCfg: OpenClawConfig, params: { - apiRoot: string; + apiRoot?: string; directMessageOnly?: boolean; + enableDirectMessages?: boolean; + additionalTesterUserIds?: string[]; + forumGroupId?: string; groupId: string; sutAccountId: string; sutToken: string; testerUserId: string; }, ): OpenClawConfig { + const testerUserIds = [params.testerUserId, ...(params.additionalTesterUserIds ?? [])]; return { ...baseCfg, agents: { @@ -71,19 +75,25 @@ export function buildTelegramQaConfig( [params.sutAccountId]: { enabled: true, botToken: params.sutToken, - apiRoot: params.apiRoot, - ...(params.directMessageOnly - ? { dmPolicy: "allowlist", allowFrom: [params.testerUserId] } + ...(params.apiRoot ? { apiRoot: params.apiRoot } : {}), + ...(params.directMessageOnly || params.enableDirectMessages + ? { dmPolicy: "allowlist", allowFrom: testerUserIds } : { dmPolicy: "disabled" }), - groups: { - [params.groupId]: { - groupPolicy: "allowlist", - allowFrom: [params.testerUserId], - // Concurrent leases share this group and QA sender. Only this - // bot's mentions or reply chain may trigger an agent turn. - requireMention: true, - }, - }, + groups: Object.fromEntries( + uniqueStrings([ + params.groupId, + ...(params.forumGroupId ? [params.forumGroupId] : []), + ]).map((groupId) => [ + groupId, + { + groupPolicy: "allowlist", + allowFrom: testerUserIds, + // Concurrent leases share this group and QA sender. Only this + // bot's mentions or reply chain may trigger an agent turn. + requireMention: true, + }, + ]), + ), }, }, }, diff --git a/extensions/qa-lab/src/live-transports/telegram/userbot-driver.runtime.test.ts b/extensions/qa-lab/src/live-transports/telegram/userbot-driver.runtime.test.ts index c6744f5be54a..365d214f646e 100644 --- a/extensions/qa-lab/src/live-transports/telegram/userbot-driver.runtime.test.ts +++ b/extensions/qa-lab/src/live-transports/telegram/userbot-driver.runtime.test.ts @@ -58,6 +58,10 @@ describe("Telegram userbot driver runtime", () => { "print(json.dumps({'type':'ready','chatId':-1001,'user':{'id':100}}), flush=True)", "for line in sys.stdin:", " request = json.loads(line)", + " if request['method'] == 'cleanup-private-forum':", + " print(json.dumps({'type':'response','id':request['id'],'result':{'ok':True,'status':'deleted'}}), flush=True)", + " continue", + " assert request['chatId'] == '-2002' and request['forumTopicId'] == 42", " message_id = 10 + int(request['id'])", " update = {'kind':'message','chatId':-1001,'messageId':message_id + 1,'senderId':200,'timestamp':1000,'text':request['text'],'entities':entities,'contentType':'messagePhoto'}", " print(json.dumps({'type':'update','update':update}), flush=True)", @@ -67,7 +71,7 @@ describe("Telegram userbot driver runtime", () => { " for rich in [rich_message, {**rich_message, 'is_full':False}, None]:", " update = {**update, 'richMessage':rich, 'text':'x', 'entities':[], 'contentType':'messageRichMessage' if rich else 'messageText'}", " print(json.dumps({'type':'update','update':update}), flush=True)", - " result = {'chatId':-1001,'messageId':message_id,'senderId':100,'timestamp':1000,'text':request['text'],'entities':entities,'contentType':'messageText'}", + " result = {'chatId':-1001,'messageId':message_id,'senderId':100,'timestamp':1000,'text':request['text'],'entities':entities,'contentType':'messageText','forumTopicId':request['forumTopicId']}", " print(json.dumps({'type':'response','id':request['id'],'result':result}), flush=True)", ].join("\n"), ); @@ -75,6 +79,7 @@ describe("Telegram userbot driver runtime", () => { const leaseFailure = new Promise(() => {}); const driver = await TelegramUserbotDriver.start({ chatId: "-1001", + expectedUserId: "100", driverEnv: {}, leaseHealth: { assertHealthy() {}, whenUnhealthy: leaseFailure }, userDriverPath: scriptPath, @@ -84,13 +89,16 @@ describe("Telegram userbot driver runtime", () => { }); expect(driver.chatId).toBe(-1001); try { - await expect(driver.send({ text })).resolves.toMatchObject({ - messageId: 11, - senderId: 100, - contentType: "messageText", - text, - entities, - }); + await expect(driver.send({ text, chatId: "-2002", forumTopicId: 42 })).resolves.toMatchObject( + { + forumTopicId: 42, + messageId: 11, + senderId: 100, + contentType: "messageText", + text, + entities, + }, + ); await vi.waitFor(() => expect(updates).toHaveLength(6)); expect(updates).toMatchObject([ { @@ -120,6 +128,7 @@ describe("Telegram userbot driver runtime", () => { ]); expect(updates[5]).not.toHaveProperty("richMessage"); expect(() => driver.assertHealthy()).not.toThrow(); + await expect(driver.cleanupPrivateForum()).resolves.toBeUndefined(); } finally { await driver.close(); } diff --git a/extensions/qa-lab/src/live-transports/telegram/userbot-driver.runtime.ts b/extensions/qa-lab/src/live-transports/telegram/userbot-driver.runtime.ts index 9c0512f58e2d..0320735bc2b3 100644 --- a/extensions/qa-lab/src/live-transports/telegram/userbot-driver.runtime.ts +++ b/extensions/qa-lab/src/live-transports/telegram/userbot-driver.runtime.ts @@ -18,12 +18,26 @@ export type TelegramUserbotUpdate = { kind: "edit" | "message"; messageId: number; replyToMessageId?: number; + forumTopicId?: number; + threadId?: number; senderId: number; senderUsername?: string; text: string; timestamp: number; }; +type PendingUserbotCommand = + | { + kind: "send"; + reject(error: Error): void; + resolve(value: TelegramUserbotUpdate): void; + } + | { + kind: "cleanup-private-forum"; + reject(error: Error): void; + resolve(): void; + }; + function isUtf16Boundary(text: string, offset: number) { const before = text.charCodeAt(offset - 1); const after = text.charCodeAt(offset); @@ -107,6 +121,8 @@ function parseUserbotUpdate(value: unknown): TelegramUserbotUpdate { ...(typeof value.replyToMessageId === "number" ? { replyToMessageId: value.replyToMessageId } : {}), + ...(typeof value.forumTopicId === "number" ? { forumTopicId: value.forumTopicId } : {}), + ...(typeof value.threadId === "number" ? { threadId: value.threadId } : {}), ...(typeof value.senderUsername === "string" ? { senderUsername: value.senderUsername } : {}), }; } @@ -131,12 +147,10 @@ function waitForChildExit(child: ChildProcessWithoutNullStreams, timeoutMs: numb export class TelegramUserbotDriver { private activeChatId: number | undefined; + private activeUserId: number | undefined; private closing = false; private commandId = 0; - private readonly pending = new Map< - string, - { reject(error: Error): void; resolve(value: TelegramUserbotUpdate): void } - >(); + private readonly pending = new Map(); private readyReject: (error: Error) => void = () => undefined; private readyResolve: () => void = () => undefined; private readonly ready: Promise; @@ -181,16 +195,28 @@ export class TelegramUserbotDriver { static async start(params: { chatId: string; + observeChatIds?: string[]; + expectedUserId?: string; driverEnv: Record; leaseHealth: { assertHealthy(): void; whenUnhealthy: Promise }; onUpdate(update: TelegramUserbotUpdate): Promise | void; userDriverPath: string; }): Promise { params.leaseHealth.assertHealthy(); - const child = spawn("python3", [params.userDriverPath, "serve", "--chat", params.chatId], { - env: { ...process.env, ...params.driverEnv }, - stdio: ["pipe", "pipe", "pipe"], - }); + const child = spawn( + "python3", + [ + params.userDriverPath, + "serve", + "--chat", + params.chatId, + ...(params.observeChatIds ?? []).flatMap((chatId) => ["--observe-chat", chatId]), + ], + { + env: { ...process.env, ...params.driverEnv }, + stdio: ["pipe", "pipe", "pipe"], + }, + ); const driver = new TelegramUserbotDriver( child, (update) => params.onUpdate(update), @@ -203,9 +229,15 @@ export class TelegramUserbotDriver { timer.unref?.(); try { await driver.ready; + if ( + params.expectedUserId !== undefined && + String(driver.activeUserId) !== params.expectedUserId + ) { + throw new Error("Telegram userbot authorization does not match the leased participant."); + } return driver; } catch (error) { - child.kill("SIGTERM"); + await driver.close(); throw error; } finally { clearTimeout(timer); @@ -231,6 +263,8 @@ export class TelegramUserbotDriver { return; } this.activeChatId = chatId; + this.activeUserId = + isRecord(message.user) && typeof message.user.id === "number" ? message.user.id : undefined; this.readyResolve(); return; } @@ -263,6 +297,17 @@ export class TelegramUserbotDriver { pending.reject(new Error("Telegram userbot emitted an invalid command result.")); return; } + if (pending.kind === "cleanup-private-forum") { + if ( + message.result.ok !== true || + (message.result.status !== "deleted" && message.result.status !== "not-created") + ) { + pending.reject(new Error("Telegram userbot did not confirm private forum cleanup.")); + return; + } + pending.resolve(); + return; + } try { pending.resolve(parseUserbotUpdate({ ...message.result, kind: "message" })); } catch (error) { @@ -298,20 +343,35 @@ export class TelegramUserbotDriver { return this.activeChatId; } - async send(params: { replyToMessageId?: number; text: string }): Promise { + async send(params: { + chatId?: string; + forumTopicId?: number; + replyToMessageId?: number; + text: string; + }): Promise { this.leaseHealth.assertHealthy(); this.assertHealthy(); this.commandId += 1; const id = String(this.commandId); const result = new Promise((resolve, reject) => { - this.pending.set(id, { resolve, reject }); + this.pending.set(id, { kind: "send", resolve, reject }); }); - this.child.stdin.write( - `${JSON.stringify({ id, method: "send", text: params.text, replyToMessageId: params.replyToMessageId })}\n`, - ); + this.child.stdin.write(`${JSON.stringify({ id, method: "send", ...params })}\n`); return await result; } + async cleanupPrivateForum() { + this.leaseHealth.assertHealthy(); + this.assertHealthy(); + this.commandId += 1; + const id = String(this.commandId); + const result = new Promise((resolve, reject) => { + this.pending.set(id, { kind: "cleanup-private-forum", resolve, reject }); + }); + this.child.stdin.write(`${JSON.stringify({ id, method: "cleanup-private-forum" })}\n`); + await result; + } + async close() { if (this.closing) { return; @@ -323,7 +383,9 @@ export class TelegramUserbotDriver { } if (!(await waitForChildExit(this.child, 2_000))) { this.child.kill("SIGKILL"); - await waitForChildExit(this.child, 2_000); + if (!(await waitForChildExit(this.child, 2_000))) { + throw new Error("Telegram userbot process exit is unconfirmed."); + } } await this.updateChain; } diff --git a/extensions/qa-lab/src/live-transports/telegram/userbot-skill.runtime.ts b/extensions/qa-lab/src/live-transports/telegram/userbot-skill.runtime.ts index 84e671431dd7..560b0a249ace 100644 --- a/extensions/qa-lab/src/live-transports/telegram/userbot-skill.runtime.ts +++ b/extensions/qa-lab/src/live-transports/telegram/userbot-skill.runtime.ts @@ -4,17 +4,23 @@ import { pathToFileURL } from "node:url"; import { isRecord } from "openclaw/plugin-sdk/string-coerce-runtime"; import { resolvePreferredOpenClawTmpDir } from "openclaw/plugin-sdk/temp-path"; -export type TelegramTestCredential = { +type TelegramUserCredential = { + testerUserId: string; + tdlibArchiveBase64: string; + tdlibArchiveSha256: string; + tdlibVersion: string; +}; + +export type TelegramTestCredential = TelegramUserCredential & { environment: "test"; groupId: string; schemaVersion: 1; sutBotId: string; sutToken: string; sutUsername: string; - tdlibArchiveBase64: string; - tdlibArchiveSha256: string; - tdlibVersion: string; - testerUserId: string; + forumGroupId?: string; + forumTopicId?: number; + participants?: Array; }; type RestoredTelegramTestCredential = TelegramTestCredential & { @@ -132,7 +138,32 @@ export async function loadTelegramUserbotSkillRuntime(params?: { createStateRoot: () => fs.mkdtempSync(path.join(resolvePreferredOpenClawTmpDir(), "openclaw-qa-telegram-")), parseCredential(value) { - return parseCredentialResult(Reflect.apply(parseCredential, undefined, [value])); + const normalized: unknown = Reflect.apply(parseCredential, undefined, [value]); + const credential = parseCredentialResult(normalized); + // The skill parser owns validation; retain only its normalized optional fixture fields. + if (isRecord(normalized)) { + if (typeof normalized.forumGroupId === "string") { + credential.forumGroupId = normalized.forumGroupId; + } + if (typeof normalized.forumTopicId === "number") { + credential.forumTopicId = normalized.forumTopicId; + } + if (Array.isArray(normalized.participants)) { + credential.participants = normalized.participants.map((participant: unknown) => { + if (!isRecord(participant)) { + throw new Error("Telegram userbot parser returned an invalid participant."); + } + return { + alias: requireString(participant, "alias"), + testerUserId: requireString(participant, "testerUserId"), + tdlibArchiveBase64: requireString(participant, "tdlibArchiveBase64"), + tdlibArchiveSha256: requireString(participant, "tdlibArchiveSha256"), + tdlibVersion: requireString(participant, "tdlibVersion"), + }; + }); + } + } + return credential; }, restoreCredential(value, stateRoot) { return parseRestoredCredential( diff --git a/extensions/qa-lab/src/live-transports/whatsapp/participant-identity-flow.test.ts b/extensions/qa-lab/src/live-transports/whatsapp/participant-identity-flow.test.ts new file mode 100644 index 000000000000..039c2135babf --- /dev/null +++ b/extensions/qa-lab/src/live-transports/whatsapp/participant-identity-flow.test.ts @@ -0,0 +1,172 @@ +import { describe, expect, it } from "vitest"; +import { createQaBusState } from "../../bus-state.js"; +import { readQaScenarioById } from "../../scenario-catalog.js"; +import { runLoadedScenarioFlow } from "../../scenario-flow-runner.test-support.js"; + +const scenarioId = "whatsapp-participant-identity-inspection"; +const driverPhone = "+15550000001"; +const groupJid = "120363000000000000@g.us"; + +type Failure = + | "missing-group" + | "extra-run" + | "unknown-person" + | "room-principal" + | "wrong-run" + | "wrong-execution" + | "raw-phone" + | "raw-group" + | "verified-generic" + | "different-person" + | "missing-human" + | "changed-after-restart"; + +async function runIdentityFlow(failure?: Failure) { + const state = createQaBusState(); + const admittedRuns: string[] = ["previous-run"]; + const inspectedExecutions = new Set(); + let restarted = false; + const result = await runLoadedScenarioFlow(scenarioId, { + state, + api: { + transport: { + id: "whatsapp", + reset: async () => state.reset(), + sendInbound: async (input: Parameters[0]) => + state.addInboundMessage({ ...input, senderId: driverPhone }), + waitForOutbound: async () => { + const inbound = state.getSnapshot().messages.at(-1); + if (!inbound || inbound.direction !== "inbound") { + throw new Error("identity proof must send a transport turn before inspecting it"); + } + const isGroup = inbound.conversation.kind === "group"; + expect(inbound.text.startsWith("openclawqa ")).toBe(isGroup); + admittedRuns.push(isGroup ? "group-run" : "dm-run"); + if (failure === "extra-run") { + admittedRuns.push("unrelated-new-run"); + } + const marker = inbound.text.split("Reply exactly: ")[1]; + if (!marker) { + throw new Error("identity proof must request an exact reply marker"); + } + return state.addOutboundMessage({ + accountId: "sut", + to: `${isGroup ? "group" : "dm"}:${inbound.conversation.id}`, + text: marker, + }); + }, + }, + env: { + providerMode: "mock-openai", + gateway: { + restartAfterStateMutation: async (mutate: () => Promise) => { + await mutate(); + restarted = true; + }, + }, + }, + // The real adapter prepares this from its lease; support tests never acquire one. + whatsappScenarioContext: { + runtimeEnv: { + driverPhoneE164: driverPhone, + sutPhoneE164: "+15550000002", + ...(failure === "missing-group" ? {} : { groupJid }), + }, + }, + runQaCli: async (_env: unknown, args: string[], options?: { json?: boolean }) => { + if (args.includes("--kind")) { + return { events: admittedRuns.map((runId) => ({ runId })) }; + } + const exact = args.includes("--execution"); + const selector = args[args.indexOf(exact ? "--execution" : "--run") + 1]; + if (!selector) { + throw new Error("identity proof must select a run or execution"); + } + const runId = exact ? selector.replace("execution-", "") : selector; + expect(admittedRuns).toContain(runId); + expect(runId).not.toBe("previous-run"); + if (exact) { + inspectedExecutions.add(selector); + } + if (!options?.json) { + return failure === "missing-human" + ? "Identity unavailable" + : "Invoker [present]\nDecisions"; + } + const isGroup = runId === "group-run"; + return { + identity: { + state: "present", + context: { + contextId: `context-${runId}`, + executionId: + failure === "wrong-execution" && exact + ? "unrelated-execution" + : `execution-${runId}`, + runId: failure === "wrong-run" ? "unrelated-run" : runId, + invoker: { + state: failure === "unknown-person" ? "unknown" : "present", + principal: { + kind: failure === "room-principal" ? "service" : "person", + principalRef: + (failure === "different-person" && isGroup) || + (failure === "changed-after-restart" && restarted) + ? "principal:other-person" + : "principal:driver", + }, + }, + ingress: { kind: "channel", state: "present" }, + }, + }, + decisionDisplays: [ + { + action: { family: "decision", operation: "record" }, + provenance: { + state: failure === "verified-generic" ? "verified" : "unverified", + }, + }, + ], + ...(failure === "raw-phone" ? { leakedReference: driverPhone.slice(1) } : {}), + ...(failure === "raw-group" ? { leakedReference: groupJid } : {}), + }; + }, + }, + }); + return { result, inspectedExecutions, restarted }; +} + +describe("WhatsApp participant identity executable flow", () => { + it("selects the real transport lane and inspects both executions across restart", async () => { + const scenario = readQaScenarioById(scenarioId); + expect(scenario.execution).toMatchObject({ + kind: "flow", + channel: "whatsapp", + suiteIsolation: "isolated", + config: { requiredChannelDriver: "live", requiredProviderMode: "mock-openai" }, + }); + expect(scenario.gatewayConfigPatch).toMatchObject({ + logging: { audit: { executionIdentity: true } }, + }); + const proof = await runIdentityFlow(); + expect(proof.result.status).toBe("pass"); + expect(proof.inspectedExecutions).toEqual(new Set(["execution-dm-run", "execution-group-run"])); + expect(proof.restarted).toBe(true); + }); + + it.each([ + ["missing-group", "requires groupJid"], + ["extra-run", "exactly one newly admitted run"], + ["unknown-person", "retain the admitted person"], + ["room-principal", "retain the admitted person"], + ["wrong-run", "retain the admitted person"], + ["wrong-execution", "must agree with run discovery"], + ["raw-phone", "exclude raw route and participant references"], + ["raw-group", "exclude raw route and participant references"], + ["verified-generic", "keep generic decisions unverified"], + ["different-person", "same participant"], + ["missing-human", "must agree with run discovery"], + ["changed-after-restart", "changed or exposed private references after restart"], + ] satisfies Array<[Failure, string]>)("rejects %s evidence", async (failure, error) => { + await expect(runIdentityFlow(failure)).rejects.toThrow(error); + }); +}); diff --git a/extensions/qa-lab/src/scenario-catalog-causality.test.ts b/extensions/qa-lab/src/scenario-catalog-causality.test.ts index 356c030ab478..98a7ad8453f8 100644 --- a/extensions/qa-lab/src/scenario-catalog-causality.test.ts +++ b/extensions/qa-lab/src/scenario-catalog-causality.test.ts @@ -11,6 +11,54 @@ import { runLoadedScenarioFlow } from "./scenario-flow-runner.test-support.js"; import { createRestartFlowFixture } from "./scenario-restart-flow.test-support.js"; describe("qa scenario catalog causality", () => { + it("exposes the message tool directly for delivery decision inspection", () => { + const scenario = readQaScenarioById("message-delivery-decision-inspection"); + + expect(scenario.gatewayConfigPatch).toMatchObject({ + tools: { + toolSearch: false, + alsoAllow: ["message"], + }, + }); + }); + + it("requires one host-owned fallback for message suppression and no duplicate after restart", () => { + const scenario = requireFlowScenario( + readQaScenarioById("message-delivery-decision-inspection"), + ); + const suppressionActions = scenario.execution.flow?.steps[1]?.actions ?? []; + const restartActions = scenario.execution.flow?.steps[2]?.actions ?? []; + const outboundCount = + "state.getSnapshot().messages.filter((message) => message.direction === 'outbound').length"; + + expect(scenario.execution.config?.expectedFallbackText).toBe( + "The tool run finished, but no final summary was produced. I did not repeat any completed actions.", + ); + expect(suppressionActions).toEqual( + expect.arrayContaining([ + expect.objectContaining({ + waitForOutbound: { + conversation: { id: "qa-message-suppression-room", kind: "direct" }, + textIncludes: { ref: "config.expectedFallbackText" }, + timeoutMs: 60000, + }, + saveAs: "suppressionOutbound", + }), + ]), + ); + expect(suppressionActions.map(readFlowAssertExpression)).toContain( + `${outboundCount} === suppressionOutboundStart + 1 && suppressionOutbound.text === config.expectedFallbackText`, + ); + expect(restartActions.map(readFlowAssertExpression)).toContain( + `${outboundCount} === suppressionOutboundStart + 1`, + ); + expect([...suppressionActions, ...restartActions].map(readFlowAssertExpression)).not.toContain( + `${outboundCount} === suppressionOutboundStart`, + ); + expect(JSON.stringify(scenario.execution.flow)).not.toContain("waitForNoOutbound"); + expect(JSON.stringify(scenario.execution.flow)).not.toContain("visibleOutbound=0"); + }); + it("treats denied Telegram admission as silent transport suppression", () => { for (const scenarioId of [ "telegram-policy-hot-reload", diff --git a/extensions/qa-lab/src/scenario-catalog-live-identity.test.ts b/extensions/qa-lab/src/scenario-catalog-live-identity.test.ts new file mode 100644 index 000000000000..bb81331c3e7d --- /dev/null +++ b/extensions/qa-lab/src/scenario-catalog-live-identity.test.ts @@ -0,0 +1,215 @@ +import { describe, expect, it, vi } from "vitest"; +import { readQaScenarioById } from "./scenario-catalog.js"; +import { runLoadedScenarioFlow } from "./scenario-flow-runner.test-support.js"; +import { selectQaFlowSuiteScenarios, splitModelRef } from "./suite-planning.js"; + +const SCENARIO = "live-frontier-execution-identity"; +type Fault = + | "none" + | "mock-lane" + | "model-fallback" + | "missing-identity" + | "raw-runtime" + | "wrong-run" + | "enforced-admission" + | "prompt-leak" + | "wrong-cli-execution" + | "reused-context" + | "restart-drift"; + +function runIdentityFlow(fault: Fault = "none") { + let turn = 0; + let marker = ""; + let restarted = false; + const contexts = new Map< + string, + { + runId: string; + contextId: string; + executionId: string; + ingress: { state: string; kind: string; boundary: string }; + invoker: { state: string }; + coverageState: string; + agentPrincipal: { kind: string; principalRef: string }; + agentDefinition: { definitionRef: string }; + trustDomain: { state: string; domainRef: string }; + runtimeInstance: { state: string; kind: string; runtimeRef: string }; + } + >(); + const call = vi.fn(async (method: string, params: Record) => { + if (method === "agent") { + turn += 1; + if (!params.message) { + throw new Error("live identity proof must send an agent message"); + } + marker = params.message.replace("Reply exactly: ", ""); + const runId = `run-${turn}`; + contexts.set(runId, { + runId, + contextId: fault === "reused-context" ? "context-1" : `context-${turn}`, + executionId: `execution-${turn}`, + ingress: { + state: "present", + kind: "gateway-client", + boundary: "gateway.ws.authenticated-connect", + }, + invoker: { state: "absent" }, + coverageState: "unattributed", + agentPrincipal: { kind: "agent", principalRef: "qa" }, + agentDefinition: { definitionRef: "qa" }, + trustDomain: { + state: "present", + domainRef: `hmac-sha256:v1:${"a".repeat(32)}:${"b".repeat(64)}`, + }, + runtimeInstance: { + state: "present", + kind: "embedded", + runtimeRef: + fault === "raw-runtime" + ? "private-runtime" + : `hmac-sha256:v1:${"a".repeat(32)}:${"c".repeat(64)}`, + }, + }); + return { status: "accepted", runId }; + } + if (method === "chat.history") { + return { + messages: [ + { + role: "assistant", + provider: fault === "model-fallback" ? "mock-openai" : "fixture-live", + model: "fixture-model", + stopReason: "stop", + content: [{ type: "text", text: marker }], + }, + ], + }; + } + if (method !== "audit.run.inspect") { + throw new Error(`unexpected Gateway method ${method}`); + } + const context = params.runId + ? contexts.get(params.runId) + : [...contexts.values()].find((entry) => entry.executionId === params.executionId); + if (!context) { + throw new Error("inspection selected an unknown execution"); + } + return { + run: { + runId: fault === "wrong-run" ? "foreign-run" : context.runId, + executionId: context.executionId, + }, + identity: + fault === "missing-identity" + ? { state: "unsupported" } + : { + state: "present", + context: { + ...context, + ...(restarted && fault === "restart-drift" ? { contextId: "replacement" } : {}), + }, + }, + decisionDisplays: [ + { + provenance: { state: "verified", producer: "run-admission" }, + decision: { + outcome: fault === "enforced-admission" ? "allowed" : "not-applicable", + reasonCode: "run_admission_identity_not_evaluated", + }, + }, + ], + ...(fault === "prompt-leak" ? { privateText: marker } : {}), + }; + }); + const restart = vi.fn(async (mutate: () => Promise) => { + await mutate(); + restarted = true; + }); + const runQaCli = vi.fn(async (_env: unknown, args: string[]) => { + expect(args.slice(0, 2)).toEqual(["audit", "--execution"]); + const executionId = args[2]; + if (!executionId) { + throw new Error("live identity proof must select an execution"); + } + const inspection = await call("audit.run.inspect", { executionId }); + if (!args.includes("--json")) { + return "Identity Invoker [absent] Decisions run_admission_identity_not_evaluated not-applicable"; + } + if (fault === "wrong-cli-execution") { + return { ...inspection, run: { executionId: "foreign-execution" } }; + } + return inspection; + }); + return { + call, + restart, + runQaCli, + result: runLoadedScenarioFlow(SCENARIO, { + api: { + env: { + providerMode: fault === "mock-lane" ? "mock-openai" : "live-frontier", + primaryModel: "fixture-live/fixture-model", + gateway: { call, restartAfterStateMutation: restart }, + }, + splitModelRef, + waitForAgentRun: async (_env: unknown, runId: string) => { + expect(contexts.has(runId)).toBe(true); + return { status: "ok" }; + }, + waitForAgentHistoryReply: async ( + _env: unknown, + _session: string, + predicate: (text: string) => boolean, + ) => { + expect(predicate(marker)).toBe(true); + return marker; + }, + runQaCli, + }, + }), + }; +} + +describe("live-frontier execution identity qualification", () => { + it("rejects mock selection and admits the operator-selected live model", () => { + const scenario = readQaScenarioById(SCENARIO); + const selection = { scenarios: [scenario], primaryModel: "fixture-live/fixture-model" }; + expect(selectQaFlowSuiteScenarios({ ...selection, providerMode: "mock-openai" })).toEqual([]); + expect(() => + selectQaFlowSuiteScenarios({ + ...selection, + providerMode: "mock-openai", + scenarioIds: [SCENARIO], + }), + ).toThrow("providerMode=live-frontier"); + expect(selectQaFlowSuiteScenarios({ ...selection, providerMode: "live-frontier" })).toEqual([ + scenario, + ]); + expect(scenario.gatewayConfigPatch).toMatchObject({ + logging: { audit: { enabled: true, executionIdentity: true } }, + }); + }); + + it("executes the registered flow through admission, CLI inspection, and replacement readback", async () => { + const fixture = runIdentityFlow(); + await expect(fixture.result).resolves.toMatchObject({ status: "pass" }); + expect(fixture.call.mock.calls.filter(([method]) => method === "agent")).toHaveLength(2); + expect(fixture.runQaCli).toHaveBeenCalledTimes(4); + expect(fixture.restart).toHaveBeenCalledOnce(); + }); + + it.each([ + ["mock-lane", "requires live-frontier"], + ["model-fallback", "selected live provider and model"], + ["missing-identity", "test condition was not met"], + ["raw-runtime", "pseudonymized runtime identity"], + ["wrong-run", "exact authenticated ingress"], + ["enforced-admission", "overstated admission authority"], + ["prompt-leak", "exposed private content"], + ["wrong-cli-execution", "different live execution"], + ["reused-context", "reused one execution identity"], + ["restart-drift", "changed across Gateway replacement"], + ] as const)("rejects %s evidence from the same executable flow", async (fault, message) => { + await expect(runIdentityFlow(fault).result).rejects.toThrow(message); + }); +}); diff --git a/qa/convex-credential-broker/README.md b/qa/convex-credential-broker/README.md index 37b0089d7d00..145415e0b756 100644 --- a/qa/convex-credential-broker/README.md +++ b/qa/convex-credential-broker/README.md @@ -152,7 +152,10 @@ For `kind: "telegram"`, broker `admin/add` validates that payload includes: For `kind: "telegram-test-userbot"`, broker `admin/add` accepts only Test Server schema version 1 with numeric chat, bot, and tester ids; a bot token and -username; a base64 TDLib archive and SHA-256 hash; and a TDLib version. +username; a base64 TDLib archive and SHA-256 hash; and a TDLib version. Optional +participant-identity fixtures add a negative `forumGroupId`, positive numeric +`forumTopicId`, and independently authorized `participants` with distinct +lowercase aliases, tester ids, TDLib archives, hashes, and versions. For `kind: "buzz"`, broker `admin/add` validates that payload includes: diff --git a/qa/convex-credential-broker/convex/payload_validation.ts b/qa/convex-credential-broker/convex/payload_validation.ts index d568d0813cca..0c4146e546e0 100644 --- a/qa/convex-credential-broker/convex/payload_validation.ts +++ b/qa/convex-credential-broker/convex/payload_validation.ts @@ -230,42 +230,114 @@ function normalizeTelegramTestUserbotCredentialPayload( `Credential payload for kind "${kind}" must use schemaVersion 1 and environment "test".`, ); } + const normalizeUser = (user: Record) => { + const testerUserId = requirePayloadString(user, "testerUserId", kind, createFailure); + if (!TELEGRAM_USER_ID_RE.test(testerUserId)) { + throwPayloadError( + createFailure, + `Credential payload for kind "${kind}" has invalid tester identity.`, + ); + } + const tdlibArchiveBase64 = requirePayloadString( + user, + "tdlibArchiveBase64", + kind, + createFailure, + ); + if (!BASE64_RE.test(tdlibArchiveBase64) || tdlibArchiveBase64.length % 4 !== 0) { + throwPayloadError( + createFailure, + `Credential payload for kind "${kind}" has invalid tdlibArchiveBase64.`, + ); + } + const tdlibArchiveSha256 = requirePayloadString( + user, + "tdlibArchiveSha256", + kind, + createFailure, + ).toLowerCase(); + if (!SHA256_HEX_RE.test(tdlibArchiveSha256)) { + throwPayloadError( + createFailure, + `Credential payload for kind "${kind}" has invalid tdlibArchiveSha256.`, + ); + } + return { + testerUserId, + tdlibArchiveBase64, + tdlibArchiveSha256, + tdlibVersion: requirePayloadString(user, "tdlibVersion", kind, createFailure), + }; + }; const groupId = requirePayloadString(payload, "groupId", kind, createFailure); const sutBotId = requirePayloadString(payload, "sutBotId", kind, createFailure); - const testerUserId = requirePayloadString(payload, "testerUserId", kind, createFailure); if (!TELEGRAM_CHAT_ID_RE.test(groupId)) { throwPayloadError(createFailure, `Credential payload for kind "${kind}" has invalid groupId.`); } - if (!TELEGRAM_USER_ID_RE.test(sutBotId) || !TELEGRAM_USER_ID_RE.test(testerUserId)) { + if (!TELEGRAM_USER_ID_RE.test(sutBotId)) { throwPayloadError( createFailure, - `Credential payload for kind "${kind}" has invalid bot or tester identity.`, + `Credential payload for kind "${kind}" has invalid bot identity.`, ); } - const tdlibArchiveBase64 = requirePayloadString( - payload, - "tdlibArchiveBase64", - kind, - createFailure, - ); - if (!BASE64_RE.test(tdlibArchiveBase64) || tdlibArchiveBase64.length % 4 !== 0) { + const primary = normalizeUser(payload); + const forumGroupId = + payload.forumGroupId === undefined + ? undefined + : requirePayloadString(payload, "forumGroupId", kind, createFailure); + if (forumGroupId && !/^-\d+$/u.test(forumGroupId)) { throwPayloadError( createFailure, - `Credential payload for kind "${kind}" has invalid tdlibArchiveBase64.`, + `Credential payload for kind "${kind}" has invalid forumGroupId.`, ); } - const tdlibArchiveSha256 = requirePayloadString( - payload, - "tdlibArchiveSha256", - kind, - createFailure, - ).toLowerCase(); - if (!SHA256_HEX_RE.test(tdlibArchiveSha256)) { + const forumTopicId = payload.forumTopicId; + if ( + forumTopicId !== undefined && + (!Number.isSafeInteger(forumTopicId) || Number(forumTopicId) <= 0) + ) { throwPayloadError( createFailure, - `Credential payload for kind "${kind}" has invalid tdlibArchiveSha256.`, + `Credential payload for kind "${kind}" has invalid forumTopicId.`, ); } + let participants: Array & { alias: string }> | undefined; + if (payload.participants !== undefined) { + if (!Array.isArray(payload.participants)) { + throwPayloadError( + createFailure, + `Credential payload for kind "${kind}" has invalid participants.`, + ); + } + const aliases = new Set(["primary"]); + const identities = new Set([primary.testerUserId]); + participants = payload.participants.map((value) => { + if (!value || typeof value !== "object" || Array.isArray(value)) { + throwPayloadError( + createFailure, + `Credential payload for kind "${kind}" has invalid participant.`, + ); + } + const participant = value as Record; + const alias = requirePayloadString(participant, "alias", kind, createFailure); + if (!/^[a-z][a-z0-9-]*$/u.test(alias) || aliases.has(alias)) { + throwPayloadError( + createFailure, + `Credential payload for kind "${kind}" requires distinct lowercase participant aliases.`, + ); + } + const user = normalizeUser(participant); + if (identities.has(user.testerUserId)) { + throwPayloadError( + createFailure, + `Credential payload for kind "${kind}" requires distinct participant identities.`, + ); + } + aliases.add(alias); + identities.add(user.testerUserId); + return { alias, ...user }; + }); + } return { schemaVersion: 1, environment: "test", @@ -276,10 +348,10 @@ function normalizeTelegramTestUserbotCredentialPayload( "", ), sutBotId, - testerUserId, - tdlibArchiveBase64, - tdlibArchiveSha256, - tdlibVersion: requirePayloadString(payload, "tdlibVersion", kind, createFailure), + ...primary, + ...(forumGroupId ? { forumGroupId } : {}), + ...(forumTopicId === undefined ? {} : { forumTopicId: Number(forumTopicId) }), + ...(participants ? { participants } : {}), } satisfies Record; } diff --git a/qa/scenarios/channels/message-delivery-decision-inspection.yaml b/qa/scenarios/channels/message-delivery-decision-inspection.yaml index 18f1b791c9d8..aec2b3a00a94 100644 --- a/qa/scenarios/channels/message-delivery-decision-inspection.yaml +++ b/qa/scenarios/channels/message-delivery-decision-inspection.yaml @@ -19,6 +19,7 @@ scenario: qa-channel: allowFrom: ["*"] tools: + toolSearch: false alsoAllow: [message] plugins: allow: [qa-lab, qa-channel] @@ -35,10 +36,10 @@ scenario: alsoAllow: [message] successCriteria: - A QA Channel reply records queued, platform-started, and delivered as attribution-only owner-native receipts. - - A message-tool metadata echo records one exact private suppression fact without a visible outbound message. + - A message-tool metadata echo records one exact private suppression fact and delivers exactly one host-owned fallback reply. - Public inspection reports the private generic fact as explicitly unknown without exposing its reason. - JSON and human CLI inspection expose no raw sender, conversation, message content, or platform message id. - - Both receipt sets remain stable after a lifecycle-owned Gateway restart. + - Both receipt sets remain stable after a lifecycle-owned Gateway restart without adding another outbound message. docsRefs: - docs/gateway/audit.md - docs/cli/audit.md @@ -57,6 +58,7 @@ scenario: summary: Drive delivered and intentionally suppressed channel turns through an ephemeral Gateway and mock provider, then inspect exact decision receipts before and after restart. config: requiredProviderMode: mock-openai + expectedFallbackText: "The tool run finished, but no final summary was produced. I did not repeat any completed actions." flow: steps: @@ -176,12 +178,14 @@ flow: - assert: expr: "suppressionPrivacySentinels.every((value) => typeof value === 'string' && value.length > 0)" message: suppression QA message did not expose complete real privacy sentinels - - waitForNoOutbound: - sinceIndex: { ref: suppressionOutboundStart } - quietMs: 1500 + - waitForOutbound: + conversation: { id: qa-message-suppression-room, kind: direct } + textIncludes: { ref: config.expectedFallbackText } + timeoutMs: 60000 + saveAs: suppressionOutbound - assert: - expr: "state.getSnapshot().messages.filter((message) => message.direction === 'outbound').length === suppressionOutboundStart" - message: suppression interval produced a visible outbound QA message + expr: "state.getSnapshot().messages.filter((message) => message.direction === 'outbound').length === suppressionOutboundStart + 1 && suppressionOutbound.text === config.expectedFallbackText" + message: suppression interval did not produce exactly one host-owned fallback reply - call: waitForCondition saveAs: suppressionRunId args: @@ -227,9 +231,9 @@ flow: expr: "suppressionPrivacySentinels.every((sentinel) => !JSON.stringify(suppressionInspect).includes(sentinel) && !suppressionText.includes(sentinel))" message: suppression JSON or human inspection exposed a raw sender, conversation, inbound message id, or content sentinel - assert: - expr: "state.getSnapshot().messages.filter((message) => message.direction === 'outbound').length === suppressionOutboundStart" - message: suppression path did not retain exact zero visible outbound messages - detailsExpr: "`suppression facts=1; publicCoverage=unknown; selectorIds=nonempty-unique; visibleOutbound=0; privacy=json+human-pass`" + expr: "state.getSnapshot().messages.filter((message) => message.direction === 'outbound').length === suppressionOutboundStart + 1" + message: suppression path did not retain exactly one visible fallback reply + detailsExpr: "`suppression facts=1; publicCoverage=unknown; selectorIds=nonempty-unique; visibleOutbound=1; privacy=json+human-pass`" - name: preserves delivery and suppression receipts across restart actions: @@ -280,6 +284,6 @@ flow: expr: "deliveryPrivacySentinels.every((sentinel) => !JSON.stringify(deliveryAfterRestart).includes(sentinel) && !deliveryTextAfterRestart.includes(sentinel)) && suppressionPrivacySentinels.every((sentinel) => !JSON.stringify(suppressionAfterRestart).includes(sentinel) && !suppressionTextAfterRestart.includes(sentinel))" message: restarted JSON or human inspection exposed a raw sender, conversation, message id, or content sentinel - assert: - expr: "state.getSnapshot().messages.filter((message) => message.direction === 'outbound').length === suppressionOutboundStart" - message: suppression path produced a visible outbound message after restart + expr: "state.getSnapshot().messages.filter((message) => message.direction === 'outbound').length === suppressionOutboundStart + 1" + message: suppression path added another outbound message after restart detailsExpr: "`restart-stable decisions=byte-identical; selectorIds=nonempty-unique; privacy=json+human-pass`" diff --git a/qa/scenarios/channels/telegram-participant-identity-inspection.yaml b/qa/scenarios/channels/telegram-participant-identity-inspection.yaml new file mode 100644 index 000000000000..0da0f0876391 --- /dev/null +++ b/qa/scenarios/channels/telegram-participant-identity-inspection.yaml @@ -0,0 +1,196 @@ +title: Telegram DM and forum participant identity inspection + +scenario: + id: telegram-participant-identity-inspection + surface: channels + coverage: + primary: + - gateway.identity-and-presence-apis + objective: Verify real Telegram DM and forum-topic admissions preserve one person identity across routes, distinguish a second participant, and retain bounded redacted execution identity across one Gateway restart. + gatewayConfigPatch: + logging: + audit: + enabled: true + executionIdentity: true + successCriteria: + - The primary QA user receives a DM reply and both participants receive replies in the configured forum topic. + - Each turn has one newly admitted run and distinct execution context; the primary person's pseudonym is stable across DM and forum while the additional person's differs. + - audit.run.inspect and audit CLI JSON and human output agree, retain bounded pseudonyms, and expose no raw participant, room, or message content. + - Exact execution identity survives one lifecycle-owned Gateway restart, and generic decision displays remain unverified. + docsRefs: + - docs/gateway/audit.md + - docs/cli/audit.md + - docs/concepts/qa-e2e-automation/channel-qa-reference.md + codeRefs: + - extensions/qa-lab/src/live-transports/telegram/adapter.runtime.ts + - extensions/qa-lab/src/live-transports/telegram/private-production.runtime.ts + execution: + kind: flow + channel: telegram + suiteIsolation: isolated + isolationReason: Enables identity before admission and inspects the same isolated audit database across a Gateway replacement restart. + timeoutMs: 900000 + retryCount: 0 + summary: Run with qa telegram --scenario telegram-participant-identity-inspection --provider-mode mock-openai. The default mode requires a Convex-leased Test Server fixture with distinct participants and a forum topic. An explicitly supplied private-production descriptor uses two operator-local Telegram.app participants and private native-observation acknowledgements without leasing a production user. Uses the live adapter and no live model; incomplete fixtures fail without fallback. + config: + requiredChannelDriver: live + requiredProviderMode: mock-openai + requireParticipantIdentityFixture: true + turns: + - conversation: { id: qa-telegram-identity-dm, kind: direct } + participantIndex: 0 + forum: false + - conversation: { id: qa-telegram-identity-forum, kind: group } + participantIndex: 0 + forum: true + - conversation: { id: qa-telegram-identity-forum, kind: group } + participantIndex: 1 + forum: true + +flow: + steps: + - name: inspects Telegram people across DM and forum executions + actions: + - assert: + expr: "transport.id === 'telegram' && env.providerMode === config.requiredProviderMode" + message: Telegram participant identity proof requires the live Telegram adapter and mock-openai provider + - assert: + expr: "typeof telegramIdentityFixture !== 'undefined' && telegramIdentityFixture.participantAliases?.[0] === 'primary' && telegramIdentityFixture.participantAliases.length >= 2 && new Set(telegramIdentityFixture.participantAliases).size === telegramIdentityFixture.participantAliases.length && Number.isSafeInteger(telegramIdentityFixture.forumTopicId) && telegramIdentityFixture.forumTopicId > 0" + message: Telegram participant identity proof requires distinct participant aliases and forumTopicId from the prepared fixture + - call: waitForGatewayHealthy + args: [{ ref: env }, 60000] + - call: waitForTransportReady + args: [{ ref: env }, 60000] + - resetTransport: true + - set: proofs + value: [] + - set: privateReferences + value: [] + - set: isSafeInspection + value: + lambda: + params: [value, privateRefs] + expr: >- + value.identity?.state === 'present' && + Buffer.byteLength(JSON.stringify(value.identity.context), 'utf8') <= 16384 && + [value.identity.context.invoker.principal?.principalRef, value.identity.context.invoker.principal?.domainRef, value.identity.context.runtimeInstance.runtimeRef, ...(value.identity.context.ingress.sourceRef === undefined ? [] : [value.identity.context.ingress.sourceRef]), ...value.identity.context.assurance.map((item) => item.evidenceRef), ...value.identity.context.applicableGrants.map((item) => item.grantRef)].every((ref) => typeof ref === 'string' && /^hmac-sha256:v1:[a-f0-9]{32}:[a-f0-9]{64}$/.test(ref)) && + value.identity.context.assurance.length <= 16 && value.identity.context.applicableGrants.length <= 16 && + value.identity.context.assurance.some((item) => item.kind === 'channel-admission' && item.strength === 'boundary-verified') && + value.decisionDisplays.length <= 100 && !Object.hasOwn(value, 'decisions') && + value.decisionDisplays.some((receipt) => receipt.action.family === 'decision') && + value.decisionDisplays.filter((receipt) => receipt.action.family === 'decision').every((receipt) => receipt.provenance.state === 'unverified') && + privateRefs.every((ref) => !JSON.stringify(value).includes(ref)) + - forEach: + items: { ref: config.turns } + item: turn + index: turnIndex + actions: + - set: marker + value: { expr: "`QA-TELEGRAM-IDENTITY-${turnIndex}-${randomUUID()}`" } + - set: runsBefore + value: + expr: "[...new Set((await runQaCli(env, ['audit', '--kind', 'agent_run', '--limit', '500', '--json'], { timeoutMs: 60000, json: true })).events.map((event) => event.runId).filter((id) => typeof id === 'string'))]" + - sendInbound: + conversation: { ref: turn.conversation } + senderId: + { expr: "telegramIdentityFixture.participantAliases[turn.participantIndex]" } + threadId: + { + expr: "turn.forum ? String(telegramIdentityFixture.forumTopicId) : undefined", + } + text: { expr: "`${turn.forum ? '@openclaw ' : ''}Reply exactly: ${marker}`" } + saveAs: inbound + - waitForOutbound: + conversation: { ref: turn.conversation } + threadId: + { + expr: "turn.forum ? String(telegramIdentityFixture.forumTopicId) : undefined", + } + textIncludes: { ref: marker } + timeoutMs: 60000 + - call: waitForCondition + saveAs: nativeReplies + args: + - lambda: + expr: "(replies => replies.length ? replies : undefined)(readTelegramMessages().filter((reply) => reply.text.includes(marker)))" + - 60000 + - 250 + - assert: + expr: "nativeReplies.every((reply) => turn.forum ? reply.forumTopicId === telegramIdentityFixture.forumTopicId && reply.chatId < 0 : !reply.forumTopicId && reply.chatId > 0)" + message: Telegram reply must be observed in the requested DM or leased forum topic + - set: privateReferences + value: + expr: "[...new Set([...privateReferences, inbound.senderId, turn.conversation.id, marker, ...nativeReplies.flatMap((reply) => [String(reply.chatId), String(reply.senderId)])])]" + - call: waitForCondition + saveAs: newRunIds + args: + - lambda: + async: true + expr: "(ids => ids.length ? ids : undefined)([...new Set((await runQaCli(env, ['audit', '--kind', 'agent_run', '--limit', '500', '--json'], { timeoutMs: 60000, json: true })).events.map((event) => event.runId).filter((id) => typeof id === 'string' && !runsBefore.includes(id)))])" + - 60000 + - 250 + - assert: + expr: "newRunIds.length === 1" + message: Telegram turn must produce exactly one newly admitted run + - call: waitForCondition + saveAs: runInspection + args: + - lambda: + async: true + expr: "env.gateway.call('audit.run.inspect', { runId: newRunIds[0], decisionLimit: 100 }).then((value) => value.identity?.state === 'present' && value.decisionDisplays.some((receipt) => receipt.action.family === 'decision') ? value : undefined)" + - 60000 + - 250 + - assert: + expr: "runInspection.run.runId === newRunIds[0] && runInspection.run.executionId === runInspection.identity.context.executionId && runInspection.identity.context.runId === newRunIds[0] && runInspection.identity.context.invoker.state === 'present' && runInspection.identity.context.invoker.principal?.kind === 'person' && runInspection.identity.context.ingress.kind === 'channel' && runInspection.identity.context.ingress.state === 'present'" + message: Telegram audit inspection must retain the admitted person and channel ingress for this exact run + - set: exactInspection + value: + expr: "await env.gateway.call('audit.run.inspect', { executionId: runInspection.identity.context.executionId, decisionLimit: 100 })" + - set: cliInspection + value: + expr: "await runQaCli(env, ['audit', '--execution', runInspection.identity.context.executionId, '--explain', '--json'], { timeoutMs: 60000, json: true })" + - set: humanInspection + value: + expr: "await runQaCli(env, ['audit', '--execution', runInspection.identity.context.executionId, '--explain'], { timeoutMs: 60000 })" + - assert: + expr: "[exactInspection, cliInspection].every((value) => value.run.executionId === runInspection.identity.context.executionId && JSON.stringify(value.identity.context) === JSON.stringify(runInspection.identity.context)) && humanInspection.includes('Invoker [present]') && humanInspection.includes(runInspection.identity.context.invoker.principal.principalRef) && humanInspection.includes('Decisions')" + message: Telegram exact RPC and CLI inspections must agree with run discovery + - assert: + expr: "[runInspection, exactInspection, cliInspection].every((value) => isSafeInspection(value, privateReferences)) && privateReferences.every((ref) => !humanInspection.includes(ref))" + message: Telegram inspection must retain bounded redacted identity and unverified generic decisions + - set: proofs + value: + expr: "[...proofs, { context: exactInspection.identity.context, senderId: inbound.senderId }]" + - assert: + expr: "proofs.length === 3 && new Set(proofs.map((proof) => proof.context.executionId)).size === 3 && new Set(proofs.map((proof) => proof.context.contextId)).size === 3 && proofs[0].senderId === proofs[1].senderId && proofs[1].senderId !== proofs[2].senderId && proofs[0].context.invoker.principal.principalRef === proofs[1].context.invoker.principal.principalRef && proofs[1].context.invoker.principal.principalRef !== proofs[2].context.invoker.principal.principalRef" + message: Telegram DM and forum must preserve the same primary person and distinguish the additional participant in three distinct execution contexts + detailsExpr: "proofs.map((proof) => `run=${proof.context.runId}; execution=${proof.context.executionId}`).join('; ')" + + - name: preserves exact Telegram identity across one Gateway restart + actions: + - call: env.gateway.restartAfterStateMutation + args: + - lambda: + async: true + expr: Promise.resolve() + - call: waitForGatewayHealthy + args: [{ ref: env }, 60000] + - call: waitForTransportReady + args: [{ ref: env }, 60000] + - forEach: + items: { ref: proofs } + item: proof + actions: + - set: restartedRpc + value: + expr: "await env.gateway.call('audit.run.inspect', { executionId: proof.context.executionId, decisionLimit: 100 })" + - set: restartedCli + value: + expr: "await runQaCli(env, ['audit', '--execution', proof.context.executionId, '--explain', '--json'], { timeoutMs: 60000, json: true })" + - set: restartedHuman + value: + expr: "await runQaCli(env, ['audit', '--execution', proof.context.executionId, '--explain'], { timeoutMs: 60000 })" + - assert: + expr: "[restartedRpc, restartedCli].every((value) => isSafeInspection(value, privateReferences) && value.run.executionId === proof.context.executionId && JSON.stringify(value.identity.context) === JSON.stringify(proof.context)) && restartedHuman.includes('Invoker [present]') && restartedHuman.includes(proof.context.invoker.principal.principalRef) && restartedHuman.includes('Decisions') && privateReferences.every((ref) => !restartedHuman.includes(ref))" + message: Telegram exact participant identity changed or exposed private references after restart + detailsExpr: "`restart-stable executions=${proofs.length}`" diff --git a/qa/scenarios/channels/whatsapp-participant-identity-inspection.yaml b/qa/scenarios/channels/whatsapp-participant-identity-inspection.yaml new file mode 100644 index 000000000000..4a5e7a03102c --- /dev/null +++ b/qa/scenarios/channels/whatsapp-participant-identity-inspection.yaml @@ -0,0 +1,149 @@ +title: WhatsApp DM and group participant identity inspection + +scenario: + id: whatsapp-participant-identity-inspection + surface: channels + coverage: + primary: + - gateway.identity-and-presence-apis + secondary: + - whatsapp.outbound-text-sends + objective: Verify one real WhatsApp participant keeps the same pseudonym across admitted DM and group turns, with distinct executions and exact audit inspection across restart. + gatewayConfigPatch: + logging: + audit: + executionIdentity: true + successCriteria: + - A leased driver account receives a marker reply through both the DM and mentioned group paths. + - Each turn resolves to one new run and exact execution with a present person invoker and channel ingress. + - The participant pseudonym stays equal across routes; phone numbers, native JIDs, and logical rooms are absent from inspection. + - Generic decision facts remain unverified, and exact JSON plus human inspection survives Gateway restart. + docsRefs: + - docs/gateway/audit.md + - docs/cli/audit.md + - docs/concepts/qa-e2e-automation/whatsapp-and-credentials.md + codeRefs: + - extensions/qa-lab/src/live-transports/whatsapp/adapter.runtime.ts + - extensions/whatsapp/src/inbound/access-control.ts + execution: + kind: flow + channel: whatsapp + suiteIsolation: isolated + isolationReason: Enables identity before admission and inspects the same isolated audit database across a Gateway replacement restart. + timeoutMs: 300000 + retryCount: 0 + summary: Run with the live WhatsApp driver and mock-openai provider. Requires two dedicated linked Web sessions, their distinct E.164 numbers, and a dedicated groupJid containing both accounts. No live model is needed. + config: + requiredChannelDriver: live + requiredProviderMode: mock-openai + turns: + - conversation: { id: qa-whatsapp-identity-dm, kind: direct } + prefix: "" + marker: QA-WHATSAPP-IDENTITY-DM-OK + - conversation: { id: qa-whatsapp-identity-group, kind: group } + prefix: "openclawqa " + marker: QA-WHATSAPP-IDENTITY-GROUP-OK + +flow: + steps: + - name: inspects one WhatsApp person across DM and group executions + actions: + - assert: + expr: "transport.id === 'whatsapp' && env.providerMode === config.requiredProviderMode" + message: WhatsApp participant identity proof requires the live WhatsApp adapter and mock-openai provider + - assert: + expr: "typeof whatsappScenarioContext.runtimeEnv.groupJid === 'string' && whatsappScenarioContext.runtimeEnv.groupJid.length > 0" + message: WhatsApp participant identity proof requires groupJid in the credential payload + - call: waitForGatewayHealthy + args: [{ ref: env }, 60000] + - call: waitForTransportReady + args: [{ ref: env }, 60000] + - resetTransport: true + - set: proofs + value: [] + - set: privateReferences + value: + expr: "[whatsappScenarioContext.runtimeEnv.driverPhoneE164, whatsappScenarioContext.runtimeEnv.sutPhoneE164, whatsappScenarioContext.runtimeEnv.groupJid, ...config.turns.map((turn) => turn.conversation.id)].flatMap((value) => [value, value.replace(/^\\+/, '')])" + - forEach: + items: { ref: config.turns } + item: turn + actions: + - set: runsBefore + value: + expr: "[...new Set((await runQaCli(env, ['audit', '--kind', 'agent_run', '--limit', '500', '--json'], { timeoutMs: 60000, json: true })).events.map((event) => event.runId).filter((id) => typeof id === 'string'))]" + - sendInbound: + conversation: { ref: turn.conversation } + senderId: qa-whatsapp-identity-driver + senderName: QA WhatsApp Driver + text: { expr: "`${turn.prefix}Reply exactly: ${turn.marker}`" } + - waitForOutbound: + conversation: { ref: turn.conversation } + textIncludes: { ref: turn.marker } + timeoutMs: 60000 + - call: waitForCondition + saveAs: newRunIds + args: + - lambda: + async: true + expr: "(ids => ids.length ? ids : undefined)([...new Set((await runQaCli(env, ['audit', '--kind', 'agent_run', '--limit', '500', '--json'], { timeoutMs: 60000, json: true })).events.map((event) => event.runId).filter((id) => typeof id === 'string' && !runsBefore.includes(id)))])" + - 60000 + - 250 + - assert: + expr: "newRunIds.length === 1" + message: WhatsApp turn must produce exactly one newly admitted run + - call: waitForCondition + saveAs: runInspection + args: + - lambda: + async: true + expr: "runQaCli(env, ['audit', '--run', newRunIds[0], '--explain', '--json'], { timeoutMs: 60000, json: true }).then((value) => value.identity?.state === 'present' && value.decisionDisplays.some((receipt) => receipt.action.family === 'decision') ? value : undefined)" + - 60000 + - 250 + - assert: + expr: "runInspection.identity.context.runId === newRunIds[0] && typeof runInspection.identity.context.executionId === 'string' && runInspection.identity.context.invoker.state === 'present' && runInspection.identity.context.invoker.principal?.kind === 'person' && typeof runInspection.identity.context.invoker.principal.principalRef === 'string' && runInspection.identity.context.ingress.kind === 'channel' && runInspection.identity.context.ingress.state === 'present'" + message: WhatsApp audit inspection must retain the admitted person and channel ingress for this run + - set: exactInspection + value: + expr: "await runQaCli(env, ['audit', '--execution', runInspection.identity.context.executionId, '--explain', '--json'], { timeoutMs: 60000, json: true })" + - set: humanInspection + value: + expr: "await runQaCli(env, ['audit', '--execution', runInspection.identity.context.executionId, '--explain'], { timeoutMs: 60000 })" + - assert: + expr: "exactInspection.identity.state === 'present' && JSON.stringify(exactInspection.identity.context) === JSON.stringify(runInspection.identity.context) && humanInspection.includes('Invoker [present]') && humanInspection.includes('Decisions')" + message: WhatsApp exact execution JSON and human inspection must agree with run discovery + - assert: + expr: "exactInspection.decisionDisplays.some((receipt) => receipt.action.family === 'decision' && receipt.provenance.state === 'unverified') && exactInspection.decisionDisplays.filter((receipt) => receipt.action.family === 'decision').every((receipt) => receipt.provenance.state === 'unverified') && privateReferences.every((value) => !JSON.stringify(exactInspection).includes(value) && !humanInspection.includes(value))" + message: WhatsApp inspection must exclude raw route and participant references and keep generic decisions unverified + - set: proofs + value: + expr: "[...proofs, exactInspection.identity.context]" + - assert: + expr: "proofs.length === 2 && new Set(proofs.map((proof) => proof.executionId)).size === 2 && new Set(proofs.map((proof) => proof.contextId)).size === 2 && new Set(proofs.map((proof) => proof.invoker.principal.principalRef)).size === 1" + message: WhatsApp DM and group must have distinct execution contexts for the same participant + detailsExpr: "proofs.map((proof) => `run=${proof.runId}; execution=${proof.executionId}`).join('; ')" + + - name: preserves exact WhatsApp identity across Gateway restart + actions: + - call: env.gateway.restartAfterStateMutation + args: + - lambda: + async: true + expr: Promise.resolve() + - call: waitForGatewayHealthy + args: [{ ref: env }, 60000] + - call: waitForTransportReady + args: [{ ref: env }, 60000] + - forEach: + items: { ref: proofs } + item: proof + actions: + - set: restartedInspection + value: + expr: "await runQaCli(env, ['audit', '--execution', proof.executionId, '--explain', '--json'], { timeoutMs: 60000, json: true })" + - set: restartedHuman + value: + expr: "await runQaCli(env, ['audit', '--execution', proof.executionId, '--explain'], { timeoutMs: 60000 })" + - assert: + expr: "restartedInspection.identity.state === 'present' && JSON.stringify(restartedInspection.identity.context) === JSON.stringify(proof) && restartedHuman.includes('Invoker [present]') && restartedHuman.includes('Decisions') && privateReferences.every((value) => !JSON.stringify(restartedInspection).includes(value) && !restartedHuman.includes(value))" + message: WhatsApp exact participant identity changed or exposed private references after restart + detailsExpr: "`restart-stable executions=${proofs.length}`" diff --git a/qa/scenarios/runtime/agent-run-identity-inspection.yaml b/qa/scenarios/runtime/agent-run-identity-inspection.yaml index 34c1518a1548..1b8989b29b33 100644 --- a/qa/scenarios/runtime/agent-run-identity-inspection.yaml +++ b/qa/scenarios/runtime/agent-run-identity-inspection.yaml @@ -10,6 +10,7 @@ scenario: successCriteria: - A real local agent turn against the deterministic mock provider records one bounded execution identity context. - Fresh and existing-install restarts keep execution-identity storage absent before explicit opt-in, and global audit disable prevents new identity rows. + - Real config.patch enable, disable, and reenable operations retain the same Gateway boot and socket, change subsequent admissions, preserve old contexts, and never backfill disabled runs. - The audit CLI renders text and JSON projections for the exact run, including every identity field and the truthful non-enforcement admission receipt. - A replacement Gateway process returns byte-equivalent normalized identity context JSON for the same run. - Two real public-ingress turns that omit runId share the session correlation but retain distinct execution and context ids. @@ -32,7 +33,7 @@ scenario: kind: script parallelSafe: true path: test/e2e/qa-lab/runtime/agent-run-identity-inspection.ts - summary: Starts an ephemeral Gateway and mock provider, runs local and repeated public-ingress turns, proves run ambiguity plus exact text/JSON selection, replaces the Gateway process, and compares normalized context bytes. + summary: Requires a current private-QA dist build (OPENCLAW_BUILD_PRIVATE_QA=1 pnpm build). Starts an ephemeral Gateway and mock provider from that build, runs local and repeated public-ingress turns, proves run ambiguity plus exact text/JSON selection, replaces the Gateway process, and compares normalized context bytes. args: - --artifact-base - ${outputDir} diff --git a/qa/scenarios/runtime/autonomous-task-lifecycle-receipts.yaml b/qa/scenarios/runtime/autonomous-task-lifecycle-receipts.yaml index 10309a1952e6..b8e338f2efd0 100644 --- a/qa/scenarios/runtime/autonomous-task-lifecycle-receipts.yaml +++ b/qa/scenarios/runtime/autonomous-task-lifecycle-receipts.yaml @@ -10,6 +10,8 @@ scenario: objective: Verify exact-bound cron, task, and flow owner lifecycle rows are explained without generic decision-fact copies. successCriteria: - A mapped hook transform suppression returns HTTP 204 before admission and creates no execution identity. + - Admitted mapped webhook POSTs before and after Gateway replacement retain distinct immutable contexts with one stable pseudonymized ingress source, absent invoker evidence, and unattributed admission coverage. + - Mapping IDs, request IDs, and webhook body sentinels stay absent from persisted contexts, JSON and human inspection, and audit-specific Gateway logs before and after replacement. - A real forced scheduled agent run binds one exact context/execution to its cron receipt and task row. - A real Gateway agent action binds one exact context/execution to its CLI task row; focused owner tests cover bound flow rows. - JSON and human audit inspection expose only bounded attribution-only lifecycle displays with closed producers, accept the cron cursor prefix, and omit task/prompt content. @@ -27,10 +29,11 @@ scenario: - src/tasks/task-flow-registry.store.sqlite.ts - src/gateway/hooks-mapping.ts - test/e2e/qa-lab/runtime/autonomous-task-lifecycle-receipts.ts + - test/e2e/qa-lab/runtime/autonomous-task-lifecycle-receipts.fixtures.ts execution: kind: script path: test/e2e/qa-lab/runtime/autonomous-task-lifecycle-receipts.ts - summary: Starts an ephemeral Gateway and mock provider, proves pre-admission suppression, runs cron and CLI task lifecycles, replaces the Gateway, and compares inspection output. + summary: Requires a current private-QA dist build (OPENCLAW_BUILD_PRIVATE_QA=1 pnpm build). Starts an ephemeral Gateway and mock provider from that build, proves pre-admission suppression and admitted mapped webhook identity/privacy, runs cron and CLI task lifecycles, replaces the Gateway, and compares persisted contexts and inspection output. args: - --artifact-base - ${outputDir} diff --git a/qa/scenarios/runtime/live-frontier-execution-identity.yaml b/qa/scenarios/runtime/live-frontier-execution-identity.yaml new file mode 100644 index 000000000000..7a28f159630c --- /dev/null +++ b/qa/scenarios/runtime/live-frontier-execution-identity.yaml @@ -0,0 +1,159 @@ +title: Live frontier execution identity inspection + +scenario: + id: live-frontier-execution-identity + surface: gateway + coverage: + secondary: + - gateway.identity-and-presence-apis + objective: Inspect distinct admitted executions from two real frontier turns in one session, with exact CLI selection and immutable identity after restart. + gatewayConfigPatch: + logging: + audit: + enabled: true + executionIdentity: true + successCriteria: + - Only the live-frontier provider lane is eligible; the selected model comes from operator input or the existing live defaults. + - Each completed turn has a matching assistant reply from the selected provider and model, plus a distinct execution and context id. + - Run discovery, exact RPC inspection, and human and JSON CLI inspection agree without claiming admission authorization or retaining prompt content. + - Exact context bytes survive a lifecycle-owned Gateway replacement. + docsRefs: + - docs/gateway/audit.md + - docs/cli/audit.md + codeRefs: + - extensions/qa-lab/src/suite-runtime-agent-process.ts + - src/gateway/server-methods/audit.ts + - src/commands/audit.ts + execution: + kind: flow + channel: qa-channel + providerMode: live-frontier + suiteIsolation: isolated + isolationReason: Enables identity collection in isolated state and replaces its Gateway before exact readback. + retryCount: 0 + summary: Run `pnpm openclaw qa suite --provider-mode live-frontier --scenario live-frontier-execution-identity`; needs authorized provider auth and the source-owned live model selection. Does not require external channel credentials. + config: + requiredProviderMode: live-frontier + +flow: + steps: + - name: inspects two real frontier executions + actions: + - assert: + expr: "env.providerMode === config.requiredProviderMode" + message: execution identity qualification requires live-frontier + - set: selected + value: { expr: "splitModelRef(env.primaryModel)" } + - assert: + expr: "selected?.provider && selected?.model && !['mock-openai', 'aimock'].includes(selected.provider)" + message: execution identity qualification requires a real provider and model + - call: waitForGatewayHealthy + args: [{ ref: env }, 60000] + - set: sessionKey + value: { expr: "`agent:qa:live-identity-${randomUUID()}`" } + - set: inspections + value: [] + - forEach: + items: [first, second] + item: turn + actions: + - set: marker + value: { expr: "`QA-LIVE-IDENTITY-${turn}-${randomUUID()}`" } + - set: started + value: + expr: "await env.gateway.call('agent', { agentId: 'qa', sessionKey, message: `Reply exactly: ${marker}`, deliver: false, provider: selected.provider, model: selected.model, idempotencyKey: randomUUID() })" + - assert: + expr: "started.status === 'accepted' && typeof started.runId === 'string' && started.runId.length > 0" + message: live agent turn was not admitted + - call: waitForAgentRun + saveAs: terminal + args: + [ + { ref: env }, + { expr: "started.runId" }, + { expr: "liveTurnTimeoutMs(env, 60000)" }, + ] + - assert: + expr: "['ok', 'completed', 'succeeded'].includes(terminal.status)" + message: live agent turn did not complete successfully + - call: waitForAgentHistoryReply + args: + - ref: env + - ref: sessionKey + - lambda: + params: [text] + expr: "text.trim() === marker" + - expr: liveTurnTimeoutMs(env, 60000) + - set: history + value: { expr: "await env.gateway.call('chat.history', { sessionKey, limit: 12 })" } + - set: assistant + value: + { + expr: "history.messages.filter((message) => message.role === 'assistant').at(-1)", + } + - assert: + expr: "assistant?.provider === selected.provider && assistant?.model === selected.model && assistant?.stopReason !== 'error' && assistant?.content?.some((part) => part.type === 'text' && part.text.trim() === marker)" + message: completed reply did not come from the selected live provider and model + - call: waitForCondition + saveAs: inspection + args: + - lambda: + async: true + expr: "env.gateway.call('audit.run.inspect', { runId: started.runId }).then((value) => value.identity.state === 'present' ? value : undefined)" + - 60000 + - 250 + - set: context + value: { expr: "inspection.identity.context" } + - assert: + expr: "inspection.run.runId === started.runId && inspection.run.executionId === context.executionId && context.runId === started.runId && context.contextId && context.executionId && context.ingress.state === 'present' && context.ingress.kind === 'gateway-client' && context.ingress.boundary === 'gateway.ws.authenticated-connect' && context.invoker.state === 'absent' && context.coverageState === 'unattributed'" + message: live execution inspection lost its exact authenticated ingress or fabricated an invoker + - assert: + expr: "context.agentPrincipal.kind === 'agent' && context.agentPrincipal.principalRef === 'qa' && context.agentDefinition.definitionRef === 'qa' && context.trustDomain.state === 'present' && context.runtimeInstance.state === 'present' && ['gateway', 'embedded', 'plugin-harness'].includes(context.runtimeInstance.kind) && [context.trustDomain.domainRef, context.runtimeInstance.runtimeRef].every((ref) => /^hmac-sha256:v1:[a-f0-9]{32}:[a-f0-9]{64}$/.test(ref)) && JSON.stringify(context).length <= 16384" + message: live execution lost bounded agent or pseudonymized runtime identity + - assert: + expr: "inspection.decisionDisplays.some((display) => display.provenance.state === 'verified' && display.provenance.producer === 'run-admission' && display.decision.outcome === 'not-applicable' && display.decision.reasonCode === 'run_admission_identity_not_evaluated') && !Object.hasOwn(inspection, 'decisions') && !JSON.stringify(inspection).includes(marker)" + message: live inspection overstated admission authority or exposed private content + - set: exact + value: + { + expr: "await runQaCli(env, ['audit', '--execution', context.executionId, '--explain', '--json'], { timeoutMs: 60000, json: true })", + } + - assert: + expr: "exact.run.executionId === context.executionId && JSON.stringify(exact.identity.context) === JSON.stringify(context)" + message: exact CLI inspection selected a different live execution + - set: human + value: + { + expr: "await runQaCli(env, ['audit', '--execution', context.executionId, '--explain'], { timeoutMs: 60000 })", + } + - assert: + expr: "['Identity', 'Invoker [absent]', 'Decisions', 'run_admission_identity_not_evaluated', 'not-applicable'].every((label) => human.includes(label)) && !human.includes(marker)" + message: human live inspection omitted non-enforcement evidence or leaked the prompt + - set: inspections + value: { expr: "[...inspections, context]" } + - assert: + expr: "inspections.length === 2 && new Set(inspections.map((context) => context.executionId)).size === 2 && new Set(inspections.map((context) => context.contextId)).size === 2" + message: repeated live turns reused one execution identity + detailsExpr: "`live turns=2; distinct executions=${inspections.length}; exact RPC and CLI identity inspected`" + - name: retains live execution identity across Gateway replacement + actions: + - call: env.gateway.restartAfterStateMutation + args: + - lambda: + async: true + expr: "Promise.resolve()" + - call: waitForGatewayHealthy + args: [{ ref: env }, 60000] + - forEach: + items: { ref: inspections } + item: before + actions: + - set: after + value: + { + expr: "await env.gateway.call('audit.run.inspect', { executionId: before.executionId })", + } + - assert: + expr: "after.run.executionId === before.executionId && after.identity.state === 'present' && JSON.stringify(after.identity.context) === JSON.stringify(before)" + message: live execution identity changed across Gateway replacement + detailsExpr: "`restart-stable exact executions=${inspections.length}`" diff --git a/qa/scenarios/runtime/restart-recovery-memory-flush-lifecycle-proof.yaml b/qa/scenarios/runtime/restart-recovery-memory-flush-lifecycle-proof.yaml index d20a927e434f..6ce714897774 100644 --- a/qa/scenarios/runtime/restart-recovery-memory-flush-lifecycle-proof.yaml +++ b/qa/scenarios/runtime/restart-recovery-memory-flush-lifecycle-proof.yaml @@ -12,7 +12,7 @@ scenario: objective: Prove a forced pre-admission memory flush cannot own the parent session lifecycle and the channel turn still receives a reply. successCriteria: - A synthetic channel turn creates the parent session. - - The next turn forces a pre-admission memory flush through the real Gateway child and mock provider. + - The next turn delivers its foreground reply, then records the forced memory flush through the real Gateway child and mock provider. - The parent session remains terminally healthy and the second turn receives a visible reply. gatewayConfigPatch: agents: @@ -31,7 +31,7 @@ scenario: execution: kind: flow runtime: openclaw - summary: Force a memory-flush child before a synthetic channel turn, then verify parent lifecycle and delivery. + summary: Force post-delivery memory maintenance for a synthetic channel turn, then verify its eventual fact, parent lifecycle, and delivery. channel: qa-channel config: conversationPrefix: restart-recovery-memory-flush @@ -103,17 +103,17 @@ flow: timeoutMs: expr: liveTurnTimeoutMs(env, 60000) saveAs: proofReply - - call: readRawQaSessionStore - saveAs: store + - call: waitForCondition + saveAs: sessionEntry args: - - ref: env - - set: sessionEntry - value: - expr: "store[sessionKey]" + - lambda: + async: true + expr: "readRawQaSessionStore(env).then((store) => { const entry = store[sessionKey]; return entry?.memoryFlush?.kind === 'succeeded' ? entry : undefined; })" + - 60000 + - 100 - assert: expr: "sessionEntry?.memoryFlush?.kind === 'succeeded'" - message: - expr: "`forced memory flush did not succeed: ${JSON.stringify({ sessionKey, storeKeys: Object.keys(store), sessionEntry })}`" + message: forced post-delivery memory flush did not succeed - assert: expr: "sessionEntry?.status === 'done' && sessionEntry?.abortedLastRun !== true" message: diff --git a/scripts/e2e/codex-npm-plugin-live-docker.sh b/scripts/e2e/codex-npm-plugin-live-docker.sh index ed2cc654aef5..f33702c3086b 100644 --- a/scripts/e2e/codex-npm-plugin-live-docker.sh +++ b/scripts/e2e/codex-npm-plugin-live-docker.sh @@ -22,6 +22,11 @@ HOST_BUILD="${OPENCLAW_CODEX_NPM_PLUGIN_HOST_BUILD:-1}" PACKAGE_TGZ="${OPENCLAW_CURRENT_PACKAGE_TGZ:-}" PROFILE_FILE="${OPENCLAW_CODEX_NPM_PLUGIN_PROFILE_FILE:-${OPENCLAW_TESTBOX_PROFILE_FILE:-$HOME/.openclaw-testbox-live.profile}}" CODEX_PLUGIN_SPEC="${OPENCLAW_CODEX_NPM_PLUGIN_SPEC:-}" +AUDIT_IDENTITY="${OPENCLAW_CODEX_NPM_PLUGIN_AUDIT_IDENTITY:-0}" +case "$AUDIT_IDENTITY" in + 0|1) ;; + *) echo "OPENCLAW_CODEX_NPM_PLUGIN_AUDIT_IDENTITY must be 0 or 1" >&2; exit 1 ;; +esac CODEX_PLUGIN_MOUNT=() CODEX_PLUGIN_PACK_DIR="" CODEX_PLUGIN_REGISTRY_PACKAGE="" @@ -214,6 +219,7 @@ if ! docker_e2e_run_with_harness \ -e OPENCLAW_CODEX_NPM_PLUGIN_FORCE_UNSAFE_INSTALL="${OPENCLAW_CODEX_NPM_PLUGIN_FORCE_UNSAFE_INSTALL:-1}" \ -e OPENCLAW_CODEX_NPM_PLUGIN_MODEL="${OPENCLAW_CODEX_NPM_PLUGIN_MODEL:-openai/gpt-5.4}" \ -e OPENCLAW_CODEX_NPM_PLUGIN_SPEC="$CODEX_PLUGIN_SPEC" \ + -e OPENCLAW_CODEX_NPM_PLUGIN_AUDIT_IDENTITY="$AUDIT_IDENTITY" \ -e OPENCLAW_CODEX_NPM_PLUGIN_REGISTRY_PACKAGE="$CODEX_PLUGIN_REGISTRY_PACKAGE" \ -e OPENCLAW_CODEX_NPM_PLUGIN_REGISTRY_TARBALL="$CODEX_PLUGIN_REGISTRY_TARBALL" \ -e OPENCLAW_CODEX_NPM_PLUGIN_REGISTRY_VERSION="$CODEX_PLUGIN_REGISTRY_VERSION" \ @@ -295,6 +301,7 @@ dump_debug_logs() { /tmp/openclaw-codex-agent-turn1.err \ /tmp/openclaw-codex-agent-turn2.json \ /tmp/openclaw-codex-agent-turn2.err \ + /tmp/openclaw-codex-audit-gateway.log \ /tmp/openclaw-codex-followthrough.json \ /tmp/openclaw-codex-followthrough.log \ /tmp/openclaw-codex-followthrough.err \ @@ -305,12 +312,14 @@ dump_debug_logs() { } registry_pid="" +audit_gateway_pid="" debug_logs_dumped=0 cleanup_scenario() { local status=$? trap - EXIT set +e openclaw_e2e_stop_process "${registry_pid:-}" + openclaw_e2e_stop_process "${audit_gateway_pid:-}" if [ "$status" -ne 0 ] && [ "$debug_logs_dumped" -eq 0 ]; then dump_debug_logs "$status" fi @@ -456,6 +465,17 @@ run_agent_turn \ node scripts/e2e/lib/codex-npm-plugin-live/assertions.mjs assert-agent-turn "$SUCCESS_MARKER" "$SESSION_ID" "$MODEL_REF" +if [ "${OPENCLAW_CODEX_NPM_PLUGIN_AUDIT_IDENTITY:-0}" = "1" ]; then + echo "Inspecting persisted Codex execution identity through the installed package Gateway..." + audit_package_root="$(openclaw_e2e_package_root "$NPM_CONFIG_PREFIX")" + audit_package_entry="$(openclaw_e2e_package_entrypoint "$audit_package_root")" + audit_gateway_pid="$(openclaw_e2e_start_gateway "$audit_package_entry" 18789 /tmp/openclaw-codex-audit-gateway.log)" + openclaw_e2e_wait_gateway_ready "$audit_gateway_pid" /tmp/openclaw-codex-audit-gateway.log + node scripts/e2e/lib/codex-npm-plugin-live/assertions.mjs assert-audit "$SUCCESS_MARKER" + openclaw_e2e_stop_process "$audit_gateway_pid" + audit_gateway_pid="" +fi + FOLLOWTHROUGH_SESSION_ID="${SESSION_ID}-followthrough" FOLLOWTHROUGH_PROGRESS_MARKER="${SUCCESS_MARKER}-FOLLOWTHROUGH-PROGRESS" FOLLOWTHROUGH_COMPLETE_MARKER="${SUCCESS_MARKER}-FOLLOWTHROUGH-COMPLETE" diff --git a/scripts/e2e/lib/codex-npm-plugin-live/assertions.mjs b/scripts/e2e/lib/codex-npm-plugin-live/assertions.mjs index 1d6263f06bb8..887ba5f94166 100644 --- a/scripts/e2e/lib/codex-npm-plugin-live/assertions.mjs +++ b/scripts/e2e/lib/codex-npm-plugin-live/assertions.mjs @@ -1,5 +1,6 @@ // Assertions for Codex npm plugin live E2E scenarios. -import { createHash } from "node:crypto"; +import { execFileSync } from "node:child_process"; +import { createHash, randomUUID } from "node:crypto"; import fs from "node:fs"; import path from "node:path"; import { DatabaseSync } from "node:sqlite"; @@ -21,6 +22,7 @@ import { stateDir, } from "../codex-install-utils.mjs"; import { assertCodexReleasePackageContract } from "../codex-release-package-assertions.mjs"; +import { inspectCodexAudit } from "./audit-inspection.mjs"; const command = process.argv[2]; const allowBetaCompatDiagnostics = @@ -289,6 +291,19 @@ function configure() { const state = stateDir(); const cfgPath = configPath(); const cfg = fs.existsSync(cfgPath) ? readJson(cfgPath) : {}; + if (process.env.OPENCLAW_CODEX_NPM_PLUGIN_AUDIT_IDENTITY === "1") { + cfg.logging = { + ...cfg.logging, + audit: { ...cfg.logging?.audit, enabled: true, executionIdentity: true }, + }; + cfg.gateway = { + ...cfg.gateway, + mode: "local", + bind: "loopback", + port: 18789, + auth: { mode: "token", token: randomUUID() }, + }; + } cfg.plugins = { ...cfg.plugins, enabled: true, @@ -819,6 +834,47 @@ function assertAgentTurn() { }); } +function assertAudit() { + const marker = process.argv[3]; + if (!marker) { + throw new Error("assert-audit requires a private reply marker"); + } + const cliPath = process.env.OPENCLAW_E2E_CLI_BIN; + if (!cliPath) { + throw new Error("assert-audit requires the installed OPENCLAW_E2E_CLI_BIN"); + } + // CLI runs mint independent run ids. This fresh state contains exactly the three + // completed local turns; select retained identity keys, never session/route guesses. + const database = new DatabaseSync(path.join(stateDir(), "state", "openclaw.sqlite"), { + readOnly: true, + }); + let selectors; + try { + selectors = database + .prepare( + "SELECT run_id AS runId, execution_id AS executionId, context_id AS contextId FROM execution_identity_contexts ORDER BY created_at, execution_id LIMIT 4", + ) + .all(); + } finally { + database.close(); + } + const result = inspectCodexAudit({ + selectors, + expectedExecutions: 3, + privateValues: [marker], + query: (args) => + JSON.parse( + execFileSync(process.execPath, [cliPath, "audit", ...args, "--json"], { + encoding: "utf8", + timeout: 120_000, + maxBuffer: MAX_TEXT_FILE_BYTES, + stdio: ["ignore", "pipe", "pipe"], + }), + ), + }); + console.log(`codex_audit_identity: ${JSON.stringify(result)}`); +} + function assertFollowthrough() { const progressMarker = process.argv[3]; const completeMarker = process.argv[4]; @@ -929,6 +985,7 @@ const commands = { "print-codex-bin": printCodexBin, "assert-preflight": assertPreflight, "assert-agent-turn": assertAgentTurn, + "assert-audit": assertAudit, "assert-followthrough": assertFollowthrough, "assert-uninstalled": assertUninstalled, "assert-agent-error": assertAgentError, diff --git a/scripts/e2e/lib/codex-npm-plugin-live/audit-inspection.mjs b/scripts/e2e/lib/codex-npm-plugin-live/audit-inspection.mjs new file mode 100644 index 000000000000..9df83ca5e113 --- /dev/null +++ b/scripts/e2e/lib/codex-npm-plugin-live/audit-inspection.mjs @@ -0,0 +1,104 @@ +// Inspect the package Gateway's public audit surface after direct-local Codex turns. +import assert from "node:assert/strict"; + +const HMAC_REF = /^hmac-sha256:v1:[a-f0-9]{32}:[a-f0-9]{64}$/u; + +export function inspectCodexAudit({ query, selectors, expectedExecutions, privateValues }) { + const inspect = (...args) => { + const result = query(args); + const encoded = JSON.stringify(result); + for (const value of privateValues) { + assert(!encoded.includes(value), "Codex audit exposed private workspace content"); + } + assert(!Object.hasOwn(result, "decisions"), "Codex audit exposed raw decision receipts"); + return result; + }; + assert.equal(selectors.length, expectedExecutions, "Codex turns did not retain three identities"); + const executionIds = new Set(); + const contextIds = new Set(); + for (const candidate of selectors) { + const { runId } = candidate; + assert.equal(typeof runId, "string"); + assert(runId.length > 0); + assert.equal(typeof candidate.executionId, "string"); + assert(candidate.executionId.length > 0); + const discovery = inspect("--run", runId, "--explain", "--limit", "50"); + assert.equal(discovery.run?.runId, runId, "Codex audit discovered the wrong run"); + assert.equal(discovery.run?.status, "known", "Codex audit run was not retained"); + assert.equal( + discovery.identity?.state, + "present", + "Codex run discovery did not select its execution", + ); + assert.equal(discovery.identity.context.executionId, candidate.executionId); + assert.equal(discovery.identity.context.contextId, candidate.contextId); + assert.equal( + discovery.nextExecutionCursor, + undefined, + "Codex execution discovery was truncated", + ); + const exact = inspect("--execution", candidate.executionId, "--explain", "--limit", "100"); + assert.equal(exact.schemaVersion, 1); + assert.equal(exact.run?.status, "known"); + assert.equal(exact.run?.runId, runId); + assert.equal(exact.run?.executionId, candidate.executionId); + assert.equal(exact.identity?.state, "present", "Codex execution identity is missing"); + const context = exact.identity.context; + assert.equal(context.runId, runId); + assert.equal(context.executionId, candidate.executionId); + assert.equal(context.contextId, candidate.contextId); + assert.equal(typeof context.contextId, "string"); + assert(context.contextId.length > 0); + assert.deepEqual(context.invoker, { state: "absent" }, "local CLI must not invent a person"); + assert.equal(context.ingress?.kind, "local-cli"); + assert.equal(context.ingress?.state, "present"); + assert.equal(context.agentPrincipal?.kind, "agent"); + assert.equal(context.agentPrincipal?.principalRef, "main"); + assert.equal(context.trustDomain?.state, "present"); + assert.equal(context.trustDomain?.kind, "gateway-cell"); + assert.match(context.trustDomain.domainRef, HMAC_REF); + assert.equal(context.agentPrincipal.domainRef, context.trustDomain.domainRef); + assert.equal(context.agentDefinition?.definitionRef, "main"); + assert.equal(context.runtimeInstance?.kind, "plugin-harness"); + assert.equal(context.runtimeInstance?.state, "present"); + assert.match(context.runtimeInstance.runtimeRef, HMAC_REF); + assert.equal(context.coverageState, "unattributed"); + assert.deepEqual(context.applicableGrants, []); + assert( + context.assurance?.some( + (evidence) => + evidence.kind === "runtime-binding" && evidence.strength === "boundary-verified", + ), + "Codex runtime binding assurance is missing", + ); + for (const evidence of context.assurance) { + assert.match(evidence.evidenceRef, HMAC_REF); + } + const admissions = exact.decisionDisplays?.filter( + (receipt) => + receipt.provenance?.state === "verified" && receipt.provenance.producer === "run-admission", + ); + assert.equal(admissions?.length, 1, "Codex admission receipt is missing or duplicated"); + const admission = admissions[0]; + assert.deepEqual(admission.action.family, "run"); + assert.deepEqual(admission.action.operation, "admission"); + assert.equal(admission.decision?.outcome, "not-applicable"); + assert.equal(admission.decision?.reasonCode, "run_admission_identity_not_evaluated"); + assert.equal(admission.enforcement?.coverageState, "unattributed"); + assert.equal(admission.enforcement?.grantCount, 0); + assert.equal(admission.enforcement?.policyCount, 0); + assert.equal(exact.nextDecisionCursor, undefined, "Codex decision inspection was truncated"); + assert( + exact.decisionDisplays.every( + (receipt) => receipt.provenance?.producer !== "operator-approval", + ), + "full-access Codex execution must not manufacture operator approval", + ); + executionIds.add(context.executionId); + contextIds.add(context.contextId); + } + assert.equal(executionIds.size, expectedExecutions, "Codex turns reused an execution id"); + assert.equal(contextIds.size, expectedExecutions, "Codex turns reused a context id"); + + return { executionCount: executionIds.size }; +} diff --git a/scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs b/scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs new file mode 100644 index 000000000000..d62ae08e925d --- /dev/null +++ b/scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs @@ -0,0 +1,156 @@ +// Installed-package identity proof; reads only the scenario's isolated audit state. +import assert from "node:assert/strict"; +import fs from "node:fs"; +import path from "node:path"; +import { DatabaseSync } from "node:sqlite"; +import { fileURLToPath } from "node:url"; + +const PSEUDONYM = /^hmac-sha256:v1:[a-f0-9]{32}:[a-f0-9]{64}$/u; + +export function readIdentityRows(stateDir) { + const file = path.join(stateDir, "state", "openclaw.sqlite"); + if (!fs.existsSync(file)) { + return []; + } + const db = new DatabaseSync(file, { readOnly: true }); + try { + if ( + !db + .prepare("SELECT 1 FROM sqlite_schema WHERE name = ? AND type = 'table'") + .get("execution_identity_contexts") + ) { + return []; + } + // Two rows suffice to reject cross-test contamination or duplicate admission. + return db + .prepare("SELECT context_json FROM execution_identity_contexts LIMIT 2") + .all() + .map((row) => row.context_json); + } finally { + db.close(); + } +} + +function assertIdentityPrivacy(text, needles) { + for (const needle of needles) { + if (needle && text.includes(needle)) { + throw new Error("execution identity exposed private fixture data"); + } + } +} + +export function assertIdentityProjection(result, persisted, needles) { + assertIdentityPrivacy(JSON.stringify(result), needles); + assertIdentityPrivacy(persisted, needles); + const persistedContext = JSON.parse(persisted); + const expectedRunId = persistedContext.runId; + assert.ok( + typeof expectedRunId === "string" && expectedRunId.length > 0, + "missing persisted run id", + ); + assert.equal(result.identity?.state, "present", "installed CLI omitted execution identity"); + assert.equal(result.run?.runId, expectedRunId, "installed CLI selected another run"); + assert.equal(Object.hasOwn(result, "decisions"), false, "private receipts reached the CLI"); + const context = result.identity.context; + assert.equal(context.runId, expectedRunId); + assert.equal(context.schemaVersion, 1); + for (const id of [context.contextId, context.executionId]) { + assert.ok(typeof id === "string" && id.length > 0, "missing opaque identity id"); + } + assert.equal(result.run.executionId, context.executionId); + assert.deepEqual(context.ingress, { + kind: "local-cli", + boundary: "agent-command.local", + state: "present", + }); + assert.deepEqual(context.invoker, { state: "absent" }); + assert.equal(context.coverageState, "unattributed"); + assert.equal(context.agentPrincipal?.principalRef, "main"); + assert.equal(context.agentDefinition?.definitionRef, "main"); + assert.equal(context.representedSubject, undefined); + assert.equal(context.sponsor, undefined); + assert.equal(context.lineage, undefined); + assert.match(context.trustDomain?.domainRef, PSEUDONYM); + assert.equal(context.agentPrincipal.domainRef, context.trustDomain.domainRef); + assert.match(context.runtimeInstance?.runtimeRef, PSEUDONYM); + for (const item of context.assurance) { + assert.match(item.evidenceRef, PSEUDONYM); + } + for (const grant of context.applicableGrants) { + assert.match(grant.grantRef, PSEUDONYM); + } + const admission = result.decisionDisplays?.find( + (display) => + display.provenance?.state === "verified" && display.provenance.producer === "run-admission", + ); + assert.equal(admission?.decision?.outcome, "not-applicable"); + assert.equal(admission?.decision?.reasonCode, "run_admission_identity_not_evaluated"); + assert.notEqual(admission?.enforcement?.coverageState, "enforced"); + assert.equal(JSON.stringify(context), persisted, "CLI context differs from persisted bytes"); + return context.executionId; +} + +function isolatedStateDir() { + const home = process.env.OPENCLAW_TEST_STATE_HOME; + assert.ok(home, "missing isolated test HOME"); + assert.equal(process.env.HOME, home); + assert.equal(process.env.OPENCLAW_HOME, home); + const stateDir = path.join(home, ".openclaw"); + assert.equal(process.env.OPENCLAW_STATE_DIR, stateDir); + assert.equal(process.env.OPENCLAW_CONFIG_PATH, path.join(stateDir, "openclaw.json")); + return stateDir; +} + +function main() { + const [command, file, beforeFile] = process.argv.slice(2); + const stateDir = isolatedStateDir(); + if (command === "clean-home") { + assert.deepEqual(fs.readdirSync(stateDir), [], "package proof inherited existing state"); + return; + } + const rows = readIdentityRows(stateDir); + if (command === "empty") { + assert.equal(rows.length, 0, "identity state existed before the opted-in turn"); + return; + } + if (command === "run-id") { + assert.equal(rows.length, 1, "expected exactly one admitted execution identity"); + const runId = JSON.parse(rows[0]).runId; + assert.ok(typeof runId === "string" && runId.length > 0, "missing persisted run id"); + process.stdout.write(runId); + return; + } + assert.equal(command, "verify"); + assert.equal(rows.length, 1, "expected exactly one admitted execution identity"); + const cfg = JSON.parse(fs.readFileSync(process.env.OPENCLAW_CONFIG_PATH, "utf8")); + const channel = cfg.channels?.[process.env.OPENCLAW_NPM_ONBOARD_CHANNEL]; + const needles = [ + process.env.HOME, + stateDir, + process.env.OPENCLAW_TEST_WORKSPACE_DIR, + process.env.OPENAI_API_KEY, + process.env.OPENCLAW_GATEWAY_TOKEN, + channel?.token, + channel?.botToken, + channel?.appToken, + cfg.models?.providers?.openai?.baseUrl, + process.env.SUCCESS_MARKER, + "Return the success marker from the test server.", + ]; + const result = JSON.parse(fs.readFileSync(file, "utf8")); + const executionId = assertIdentityProjection(result, rows[0], needles); + if (beforeFile) { + const before = JSON.parse(fs.readFileSync(beforeFile, "utf8")); + assertIdentityProjection(before, rows[0], needles); + assert.equal( + JSON.stringify(result.identity.context), + JSON.stringify(before.identity.context), + "execution identity changed across Gateway restart", + ); + } + process.stdout.write(executionId); +} + +if (process.argv[1] && fileURLToPath(import.meta.url) === path.resolve(process.argv[1])) { + main(); +} diff --git a/scripts/e2e/npm-onboard-channel-agent-docker.sh b/scripts/e2e/npm-onboard-channel-agent-docker.sh index 59a1b76639c5..17e5b6e961de 100644 --- a/scripts/e2e/npm-onboard-channel-agent-docker.sh +++ b/scripts/e2e/npm-onboard-channel-agent-docker.sh @@ -17,6 +17,9 @@ TARGET_ROOT_DIR="$(cd "${OPENCLAW_DOCKER_E2E_REPO_ROOT:-$ROOT_DIR}" && pwd)" ONBOARD_ASSERTIONS="$(openclaw_resolve_frozen_target_file "$TARGET_ROOT_DIR" \ scripts/e2e/lib/npm-onboard-channel-agent/assertions.mjs \ "$ROOT_DIR/scripts/e2e/lib/npm-onboard-channel-agent/assertions.mjs")" +ONBOARD_IDENTITY_ASSERTIONS="$(openclaw_resolve_frozen_target_file "$TARGET_ROOT_DIR" \ + scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs \ + "$ROOT_DIR/scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs")" # The assertion and its config producer are one target-owned contract; mixing # generations can make a valid frozen package appear to change its default model. ONBOARD_MOCK_OPENAI_CONFIG="$(openclaw_resolve_frozen_target_file "$TARGET_ROOT_DIR" \ @@ -84,6 +87,7 @@ if ! docker_e2e_run_with_harness \ -e "OPENCLAW_NPM_ONBOARD_STATUS_TEXT_MAX_BYTES=$STATUS_TEXT_MAX_BYTES" \ -e "OPENCLAW_TEST_STATE_SCRIPT_B64=$OPENCLAW_TEST_STATE_SCRIPT_B64" \ -v "$ONBOARD_ASSERTIONS:/app/scripts/e2e/lib/npm-onboard-channel-agent/assertions.mjs:ro" \ + -v "$ONBOARD_IDENTITY_ASSERTIONS:/app/scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs:ro" \ -v "$ONBOARD_MOCK_OPENAI_CONFIG:/app/scripts/e2e/lib/fixtures/mock-openai-config.mjs:ro" \ "${DOCKER_E2E_PACKAGE_ARGS[@]}" \ -i "$IMAGE_NAME" bash -s >"$run_log" 2>&1 <<'EOF'; then @@ -92,6 +96,9 @@ set -Eeuo pipefail source scripts/lib/openclaw-e2e-instance.sh source scripts/e2e/lib/prepublish-plugin-registry.sh openclaw_e2e_eval_test_state_from_b64 "${OPENCLAW_TEST_STATE_SCRIPT_B64:?missing OPENCLAW_TEST_STATE_SCRIPT_B64}" +export OPENCLAW_TEST_STATE_HOME +identity_assertions=scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs +node "$identity_assertions" clean-home export NPM_CONFIG_PREFIX="$HOME/.npm-global" export PATH="$NPM_CONFIG_PREFIX/bin:$PATH" export OPENAI_API_KEY="sk-openclaw-npm-onboard-e2e" @@ -106,6 +113,7 @@ MOCK_REQUEST_LOG="$scenario_tmp/mock-openai-requests.jsonl" export SUCCESS_MARKER MOCK_REQUEST_LOG mock_pid="" plugin_registry_pid="" +gateway_pid="" case "$CHANNEL" in telegram) @@ -134,6 +142,7 @@ case "$CHANNEL" in esac cleanup() { + openclaw_e2e_stop_process "${gateway_pid:-}" openclaw_e2e_stop_process "${mock_pid:-}" openclaw_e2e_stop_process "${plugin_registry_pid:-}" rm -rf "$scenario_tmp" @@ -157,6 +166,8 @@ dump_debug_logs() { /tmp/openclaw-agent.err \ /tmp/openclaw-agent.json \ /tmp/openclaw-mock-openai.log \ + "$scenario_tmp/gateway-before.log" \ + "$scenario_tmp/gateway-after.log" \ "$MOCK_REQUEST_LOG" \ "$OPENCLAW_HOME/.openclaw/openclaw.json" \ "$OPENCLAW_HOME/.openclaw/agents/main/agent/auth-profiles.json" @@ -254,6 +265,9 @@ fi node scripts/e2e/lib/npm-onboard-channel-agent/assertions.mjs configure-mock-model "$MOCK_PORT" node scripts/e2e/lib/npm-onboard-channel-agent/assertions.mjs assert-mock-model-config "$MOCK_PORT" +node "$identity_assertions" empty +openclaw config set logging.audit.enabled true +openclaw config set logging.audit.executionIdentity true echo "Running local agent turn against mocked OpenAI..." if openclaw agent --local \ @@ -272,6 +286,23 @@ if [ "$agent_status" -ne 0 ]; then fi node scripts/e2e/lib/npm-onboard-channel-agent/assertions.mjs assert-agent-turn "$SUCCESS_MARKER" "$MOCK_REQUEST_LOG" +run_id="$(node "$identity_assertions" run-id)" + +# The local CLI flushes its audit writer before exiting. Inspect once after that +# boundary; polling a read-only inspector would hide lost admission writes. +entry="$(openclaw_e2e_package_entrypoint "$package_root")" +export OPENCLAW_SKIP_CHANNELS=1 OPENCLAW_SKIP_GMAIL_WATCHER=1 OPENCLAW_SKIP_CANVAS_HOST=1 +gateway_pid="$(openclaw_e2e_start_gateway "$entry" "$PORT" "$scenario_tmp/gateway-before.log")" +openclaw_e2e_wait_gateway_ready "$gateway_pid" "$scenario_tmp/gateway-before.log" 300 "$PORT" +openclaw audit --run "$run_id" --explain --json >"$scenario_tmp/identity-before.json" +execution_id="$(node "$identity_assertions" verify "$scenario_tmp/identity-before.json")" +openclaw_e2e_stop_process "$gateway_pid" +gateway_pid="" +gateway_pid="$(openclaw_e2e_start_gateway "$entry" "$PORT" "$scenario_tmp/gateway-after.log")" +openclaw_e2e_wait_gateway_ready "$gateway_pid" "$scenario_tmp/gateway-after.log" 300 "$PORT" +openclaw audit --execution "$execution_id" --explain --json >"$scenario_tmp/identity-after.json" +node "$identity_assertions" verify "$scenario_tmp/identity-after.json" "$scenario_tmp/identity-before.json" >/dev/null +echo "Installed CLI execution identity survived Gateway restart with private fixture data omitted." echo "npm tarball onboard/channel/agent Docker E2E passed for $CHANNEL" EOF diff --git a/src/audit/execution-decision-cursors.test.ts b/src/audit/execution-decision-cursors.test.ts new file mode 100644 index 000000000000..0b689985f35f --- /dev/null +++ b/src/audit/execution-decision-cursors.test.ts @@ -0,0 +1,248 @@ +import type { DatabaseSync } from "node:sqlite"; +import { afterAll, beforeAll, describe, expect, it, vi } from "vitest"; +import { useAutoCleanupTempDirTracker } from "../../test/helpers/temp-dir.js"; +import { + closeOpenClawStateDatabaseAsync, + closeOpenClawStateDatabaseForTest, + openOpenClawStateDatabase, +} from "../state/openclaw-state-db.js"; +import { recordAuditEventInDatabase } from "./audit-event-store.js"; +import { recordExecutionDecisionFactInDatabase } from "./execution-decision-facts.js"; +import { receipt } from "./execution-decision-facts.test-support.js"; +import { ExecutionDecisionCursorError } from "./execution-decision-receipts.js"; +import { createExecutionIdentityAdmissionToken } from "./execution-identity-admission.js"; +import { inspectExecutionIdentityRun } from "./execution-identity-context.js"; +import { + prepareExecutionIdentityContextAtAdmission, + recordDeniedApprovalForRun, +} from "./execution-identity.test-support.js"; +import { bindExecutionOwnerLifecycleMetadata } from "./execution-owner-lifecycle-binding-store.js"; + +const tempDirs = useAutoCleanupTempDirTracker(afterAll); +const now = 200; +const stages = [ + { prefix: "a", owner: "operator_approvals" }, + { prefix: "m", owner: "audit_events" }, + { prefix: "g", owner: "tool-policy" }, + { prefix: "c", owner: "cron_run_receipts" }, + { prefix: "t", owner: "task_runs" }, + { prefix: "f", owner: "flow_runs" }, +] as const; +type Stage = (typeof stages)[number]["prefix"]; +let options: { env: { OPENCLAW_STATE_DIR: string } }; +let db: DatabaseSync; + +function inspect(decisionCursor: string, executionId = "execution-local") { + return inspectExecutionIdentityRun( + { executionId, decisionCursor, decisionLimit: 1 }, + { ...options, now }, + ); +} + +beforeAll(async () => { + vi.spyOn(Date, "now").mockReturnValue(now); + options = { env: { OPENCLAW_STATE_DIR: tempDirs.make("decision-cursors-") } }; + const database = openOpenClawStateDatabase(options); + db = database.db; + for (const scope of ["local", "foreign", "sibling"]) { + const context = prepareExecutionIdentityContextAtAdmission( + { + runId: scope === "sibling" ? "run-local" : `run-${scope}`, + agentId: "main", + ingress: { kind: "local-cli", boundary: "agent-command.local", state: "present" }, + runtime: { kind: "embedded" }, + }, + { + ...options, + contextId: `context-${scope}`, + executionId: `execution-${scope}`, + runtimeInstanceId: `runtime-${scope}`, + now: 50, + }, + ); + const token = createExecutionIdentityAdmissionToken(context.runId, { + contextId: context.contextId, + executionId: context.executionId, + now: context.createdAt, + }); + for (const index of [1, 2]) { + const id = `${scope}-${index}`; + await recordDeniedApprovalForRun(context.runId, options, `approval-${id}`, context); + expect( + recordAuditEventInDatabase( + { + sourceId: `message-${id}`, + sourceSequence: 1, + occurredAt: now, + kind: "message", + action: "message.outbound.finished", + status: "succeeded", + outcome: "sent", + actorType: "agent", + actorId: "main", + agentId: "main", + runId: context.runId, + executionIdentityToken: token, + direction: "outbound", + channel: "qa-channel", + conversationKind: "direct", + resultCount: 1, + }, + { ...options, database }, + ), + ).toBeDefined(); + expect( + recordExecutionDecisionFactInDatabase( + { + ...receipt(`generic-${id}`, now), + contextId: context.contextId, + executionId: context.executionId, + runId: context.runId, + }, + { ...options, database, now }, + ), + ).toBe("inserted"); + db.prepare(`INSERT INTO cron_run_receipts ( + receipt_id, store_key, job_id, config_revision, agent_id, request_run_id, + status, owner_pid, started_at_ms, finished_at_ms + ) VALUES (?, 'default', ?, 'revision-1', 'main', ?, 'ok', 1, ?, ?)`).run( + `cron-${id}`, + `job-${id}`, + context.runId, + now, + now, + ); + db.prepare(`INSERT INTO task_runs ( + task_id, runtime, owner_key, scope_kind, task, status, delivery_status, + notify_policy, created_at + ) VALUES (?, 'cron', 'fixture-owner', 'system', 'fixture task', 'succeeded', + 'not_applicable', 'silent', ?)`).run(`task-${id}`, now); + db.prepare(`INSERT INTO flow_runs ( + flow_id, owner_key, status, notify_policy, goal, created_at, updated_at + ) VALUES (?, 'fixture-owner', 'succeeded', 'silent', 'fixture goal', ?, ?)`).run( + `flow-${id}`, + now, + now, + ); + for (const ownerKind of ["cron", "task", "flow"] as const) { + expect( + bindExecutionOwnerLifecycleMetadata({ + db, + ownerKind, + ownerId: `${ownerKind}-${id}`, + binding: context, + }), + ).toBe("bound"); + } + } + } +}); + +afterAll(async () => { + await closeOpenClawStateDatabaseAsync(); + closeOpenClawStateDatabaseForTest(); + vi.restoreAllMocks(); +}); + +const removeAnchor: Record void> = { + a: () => { + db.prepare("DELETE FROM operator_approvals WHERE approval_id = ?").run("approval-local-1"); + }, + m: () => { + db.prepare( + "DELETE FROM audit_events WHERE sequence = (SELECT MIN(sequence) FROM audit_events)", + ).run(); + }, + g: () => { + db.prepare("DELETE FROM execution_decision_facts WHERE receipt_id = ?").run("generic-local-1"); + }, + c: () => { + db.prepare("DELETE FROM cron_run_receipts WHERE receipt_id = ?").run("cron-local-1"); + }, + t: () => { + db.prepare("DELETE FROM task_runs WHERE task_id = ?").run("task-local-1"); + }, + f: () => { + db.prepare("DELETE FROM flow_runs WHERE flow_id = ?").run("flow-local-1"); + }, +}; + +describe("inspection decision cursor rejection matrix", () => { + it.each(stages)( + "rejects malformed and unretained $prefix anchors through inspection", + async ({ prefix, owner }) => { + const first = await inspect(`${prefix}:0:0`); + expect(first.decisions).toHaveLength(1); + expect(first.decisions[0]?.source.owner).toBe(owner); + const cursor = first.nextDecisionCursor; + expect(cursor).toMatch(new RegExp(`^${prefix}:${now}:[1-9]\\d*$`)); + if (!cursor) { + throw new Error("Expected an owner cursor with another row at the same timestamp"); + } + const second = await inspect(cursor); + expect(second.decisions).toHaveLength(1); + expect(second.decisions[0]?.source.owner).toBe(owner); + expect(second.decisions[0]?.receiptId).not.toBe(first.decisions[0]?.receiptId); + + for (const suffix of [ + "-1:1", + "1:-1", + "1.5:1", + "1:1.5", + "01:1", + "1:01", + "9007199254740992:1", + "1:9007199254740992", + "1", + "1:1:extra", + ]) { + await expect(inspect(`${prefix}:${suffix}`)).rejects.toThrow( + "invalid execution decision cursor", + ); + } + const rowId = cursor.split(":")[2]; + for (const invalid of [ + `${prefix}:${now + 1}:${rowId}`, + `${prefix}:${now}:9007199254740991`, + ]) { + await expect(inspect(invalid)).rejects.toThrow( + "decision cursor is no longer retained; restart inspection without --cursor", + ); + } + await expect(inspect(cursor, "execution-foreign")).rejects.toBeInstanceOf( + ExecutionDecisionCursorError, + ); + await expect(inspect(cursor, "execution-foreign")).rejects.toThrow( + "decision cursor is no longer retained; restart inspection without --cursor", + ); + if (prefix === "a") { + // Approvals page the run correlation and expose a mismatched execution + // only as unknown evidence; the other owners scope their cursor itself. + const sibling = await inspect(cursor, "execution-sibling"); + expect(sibling.decisions).toHaveLength(1); + expect(sibling.decisions[0]).toMatchObject({ + contextId: "context-sibling", + executionId: "execution-sibling", + runId: "run-local", + decision: { + outcome: "unknown", + reasonCode: "operator_approval_execution_link_mismatch", + }, + enforcement: { coverageState: "unknown", grantRefs: [] }, + missingEvidence: ["decision.execution_link"], + }); + } else { + await expect(inspect(cursor, "execution-sibling")).rejects.toThrow( + "decision cursor is no longer retained; restart inspection without --cursor", + ); + } + removeAnchor[prefix](); + await expect(inspect(cursor)).rejects.toThrow( + "decision cursor is no longer retained; restart inspection without --cursor", + ); + expect((await inspect(`${prefix}:0:0`)).decisions[0]?.receiptId).toBe( + second.decisions[0]?.receiptId, + ); + }, + ); +}); diff --git a/src/gateway/operator-approval-store.execution-identity.test.ts b/src/gateway/operator-approval-store.execution-identity.test.ts index 0939d38457d3..ff4bd2d2480a 100644 --- a/src/gateway/operator-approval-store.execution-identity.test.ts +++ b/src/gateway/operator-approval-store.execution-identity.test.ts @@ -1,6 +1,7 @@ import fs from "node:fs"; import { afterEach, describe, expect, it } from "vitest"; import { useAutoCleanupTempDirTracker } from "../../test/helpers/temp-dir.js"; +import { assertSqliteSchemaContains } from "../infra/sqlite-schema-contract.js"; import { closeOpenClawStateDatabaseAsync, closeOpenClawStateDatabaseForTest, @@ -14,6 +15,11 @@ import { resolveOperatorApproval, } from "./operator-approval-store.js"; import { insertOperatorApprovalInDatabase as insertOperatorApprovalNative } from "./operator-approval-store.kernel.js"; +import { + getOperatorApprovalDetailed as getOlderOperatorApproval, + OLDER_OPERATOR_APPROVAL_SCHEMA_SQL, + resolveOperatorApproval as resolveOlderOperatorApproval, +} from "./operator-approval-store.older-reader.test-support.js"; type NewOperatorApproval = Parameters[0]["approval"]; const tempDirs = useAutoCleanupTempDirTracker(afterEach); @@ -259,4 +265,86 @@ describe("operator approval execution identity", () => { }); } }); + + it("preserves companion identity through an older approval reader write and candidate reopen", async () => { + const options = databaseOptions(); + await insertOperatorApproval({ + approval: approval("older-reader", token()), + databaseOptions: options, + }); + const candidate = openOpenClawStateDatabase(options); + const version = candidate.db.prepare("PRAGMA user_version").get(); + const binding = candidate.db + .prepare("SELECT * FROM operator_approval_execution_identities") + .all(); + expect(binding).toEqual([ + { + approval_id: "older-reader", + source_context_id: "context-1", + source_execution_id: "execution-1", + }, + ]); + await closeOpenClawStateDatabaseAsync(); + closeOpenClawStateDatabaseForTest(); + + // The pinned reader predates companion identities. Its original decoder and + // decision transition reopen through the current shared database owner. + expect( + getOlderOperatorApproval({ id: "older-reader", nowMs: 2_000, databaseOptions: options }), + ).toMatchObject({ outcome: "found", record: { status: "pending", decision: null } }); + const older = openOpenClawStateDatabase(options); + assertSqliteSchemaContains(older.db, older.path, OLDER_OPERATOR_APPROVAL_SCHEMA_SQL); + expect( + resolveOlderOperatorApproval({ + id: "older-reader", + decision: "allow-once", + resolver: { kind: "device", id: "older-reviewer" }, + expectedKind: "exec", + runtimeEpoch: "runtime-a", + nowMs: 2_000, + databaseOptions: options, + }), + ).toMatchObject({ outcome: "resolved", record: { decision: "allow-once" } }); + expect( + getOlderOperatorApproval({ id: "older-reader", nowMs: 2_000, databaseOptions: options }), + ).toMatchObject({ outcome: "found", record: { status: "allowed", decision: "allow-once" } }); + expect(older.db.prepare("PRAGMA user_version").get()).toEqual(version); + await closeOpenClawStateDatabaseAsync(); + closeOpenClawStateDatabaseForTest(); + + expect( + await getOperatorApprovalDetailed({ + id: "older-reader", + nowMs: 3_000, + databaseOptions: options, + }), + ).toMatchObject({ + outcome: "found", + record: { decision: "allow-once", resolver: { id: "older-reviewer" } }, + }); + expect( + await consumeOperatorApprovalAllowOnce({ + id: "older-reader", + consumerId: "candidate-consumer", + expectedKind: "exec", + runtimeEpoch: "runtime-a", + nowMs: 3_000, + databaseOptions: options, + }), + ).toMatchObject({ outcome: "consumed" }); + expect( + await getOperatorApprovalDetailed({ + id: "older-reader", + nowMs: 3_000, + databaseOptions: options, + }), + ).toMatchObject({ outcome: "found", record: { consumedBy: "candidate-consumer" } }); + const reopened = openOpenClawStateDatabase(options).db; + expect(reopened.prepare("SELECT * FROM operator_approval_execution_identities").all()).toEqual( + binding, + ); + expect(reopened.prepare("PRAGMA user_version").get()).toEqual(version); + expect(reopened.prepare("PRAGMA integrity_check").all()).toEqual([{ integrity_check: "ok" }]); + expect(reopened.prepare("PRAGMA foreign_key_check").all()).toEqual([]); + }); }); diff --git a/src/gateway/operator-approval-store.older-reader.test-support.ts b/src/gateway/operator-approval-store.older-reader.test-support.ts new file mode 100644 index 000000000000..eca3f42c5d00 --- /dev/null +++ b/src/gateway/operator-approval-store.older-reader.test-support.ts @@ -0,0 +1,613 @@ +import { normalizeNullableString as normalizeString } from "@openclaw/normalization-core/string-coerce"; +import { validateApprovalPresentation } from "../../packages/gateway-protocol/src/approval-result-validators.js"; +import { isWellFormedApprovalId } from "../../packages/gateway-protocol/src/schema/approval-id.js"; +// Frozen pre-companion approval reader from aa7cf44c75a123a2724f20b05cd10d66cf7e65f3. +// get/resolve and their validation/query helpers below are unmodified historical source. +// Database, protocol, and nullable-string normalization dependencies are current; +// normalization is equivalent for the historical string/null/undefined input domain. +// This qualifies the older reader, not an older binary's schema-version admission. +import type { ApprovalPresentation } from "../../packages/gateway-protocol/src/schema/approvals.js"; +import { + buildApprovalResolutionRef, + isApprovalResolutionRef, +} from "../infra/approval-resolution-ref.js"; +import { + executeSqliteQuerySync, + executeSqliteQueryTakeFirstSync, + getNodeSqliteKysely, +} from "../infra/kysely-sync.js"; +import { + type OpenClawStateDatabaseOptions, + openOpenClawStateDatabase, + runOpenClawStateWriteTransaction, +} from "../state/openclaw-state-db.js"; +import type { + GetOperatorApprovalResult, + OperatorApprovalDatabase, + OperatorApprovalDecision, + OperatorApprovalKind, + OperatorApprovalRecord, + OperatorApprovalResolver, + OperatorApprovalResolverKind, + OperatorApprovalRow, + OperatorApprovalStatus, + OperatorApprovalTerminalReason, + ResolveOperatorApprovalResult, +} from "./operator-approval-store.types.js"; + +const OPERATOR_APPROVAL_MAX_AUDIENCE_SESSION_KEYS = 64; + +const OPERATOR_APPROVAL_DECISIONS = new Set([ + "allow-once", + "allow-always", + "deny", +]); +const OPERATOR_APPROVAL_KINDS = new Set(["exec", "plugin", "system-agent"]); +const OPERATOR_APPROVAL_STATUSES = new Set([ + "pending", + "allowed", + "denied", + "expired", + "cancelled", +]); +const OPERATOR_APPROVAL_TERMINAL_REASONS = new Set([ + "user", + "timeout", + "malformed-verdict", + "no-route", + "run-aborted", + "gateway-restart", + "storage-corrupt", +]); +const OPERATOR_APPROVAL_RESOLVER_KINDS = new Set([ + "device", + "channel", + "runtime", + "system", +]); + +function parseApprovalPresentation(raw: string): ApprovalPresentation | null { + try { + const value: unknown = JSON.parse(raw); + return validateApprovalPresentation(value) ? value : null; + } catch { + return null; + } +} + +function parseStringArray(raw: string): string[] | null { + try { + const value: unknown = JSON.parse(raw); + if ( + !Array.isArray(value) || + value.some((entry) => typeof entry !== "string" || !entry.trim()) + ) { + return null; + } + return value as string[]; + } catch { + return null; + } +} + +function requireString(value: string, label: string): string { + const normalized = normalizeString(value); + if (!normalized) { + throw new Error(`${label} must not be empty`); + } + return normalized; +} + +function requireApprovalId(value: string): string { + if (!isWellFormedApprovalId(value)) { + throw new Error("operator approval id must be non-empty, well-formed Unicode, and not . or .."); + } + return value; +} + +function isValidTimestamp(value: number): boolean { + return Number.isSafeInteger(value) && value >= 0; +} + +function clampAuditTimestamp(nowMs: number, ...minimums: Array): number { + return Math.max(nowMs, ...minimums.filter((value): value is number => value !== null)); +} + +function hasValidLifecycleTuple(params: { + row: OperatorApprovalRow; + status: OperatorApprovalStatus; + decision: OperatorApprovalDecision | null; + terminalReason: OperatorApprovalTerminalReason | null; + resolverKind: OperatorApprovalResolverKind | null; +}): boolean { + const { row, status, decision, terminalReason, resolverKind } = params; + const noConsumption = row.consumed_at_ms === null && row.consumed_by === null; + if (status === "pending") { + return ( + decision === null && + terminalReason === null && + row.resolved_at_ms === null && + resolverKind === null && + row.resolver_id === null && + noConsumption + ); + } + if (row.resolved_at_ms === null || resolverKind === null) { + return false; + } + if (status === "allowed") { + const validConsumption = + decision === "allow-once" + ? noConsumption || (row.consumed_at_ms !== null && Boolean(row.consumed_by?.trim())) + : noConsumption; + return ( + (decision === "allow-once" || decision === "allow-always") && + terminalReason === "user" && + validConsumption + ); + } + if (decision !== "deny" || !noConsumption) { + return false; + } + if (status === "denied") { + return ( + terminalReason === "user" || + terminalReason === "malformed-verdict" || + terminalReason === "no-route" || + terminalReason === "storage-corrupt" + ); + } + if (status === "expired") { + return terminalReason === "timeout"; + } + return ( + status === "cancelled" && + (terminalReason === "run-aborted" || terminalReason === "gateway-restart") + ); +} + +function decodeOperatorApprovalRow(row: OperatorApprovalRow): OperatorApprovalRecord | null { + const presentation = parseApprovalPresentation(row.presentation_json); + const reviewerDeviceIds = parseStringArray(row.reviewer_device_ids_json); + const audienceSessionKeys = parseStringArray(row.audience_session_keys_json); + const kind = row.kind as OperatorApprovalKind; + const status = row.status as OperatorApprovalStatus; + const decision = row.decision as OperatorApprovalDecision | null; + const terminalReason = row.terminal_reason as OperatorApprovalTerminalReason | null; + const resolverKind = row.resolver_kind as OperatorApprovalResolverKind | null; + if ( + !presentation || + !isWellFormedApprovalId(row.approval_id) || + !isApprovalResolutionRef(row.resolution_ref) || + !reviewerDeviceIds || + !audienceSessionKeys || + audienceSessionKeys.length > OPERATOR_APPROVAL_MAX_AUDIENCE_SESSION_KEYS || + !OPERATOR_APPROVAL_KINDS.has(kind) || + !OPERATOR_APPROVAL_STATUSES.has(status) || + !isValidTimestamp(row.created_at_ms) || + !isValidTimestamp(row.expires_at_ms) || + !isValidTimestamp(row.updated_at_ms) || + row.expires_at_ms < row.created_at_ms || + row.updated_at_ms < row.created_at_ms || + (row.resolved_at_ms !== null && + (!isValidTimestamp(row.resolved_at_ms) || + row.resolved_at_ms < row.created_at_ms || + row.resolved_at_ms > row.updated_at_ms)) || + (row.consumed_at_ms !== null && + (!isValidTimestamp(row.consumed_at_ms) || + row.resolved_at_ms === null || + row.consumed_at_ms < row.resolved_at_ms || + row.consumed_at_ms > row.updated_at_ms)) || + (row.requested_by_device_token_auth !== 0 && row.requested_by_device_token_auth !== 1) || + (decision !== null && !OPERATOR_APPROVAL_DECISIONS.has(decision)) || + (terminalReason !== null && !OPERATOR_APPROVAL_TERMINAL_REASONS.has(terminalReason)) || + (resolverKind !== null && !OPERATOR_APPROVAL_RESOLVER_KINDS.has(resolverKind)) + ) { + return null; + } + if ( + presentation.kind !== kind || + row.resolution_ref !== + buildApprovalResolutionRef({ approvalId: row.approval_id, approvalKind: kind }) || + !hasValidLifecycleTuple({ row, status, decision, terminalReason, resolverKind }) || + (status === "allowed" && + (!decision || !Array.prototype.includes.call(presentation.allowedDecisions, decision))) + ) { + return null; + } + + return { + id: row.approval_id, + resolutionRef: row.resolution_ref, + kind, + status, + presentation, + requester: { + deviceId: row.requested_by_device_id, + clientId: row.requested_by_client_id, + deviceTokenAuth: row.requested_by_device_token_auth === 1, + }, + reviewerDeviceIds, + source: { + agentId: row.source_agent_id, + sessionKey: row.source_session_key, + sessionId: row.source_session_id, + runId: row.source_run_id, + toolCallId: row.source_tool_call_id, + toolName: row.source_tool_name, + }, + audienceSessionKeys, + runtimeEpoch: row.runtime_epoch, + createdAtMs: row.created_at_ms, + expiresAtMs: row.expires_at_ms, + updatedAtMs: row.updated_at_ms, + decision, + terminalReason, + resolvedAtMs: row.resolved_at_ms, + resolver: + resolverKind === null + ? null + : { + kind: resolverKind, + id: row.resolver_id, + }, + consumedAtMs: row.consumed_at_ms, + consumedBy: row.consumed_by, + }; +} + +function selectOperatorApprovalRow( + database: ReturnType, + id: string, +): OperatorApprovalRow | undefined { + const stateDb = getNodeSqliteKysely(database.db); + return executeSqliteQueryTakeFirstSync( + database.db, + stateDb.selectFrom("operator_approvals").selectAll().where("approval_id", "=", id), + ); +} + +function selectOperatorApprovalRowByLocator( + database: ReturnType, + locator: string, +): OperatorApprovalRow | undefined { + const stateDb = getNodeSqliteKysely(database.db); + const rows = executeSqliteQuerySync( + database.db, + stateDb + .selectFrom("operator_approvals") + .selectAll() + .where((eb) => eb.or([eb("approval_id", "=", locator), eb("resolution_ref", "=", locator)])) + .limit(2), + ).rows; + return rows.length === 1 ? rows[0] : undefined; +} + +function matchesExpectedApprovalOwner(params: { + row: OperatorApprovalRow; + expectedKind?: OperatorApprovalKind; + runtimeEpoch?: string; +}): boolean { + return ( + (params.expectedKind === undefined || params.row.kind === params.expectedKind) && + (params.runtimeEpoch === undefined || params.row.runtime_epoch === params.runtimeEpoch) + ); +} + +function denyCorruptPendingRow(params: { + database: ReturnType; + id: string; + nowMs: number; + createdAtMs: number; +}): void { + const auditTimestampMs = clampAuditTimestamp(params.nowMs, params.createdAtMs); + const stateDb = getNodeSqliteKysely(params.database.db); + executeSqliteQuerySync( + params.database.db, + stateDb + .updateTable("operator_approvals") + .set({ + status: "denied", + decision: "deny", + terminal_reason: "storage-corrupt", + resolved_at_ms: auditTimestampMs, + resolver_kind: "system", + resolver_id: null, + updated_at_ms: auditTimestampMs, + }) + .where("approval_id", "=", params.id) + .where("status", "=", "pending"), + ); +} + +function expirePendingRow(params: { + database: ReturnType; + id: string; + nowMs: number; + createdAtMs: number; +}): OperatorApprovalRow | undefined { + const auditTimestampMs = clampAuditTimestamp(params.nowMs, params.createdAtMs); + const stateDb = getNodeSqliteKysely(params.database.db); + executeSqliteQuerySync( + params.database.db, + stateDb + .updateTable("operator_approvals") + .set({ + status: "expired", + decision: "deny", + terminal_reason: "timeout", + resolved_at_ms: auditTimestampMs, + resolver_kind: "system", + resolver_id: null, + updated_at_ms: auditTimestampMs, + }) + .where("approval_id", "=", params.id) + .where("status", "=", "pending") + .where("expires_at_ms", "<=", params.nowMs), + ); + return selectOperatorApprovalRow(params.database, params.id); +} + +function requireDecodedRecord(row: OperatorApprovalRow): OperatorApprovalRecord { + const record = decodeOperatorApprovalRow(row); + if (!record) { + throw new Error(`operator approval '${row.approval_id}' became corrupt during a transaction`); + } + return record; +} + +export function getOperatorApprovalDetailed(params: { + id: string; + allowTransportRef?: boolean; + nowMs?: number; + databaseOptions?: OpenClawStateDatabaseOptions; +}): GetOperatorApprovalResult { + const locator = requireApprovalId(params.id); + return runOpenClawStateWriteTransaction((database) => { + const nowMs = params.nowMs ?? Date.now(); + let row = params.allowTransportRef + ? selectOperatorApprovalRowByLocator(database, locator) + : selectOperatorApprovalRow(database, locator); + if (!row) { + return { outcome: "not-found" }; + } + const id = row.approval_id; + if (row.status === "pending" && row.expires_at_ms <= nowMs) { + row = expirePendingRow({ database, id, nowMs, createdAtMs: row.created_at_ms }); + if (!row) { + return { outcome: "not-found" }; + } + } + const record = decodeOperatorApprovalRow(row); + if (record) { + return { outcome: "found", record }; + } + denyCorruptPendingRow({ database, id, nowMs, createdAtMs: row.created_at_ms }); + return params.allowTransportRef ? { outcome: "corrupt", id } : { outcome: "corrupt" }; + }, params.databaseOptions); +} + +export function resolveOperatorApproval(params: { + id: string; + decision: OperatorApprovalDecision; + resolver: OperatorApprovalResolver; + expectedKind?: OperatorApprovalKind; + runtimeEpoch?: string; + nowMs?: number; + databaseOptions?: OpenClawStateDatabaseOptions; +}): ResolveOperatorApprovalResult { + const id = requireApprovalId(params.id); + const resolverId = normalizeString(params.resolver.id); + const runtimeEpoch = + params.runtimeEpoch === undefined + ? undefined + : requireString(params.runtimeEpoch, "operator approval runtime epoch"); + return runOpenClawStateWriteTransaction((database) => { + const nowMs = params.nowMs ?? Date.now(); + let row = selectOperatorApprovalRow(database, id); + if (!row) { + return { outcome: "not-found" }; + } + if (!matchesExpectedApprovalOwner({ row, expectedKind: params.expectedKind, runtimeEpoch })) { + return { outcome: "not-found" }; + } + let record = decodeOperatorApprovalRow(row); + if (!record) { + denyCorruptPendingRow({ database, id, nowMs, createdAtMs: row.created_at_ms }); + return { outcome: "corrupt" }; + } + if (record.status !== "pending") { + return { + outcome: "already-resolved", + retry: record.decision === params.decision ? "same" : "conflict", + record, + }; + } + if (record.expiresAtMs <= nowMs) { + row = expirePendingRow({ database, id, nowMs, createdAtMs: row.created_at_ms }); + if (!row) { + return { outcome: "not-found" }; + } + record = requireDecodedRecord(row); + return { outcome: "expired", record }; + } + if (!Array.prototype.includes.call(record.presentation.allowedDecisions, params.decision)) { + return { outcome: "decision-not-allowed", record }; + } + + const auditTimestampMs = clampAuditTimestamp(nowMs, record.createdAtMs); + const stateDb = getNodeSqliteKysely(database.db); + let resolveQuery = stateDb + .updateTable("operator_approvals") + .set({ + status: params.decision === "deny" ? "denied" : "allowed", + decision: params.decision, + terminal_reason: "user", + resolved_at_ms: auditTimestampMs, + resolver_kind: params.resolver.kind, + resolver_id: resolverId, + updated_at_ms: auditTimestampMs, + }) + .where("approval_id", "=", id) + .where("status", "=", "pending") + .where("expires_at_ms", ">", nowMs); + if (params.expectedKind !== undefined) { + resolveQuery = resolveQuery.where("kind", "=", params.expectedKind); + } + if (runtimeEpoch !== undefined) { + resolveQuery = resolveQuery.where("runtime_epoch", "=", runtimeEpoch); + } + const result = executeSqliteQuerySync(database.db, resolveQuery); + row = selectOperatorApprovalRow(database, id); + if (!row) { + return { outcome: "not-found" }; + } + record = requireDecodedRecord(row); + if (result.numAffectedRows === 1n) { + return { outcome: "resolved", record }; + } + if (record.status === "pending" && record.expiresAtMs <= nowMs) { + const expiredRow = expirePendingRow({ + database, + id, + nowMs, + createdAtMs: record.createdAtMs, + }); + if (!expiredRow) { + return { outcome: "not-found" }; + } + return { outcome: "expired", record: requireDecodedRecord(expiredRow) }; + } + return { + outcome: "already-resolved", + retry: record.decision === params.decision ? "same" : "conflict", + record, + }; + }, params.databaseOptions); +} + +export const OLDER_OPERATOR_APPROVAL_SCHEMA_SQL = ` +CREATE TABLE IF NOT EXISTS operator_approvals ( + approval_id TEXT NOT NULL PRIMARY KEY CHECK ( + length(approval_id) > 0 AND approval_id NOT IN ('.', '..') + ), + resolution_ref TEXT NOT NULL CHECK ( + length(resolution_ref) = 43 AND resolution_ref NOT GLOB '*[^A-Za-z0-9_-]*' + ), + kind TEXT NOT NULL CHECK (kind IN ('exec', 'plugin', 'system-agent')), + status TEXT NOT NULL CHECK (status IN ('pending', 'allowed', 'denied', 'expired', 'cancelled')), + presentation_json TEXT NOT NULL, + requested_by_device_id TEXT, + requested_by_client_id TEXT, + requested_by_device_token_auth INTEGER NOT NULL DEFAULT 0, + reviewer_device_ids_json TEXT NOT NULL, + source_agent_id TEXT, + source_session_key TEXT, + source_session_id TEXT, + source_run_id TEXT, + source_tool_call_id TEXT, + source_tool_name TEXT, + audience_session_keys_json TEXT NOT NULL, + runtime_epoch TEXT NOT NULL, + created_at_ms INTEGER NOT NULL, + expires_at_ms INTEGER NOT NULL, + updated_at_ms INTEGER NOT NULL, + decision TEXT CHECK (decision IN ('allow-once', 'allow-always', 'deny')), + terminal_reason TEXT CHECK ( + terminal_reason IN ( + 'user', + 'timeout', + 'malformed-verdict', + 'no-route', + 'run-aborted', + 'gateway-restart', + 'storage-corrupt' + ) + ), + resolved_at_ms INTEGER, + resolver_kind TEXT CHECK (resolver_kind IN ('device', 'channel', 'runtime', 'system')), + resolver_id TEXT, + consumed_at_ms INTEGER, + consumed_by TEXT, + CHECK (expires_at_ms >= created_at_ms), + CHECK (updated_at_ms >= created_at_ms), + CHECK (resolved_at_ms IS NULL OR resolved_at_ms >= created_at_ms), + CHECK (resolved_at_ms IS NULL OR resolved_at_ms <= updated_at_ms), + CHECK (consumed_at_ms IS NULL OR consumed_at_ms >= resolved_at_ms), + CHECK (consumed_at_ms IS NULL OR consumed_at_ms <= updated_at_ms), + CHECK (requested_by_device_token_auth IN (0, 1)), + CHECK ( + ( + status = 'pending' + AND decision IS NULL + AND terminal_reason IS NULL + AND resolved_at_ms IS NULL + AND resolver_kind IS NULL + AND resolver_id IS NULL + AND consumed_at_ms IS NULL + AND consumed_by IS NULL + ) + OR ( + status = 'allowed' + AND decision IN ('allow-once', 'allow-always') + AND terminal_reason = 'user' + AND resolved_at_ms IS NOT NULL + AND resolver_kind IS NOT NULL + ) + OR ( + status = 'denied' + AND decision = 'deny' + AND terminal_reason IN ('user', 'malformed-verdict', 'no-route', 'storage-corrupt') + AND resolved_at_ms IS NOT NULL + AND resolver_kind IS NOT NULL + AND consumed_at_ms IS NULL + AND consumed_by IS NULL + ) + OR ( + status = 'expired' + AND decision = 'deny' + AND terminal_reason = 'timeout' + AND resolved_at_ms IS NOT NULL + AND resolver_kind IS NOT NULL + AND consumed_at_ms IS NULL + AND consumed_by IS NULL + ) + OR ( + status = 'cancelled' + AND decision = 'deny' + AND terminal_reason IN ('run-aborted', 'gateway-restart') + AND resolved_at_ms IS NOT NULL + AND resolver_kind IS NOT NULL + AND consumed_at_ms IS NULL + AND consumed_by IS NULL + ) + ), + CHECK ( + (consumed_at_ms IS NULL AND consumed_by IS NULL) + OR ( + status = 'allowed' + AND decision = 'allow-once' + AND consumed_at_ms IS NOT NULL + AND consumed_by IS NOT NULL + ) + ) +) STRICT; + +CREATE INDEX IF NOT EXISTS idx_operator_approvals_status_expiry + ON operator_approvals(status, expires_at_ms, approval_id); + +CREATE UNIQUE INDEX IF NOT EXISTS idx_operator_approvals_resolution_ref + ON operator_approvals(resolution_ref); + +CREATE INDEX IF NOT EXISTS idx_operator_approvals_source_session_created + ON operator_approvals(source_session_key, created_at_ms DESC, approval_id); + +CREATE INDEX IF NOT EXISTS idx_operator_approvals_resolved + ON operator_approvals(resolved_at_ms, approval_id) + WHERE resolved_at_ms IS NOT NULL; + +CREATE INDEX IF NOT EXISTS idx_operator_approvals_runtime_pending + ON operator_approvals(runtime_epoch, approval_id) + WHERE status = 'pending'; + +`; diff --git a/test/e2e/qa-lab/runtime/agent-run-identity-hot-reload.ts b/test/e2e/qa-lab/runtime/agent-run-identity-hot-reload.ts new file mode 100644 index 000000000000..c155a5eabdd4 --- /dev/null +++ b/test/e2e/qa-lab/runtime/agent-run-identity-hot-reload.ts @@ -0,0 +1,119 @@ +import assert from "node:assert/strict"; +import { randomUUID } from "node:crypto"; +import type { QaGatewayChild } from "../../../../extensions/qa-lab/src/gateway-child.js"; +import type { AuditRunInspectResult } from "../../../../packages/gateway-protocol/src/index.js"; +import { + connectHotReloadClient, + waitForHotReloadFact, + type HotReloadConnection, +} from "./gateway-config-hot-reload-fixtures.js"; + +export async function runIdentityGatewayTurn(gateway: QaGatewayChild, label: string) { + const accepted = (await gateway.call( + "agent", + { + sessionKey: `agent:qa:identity-hot-${randomUUID()}`, + message: `Reply exactly: ${label}`, + deliver: false, + idempotencyKey: randomUUID(), + }, + { expectFinal: false }, + )) as { status: string; runId: string }; + assert.equal(accepted.status, "accepted"); + assert.equal(typeof accepted.runId, "string"); + const terminal = (await gateway.call( + "agent.wait", + { runId: accepted.runId, timeoutMs: 60_000 }, + { timeoutMs: 65_000 }, + )) as { status: string }; + assert.equal(terminal.status, "ok", `${label} must complete through the real Gateway`); + return accepted.runId; +} + +export async function patchExecutionIdentity( + gateway: QaGatewayChild, + connection: HotReloadConnection, + values: { enabled?: boolean; executionIdentity: boolean }, +) { + const pid = gateway.pid; + const bootId = connection.bootId; + const snapshot = await connection.client.request<{ hash: string }>("config.get", {}); + // config.patch acknowledges the runtime application, not merely the file write. + const applied = await connection.client.request<{ + sentinel: { payload: { stats: { requiresRestart: boolean } } }; + }>("config.patch", { + baseHash: snapshot.hash, + raw: JSON.stringify({ logging: { audit: values } }), + }); + assert.equal(applied.sentinel.payload.stats.requiresRestart, false); + assert.equal((await connection.client.request<{ pid: number }>("system.info", {})).pid, pid); + assert.equal(connection.hellos, 1, "hot audit changes must retain the connected socket"); + assert.equal(connection.closes, 0, "hot audit changes must not disconnect operators"); + const fresh = await connectHotReloadClient(gateway); + try { + assert.equal(fresh.bootId, bootId, "hot audit changes must retain the same Gateway boot"); + } finally { + await fresh.client.stopAndWait({ timeoutMs: 2_000 }); + } +} + +export async function waitForIdentityAuditFence(gateway: QaGatewayChild, runId: string) { + return waitForHotReloadFact(`identity and terminal activity for ${runId}`, async () => { + const result = (await gateway.call("audit.run.inspect", { runId })) as AuditRunInspectResult; + const activity = (await gateway.call("audit.activity.list", { runId })) as { + events: Array<{ action: string }>; + }; + return result.identity.state === "present" && + activity.events.some((event) => event.action === "agent.run.finished") + ? result.identity.context + : undefined; + }); +} + +export async function proveHotExecutionIdentity(gateway: QaGatewayChild) { + const connection = await connectHotReloadClient(gateway); + const inspect = (runId: string) => + connection.client.request("audit.run.inspect", { runId }); + const present = (runId: string) => + waitForHotReloadFact(`persisted identity for ${runId}`, async () => { + const result = await inspect(runId); + return result.identity.state === "present" ? result.identity.context : undefined; + }); + try { + const beforeEnable = await runIdentityGatewayTurn(gateway, "HOT-DEFAULT-OFF"); + await patchExecutionIdentity(gateway, connection, { executionIdentity: true }); + const enabled = await runIdentityGatewayTurn(gateway, "HOT-ENABLED"); + const retained = await present(enabled); + assert.equal(retained.runId, enabled); + + await patchExecutionIdentity(gateway, connection, { executionIdentity: false }); + const whileDisabled = await runIdentityGatewayTurn(gateway, "HOT-DISABLED"); + assert.deepEqual(await present(enabled), retained, "disable must retain immutable evidence"); + + await patchExecutionIdentity(gateway, connection, { executionIdentity: true }); + const reenabled = await runIdentityGatewayTurn(gateway, "HOT-REENABLED"); + const next = await present(reenabled); + assert.notEqual(next.executionId, retained.executionId); + assert.notEqual(next.contextId, retained.contextId); + // The later positive write settles the same audit FIFO. Earlier disabled + // admissions must remain without identity after both enabling transitions. + for (const runId of [beforeEnable, whileDisabled]) { + const result = await inspect(runId); + assert.equal(result.identity.state, "unsupported", "enabling must not backfill a run"); + assert.deepEqual(result.decisionDisplays, []); + } + assert.deepEqual(await present(enabled), retained, "reenable must not rewrite an old context"); + return { + bootId: connection.bootId, + pid: gateway.pid, + beforeEnable, + enabled, + whileDisabled, + reenabled, + noBackfill: true, + retainedContextUnchanged: true, + }; + } finally { + await connection.client.stopAndWait({ timeoutMs: 2_000 }); + } +} diff --git a/test/e2e/qa-lab/runtime/agent-run-identity-inspection.ts b/test/e2e/qa-lab/runtime/agent-run-identity-inspection.ts index e16dc8cb243c..5424718f5a18 100644 --- a/test/e2e/qa-lab/runtime/agent-run-identity-inspection.ts +++ b/test/e2e/qa-lab/runtime/agent-run-identity-inspection.ts @@ -19,6 +19,13 @@ import { startQaMockOpenAiServer } from "../../../../extensions/qa-lab/src/provi import type { AuditRunInspectResult } from "../../../../packages/gateway-protocol/src/index.js"; import { formatErrorMessage } from "../../../../src/infra/errors.js"; import { stopQaGatewayFixture } from "../../../helpers/qa-gateway-cleanup.js"; +import { + proveHotExecutionIdentity, + patchExecutionIdentity, + runIdentityGatewayTurn, + waitForIdentityAuditFence, +} from "./agent-run-identity-hot-reload.js"; +import { connectHotReloadClient } from "./gateway-config-hot-reload-fixtures.js"; import { createQaScriptEvidenceWriter, type QaScriptEvidenceStatus } from "./script-evidence.js"; const SCENARIO_ID = "agent-run-identity-inspection"; @@ -85,24 +92,6 @@ async function assertUntrustedProxyHeadersRejected( }); } -async function updateExecutionIdentityConfig( - configPath: string, - values: { enabled?: boolean; executionIdentity: boolean }, -) { - const raw = await fs.readFile(configPath, "utf8"); - const config = parseJson(raw || "{}", "QA Gateway config") as Record; - const logging = - config.logging && typeof config.logging === "object" - ? (config.logging as Record) - : {}; - const audit = - logging.audit && typeof logging.audit === "object" - ? (logging.audit as Record) - : {}; - config.logging = { ...logging, audit: { ...audit, ...values } }; - await fs.writeFile(configPath, `${JSON.stringify(config, null, 2)}\n`, "utf8"); -} - function parseOptions(argv: readonly string[]): ProducerOptions { const readValue = (name: string) => { const index = argv.indexOf(name); @@ -400,7 +389,12 @@ async function runProof(options: ProducerOptions): Promise { try { gateway = await gatewayOwner.start({ repoRoot: options.repoRoot, - useRepoCli: true, + command: { + executablePath: process.execPath, + argsPrefix: [path.join(options.repoRoot, "dist/index.js")], + cwd: options.repoRoot, + usePackagedPlugins: true, + }, providerBaseUrl: `${mock.baseUrl}/v1`, providerMode: "mock-openai", transportBaseUrl: "http://127.0.0.1", @@ -425,8 +419,8 @@ async function runProof(options: ProducerOptions): Promise { if (inspectExecutionIdentityStorage(gateway).tablePresent) { throw new Error("existing-install restart unexpectedly created execution identity storage"); } - await gateway.restartAfterStateMutation(async ({ configPath }) => { - await updateExecutionIdentityConfig(configPath, { executionIdentity: true }); + const hotReload = await proveHotExecutionIdentity(gateway); + await gateway.restartAfterStateMutation(async () => { await runLocalTurn(gateway!, "Reply exactly: IDENTITY-INSPECTION-OK"); }); const runId = findLocalRunId(gateway); @@ -593,22 +587,42 @@ async function runProof(options: ProducerOptions): Promise { } } const retainedBeforeGlobalDisable = inspectExecutionIdentityStorage(gateway).rowCount; - await gateway.restartAfterStateMutation(async ({ configPath }) => { - await updateExecutionIdentityConfig(configPath, { + const globalDisableConnection = await connectHotReloadClient(gateway); + let globalDisabledRunId: string; + try { + await patchExecutionIdentity(gateway, globalDisableConnection, { enabled: false, executionIdentity: true, }); - await runLocalTurn(gateway!, "Reply exactly: IDENTITY-DISABLED-GLOBAL"); - }); - if (inspectExecutionIdentityStorage(gateway).rowCount !== retainedBeforeGlobalDisable) { - throw new Error("global audit disable unexpectedly retained a new execution context"); + globalDisabledRunId = await runIdentityGatewayTurn(gateway, "IDENTITY-DISABLED-GLOBAL"); + const afterGlobalDisable = parseJson( + await gateway.runCli(["audit", "--run", runId, "--explain", "--json"]), + "global-disabled retained inspection", + ) as AuditRunInspectResult; + if (normalizedContextJson(afterGlobalDisable) !== beforeContext) { + throw new Error("global audit disable hid or changed retained identity evidence"); + } + await patchExecutionIdentity(gateway, globalDisableConnection, { + enabled: true, + executionIdentity: true, + }); + const fenceRunId = await runIdentityGatewayTurn(gateway, "IDENTITY-GLOBAL-REENABLED"); + await waitForIdentityAuditFence(gateway, fenceRunId); + } finally { + await globalDisableConnection.client.stopAndWait({ timeoutMs: 2_000 }); } - const afterGlobalDisable = parseJson( - await gateway.runCli(["audit", "--run", runId, "--explain", "--json"]), - "global-disabled retained inspection", - ) as AuditRunInspectResult; - if (normalizedContextJson(afterGlobalDisable) !== beforeContext) { - throw new Error("global audit disable hid or changed retained identity evidence"); + // A later identity and terminal event have crossed the same writer FIFO; + // agent.wait alone does not settle queued audit persistence. + if (inspectExecutionIdentityStorage(gateway).rowCount !== retainedBeforeGlobalDisable + 1) { + throw new Error( + "global audit disable unexpectedly retained or backfilled an execution context", + ); + } + const disabledActivity = (await gateway.call("audit.activity.list", { + runId: globalDisabledRunId, + })) as { events: unknown[] }; + if (disabledActivity.events.length !== 0) { + throw new Error("hot global audit disable recorded activity for a subsequent run"); } const snapshotPath = path.join(options.artifactBase, SNAPSHOT_FILE); @@ -635,6 +649,7 @@ async function runProof(options: ProducerOptions): Promise { contextSha256: sha256(beforeContext), byteEquivalentAfterRestart: true, byteEquivalentPersistedReadback: true, + hotReload, optIn: { explicitEnablement: true, freshInstallDisabled: true, diff --git a/test/e2e/qa-lab/runtime/autonomous-task-lifecycle-receipts.fixtures.ts b/test/e2e/qa-lab/runtime/autonomous-task-lifecycle-receipts.fixtures.ts new file mode 100644 index 000000000000..b7fc8cca6bd0 --- /dev/null +++ b/test/e2e/qa-lab/runtime/autonomous-task-lifecycle-receipts.fixtures.ts @@ -0,0 +1,448 @@ +// Shared receipt evidence helpers and admitted webhook proof for the real QA Gateway. +import { randomUUID } from "node:crypto"; +import fs from "node:fs/promises"; +import path from "node:path"; +import { DatabaseSync } from "node:sqlite"; +import { setTimeout as delay } from "node:timers/promises"; +import { isRecord } from "@openclaw/normalization-core/record-coerce"; +import type { QaGatewayChild } from "../../../../extensions/qa-lab/src/gateway-child.js"; +import { validateExecutionIdentityContextV1 } from "../../../../packages/gateway-protocol/src/audit-run-validators.js"; +import { lazyCompile } from "../../../../packages/gateway-protocol/src/protocol-validator.js"; +import { + AuditRunInspectResultSchema, + type AuditRunInspectResult, + type ExecutionIdentityContextV1, +} from "../../../../packages/gateway-protocol/src/schema/audit-run.js"; +import { formatErrorMessage } from "../../../../src/infra/errors.js"; + +export type ExactOwnerRow = { + context_id: string; + execution_id: string; + run_id: string; + status: string; +}; +type OwnerDisplayProducer = "cron-lifecycle" | "task-lifecycle" | "flow-lifecycle"; +type ContextRow = { + context_id: string; + execution_id: string; + run_id: string; + context_json: string; +}; + +export function hasSqliteColumns( + db: DatabaseSync, + table: string, + columns: readonly string[], +): boolean { + const exists = db + .prepare("SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = ?") + .get(table); + if (!exists) { + return false; + } + const present = new Set( + (db.prepare(`PRAGMA table_info(${table})`).all() as Array<{ name: string }>).map( + (row) => row.name, + ), + ); + return columns.every((column) => present.has(column)); +} + +function parseJson(raw: string, label: string): unknown { + try { + return JSON.parse(raw); + } catch (error: unknown) { + throw new Error(`${label} was not JSON: ${formatErrorMessage(error)}`, { cause: error }); + } +} + +const validateAuditInspection = lazyCompile(AuditRunInspectResultSchema); + +export function parseAuditInspection(raw: string, label: string): AuditRunInspectResult { + const value = parseJson(raw, label); + if (!validateAuditInspection(value)) { + throw new Error(`${label} did not match the audit inspection contract`); + } + return value; +} + +export function stateDatabasePath(gateway: QaGatewayChild): string { + const stateDir = gateway.runtimeEnv.OPENCLAW_STATE_DIR; + if (!stateDir) { + throw new Error("QA Gateway did not expose its isolated state directory"); + } + return path.join(stateDir, "state", "openclaw.sqlite"); +} + +export function countExecutionContexts(gateway: QaGatewayChild): number { + const db = new DatabaseSync(stateDatabasePath(gateway), { readOnly: true }); + try { + if (!hasSqliteColumns(db, "execution_identity_contexts", ["context_id"])) { + return 0; + } + const row = db.prepare("SELECT COUNT(*) AS count FROM execution_identity_contexts").get() as { + count: number; + }; + return row.count; + } finally { + db.close(); + } +} + +function readExecutionContexts(gateway: QaGatewayChild): ContextRow[] { + const db = new DatabaseSync(stateDatabasePath(gateway), { readOnly: true }); + try { + if (!hasSqliteColumns(db, "execution_identity_contexts", ["context_json"])) { + return []; + } + return db + .prepare( + "SELECT context_id, execution_id, run_id, context_json FROM execution_identity_contexts ORDER BY execution_id", + ) + .all() as ContextRow[]; + } finally { + db.close(); + } +} + +function requirePrivateSentinelsAbsent( + text: string, + sentinels: readonly string[], + surface: string, +) { + if (sentinels.some((sentinel) => text.includes(sentinel))) { + throw new Error(`${surface} leaked a private webhook or prompt sentinel`); + } +} + +function requireWebhookContext(context: ExecutionIdentityContextV1) { + if ( + context.ingress.kind !== "webhook" || + context.ingress.boundary !== "gateway.hooks.agent" || + context.ingress.state !== "present" || + !/^hmac-sha256:v1:[a-f0-9]{32}:[a-f0-9]{64}$/u.test(context.ingress.sourceRef ?? "") + ) { + throw new Error("admitted mapped webhook omitted its pseudonymized ingress source"); + } + if ( + JSON.stringify(context.invoker) !== JSON.stringify({ state: "absent" }) || + context.coverageState !== "unattributed" || + !context.missingEvidence.includes("invoker.principal") + ) { + throw new Error("mapped webhook source or shared authentication became invoker evidence"); + } +} + +function requireWebhookIdentity(result: AuditRunInspectResult, row: ContextRow) { + if (result.identity.state !== "present") { + throw new Error("mapped webhook inspection did not retain its exact persisted context"); + } + const context = result.identity.context; + if ( + JSON.stringify(context) !== row.context_json || + context.contextId !== row.context_id || + context.executionId !== row.execution_id || + context.runId !== row.run_id + ) { + throw new Error("mapped webhook inspection did not retain its exact persisted context"); + } + requireWebhookContext(context); + const admission = result.decisionDisplays.find( + (display) => + display.provenance.state === "verified" && display.provenance.producer === "run-admission", + ); + if ( + admission?.enforcement.coverageState !== "unattributed" || + admission.decision.outcome !== "not-applicable" + ) { + throw new Error("mapped webhook source or shared authentication became invoker evidence"); + } + return context; +} + +export function readCliOwnerRows( + gateway: QaGatewayChild, + runId: string, +): { task: ExactOwnerRow } | undefined { + const db = new DatabaseSync(stateDatabasePath(gateway), { readOnly: true }); + try { + if ( + !hasSqliteColumns(db, "execution_identity_contexts", ["context_id", "execution_id"]) || + !hasSqliteColumns(db, "execution_owner_lifecycle_bindings", [ + "owner_kind", + "owner_id", + "context_id", + "execution_id", + ]) || + !hasSqliteColumns(db, "task_runs", ["task_id"]) + ) { + return undefined; + } + const task = db + .prepare( + `SELECT binding.context_id, binding.execution_id, context.run_id, task.status + FROM task_runs AS task + JOIN execution_owner_lifecycle_bindings AS binding + ON binding.owner_kind = 'task' AND binding.owner_id = task.task_id + JOIN execution_identity_contexts AS context + ON context.context_id = binding.context_id + AND context.execution_id = binding.execution_id + WHERE task.runtime = 'cli' AND task.run_id = ? AND task.ended_at IS NOT NULL + LIMIT 1`, + ) + .get(runId) as ExactOwnerRow | undefined; + return task ? { task } : undefined; + } finally { + db.close(); + } +} + +export async function waitFor(label: string, read: () => T | undefined): Promise { + const deadline = Date.now() + 30_000; + while (Date.now() < deadline) { + const value = read(); + if (value !== undefined) { + return value; + } + await delay(50); + } + throw new Error(`timed out waiting for ${label}`); +} + +export function requireOwnerDisplay(result: AuditRunInspectResult, producer: OwnerDisplayProducer) { + const receipt = result.decisionDisplays.find( + (candidate) => + candidate.provenance.state === "verified" && candidate.provenance.producer === producer, + ); + if ( + !receipt || + receipt.enforcement.coverageState !== "attribution-only" || + receipt.decision.outcome !== "not-applicable" + ) { + throw new Error(`inspection omitted exact attribution-only ${producer} display`); + } + return receipt; +} + +export async function inspectExecution(params: { + gateway: QaGatewayChild; + executionId: string; + producers: OwnerDisplayProducer[]; + privateSentinels: string[]; +}) { + const jsonRaw = await params.gateway.runCli([ + "audit", + "--execution", + params.executionId, + "--explain", + "--json", + ]); + const json = parseAuditInspection(jsonRaw, "owner lifecycle inspection"); + for (const producer of params.producers) { + requireOwnerDisplay(json, producer); + } + requirePrivateSentinelsAbsent(jsonRaw, params.privateSentinels, "JSON inspection"); + const human = await params.gateway.runCli([ + "audit", + "--execution", + params.executionId, + "--explain", + ]); + requirePrivateSentinelsAbsent(human, params.privateSentinels, "human inspection"); + for (const producer of params.producers) { + if (!human.includes(`Display producer: ${producer}`)) { + throw new Error(`human inspection omitted ${producer}`); + } + } + return { json, jsonRaw, human }; +} + +export function createMappedWebhookProof(hookToken: string) { + const mappingId = `PRIVATE-MAPPING-${randomUUID()}`; + const createRequest = () => ({ + requestId: `PRIVATE-REQUEST-${randomUUID()}`, + body: `PRIVATE-WEBHOOK-BODY-${randomUUID()}`, + }); + const webhookRequests = [createRequest(), createRequest()]; + const restartRequest = createRequest(); + const webhookSentinels = [ + mappingId, + ...[...webhookRequests, restartRequest].flatMap(({ requestId, body }) => [requestId, body]), + ]; + const admitRequest = async ( + gateway: QaGatewayChild, + request: ReturnType, + ) => { + const before = new Set(readExecutionContexts(gateway).map((row) => row.execution_id)); + const response = await fetch(`${gateway.baseUrl}/hooks/admitted`, { + method: "POST", + headers: { + Authorization: `Bearer ${hookToken}`, + "Content-Type": "application/json", + "Idempotency-Key": request.requestId, + "X-Request-Id": request.requestId, + }, + body: JSON.stringify({ value: request.body }), + }); + const accepted = parseJson(await response.text(), "mapped webhook admission"); + if ( + response.status !== 200 || + !isRecord(accepted) || + accepted.ok !== true || + typeof accepted.runId !== "string" || + !accepted.runId + ) { + throw new Error(`mapped webhook admission returned HTTP ${response.status}`); + } + // The HTTP dispatch id differs from the inner admitted run id. One POST + // at a time binds the new context without guessing from either id. + const row = await waitFor("one admitted webhook context", () => { + const added = readExecutionContexts(gateway).filter( + (candidate) => !before.has(candidate.execution_id), + ); + if (added.length > 1) { + throw new Error("one mapped webhook allocated multiple execution contexts"); + } + return added[0]; + }); + const persistedContext = parseJson(row.context_json, "persisted webhook context"); + if (!validateExecutionIdentityContextV1(persistedContext)) { + throw new Error("persisted webhook context did not match the execution identity contract"); + } + requireWebhookContext(persistedContext); + // Hooks invoke the isolated runner directly; only scheduled cron owns task rows. + // Observe the admitted runner's terminal event before inspection or replacement. + const completed = (await gateway.call( + "agent.wait", + { runId: row.run_id, timeoutMs: 30_000 }, + { timeoutMs: 35_000 }, + )) as { runId: string; status: string }; + if (completed.runId !== row.run_id || completed.status !== "ok") { + throw new Error(`admitted mapped webhook did not complete successfully: ${completed.status}`); + } + const inspection = await inspectExecution({ + gateway, + executionId: row.execution_id, + producers: [], + privateSentinels: webhookSentinels, + }); + const context = requireWebhookIdentity(inspection.json, row); + for (const text of [ + "Invoker [absent]", + "Ingress [present]", + "webhook at gateway.hooks.agent", + context.ingress.sourceRef!, + ]) { + if (!inspection.human.includes(text)) { + throw new Error("human inspection omitted mapped webhook identity evidence"); + } + } + return { row, context, inspection }; + }; + return { + mapping: { + id: mappingId, + match: { path: "admitted" }, + action: "agent" as const, + messageTemplate: "{{payload.value}}: reply WEBHOOK-DONE", + deliver: false, + wakeMode: "next-heartbeat" as const, + }, + async admit(gateway: QaGatewayChild) { + const webhookProofs: Array>> = []; + for (const request of webhookRequests) { + webhookProofs.push(await admitRequest(gateway, request)); + } + const [firstWebhook, secondWebhook] = webhookProofs; + if ( + !firstWebhook || + !secondWebhook || + firstWebhook.context.ingress.sourceRef !== secondWebhook.context.ingress.sourceRef || + firstWebhook.row.context_id === secondWebhook.row.context_id + ) { + throw new Error( + "one mapping did not retain a stable source across distinct webhook requests", + ); + } + requirePrivateSentinelsAbsent( + JSON.stringify(readExecutionContexts(gateway)), + webhookSentinels, + "persisted execution contexts", + ); + + return { + contexts: webhookProofs.map(({ context }) => ({ + contextId: context.contextId, + executionId: context.executionId, + ingress: context.ingress, + invoker: context.invoker, + coverageState: context.coverageState, + })), + inspections: webhookProofs.map(({ inspection }) => inspection.json), + async verifyAfterRestart(replacementGateway: QaGatewayChild) { + const restarted = await admitRequest(replacementGateway, restartRequest); + if ( + webhookProofs.some( + ({ row, context }) => + context.ingress.sourceRef !== restarted.context.ingress.sourceRef || + row.context_id === restarted.row.context_id || + row.execution_id === restarted.row.execution_id, + ) + ) { + throw new Error( + "new webhook admission changed its mapping source across Gateway replacement", + ); + } + const webhooksAfter = []; + const persistedAfter = readExecutionContexts(replacementGateway); + for (const proof of webhookProofs) { + const row = persistedAfter.find( + (candidate) => candidate.execution_id === proof.row.execution_id, + ); + if (!row || row.context_json !== proof.row.context_json) { + throw new Error("persisted webhook context changed across Gateway replacement"); + } + const inspection = await inspectExecution({ + gateway: replacementGateway, + executionId: row.execution_id, + producers: [], + privateSentinels: webhookSentinels, + }); + requireWebhookIdentity(inspection.json, row); + if (inspection.human !== proof.inspection.human) { + throw new Error("human webhook inspection changed across Gateway replacement"); + } + webhooksAfter.push(inspection.json); + } + requirePrivateSentinelsAbsent( + JSON.stringify(persistedAfter), + webhookSentinels, + "persisted execution contexts after restart", + ); + const auditLogLines = ( + await Promise.all( + ["gateway.stdout.log", "gateway.stderr.log"].map((file) => + fs.readFile(path.join(replacementGateway.tempRoot, file), "utf8"), + ), + ) + ) + .join("\n") + .split(/\r?\n/u) + .filter((line) => /\baudit\b|execution identity|execution decision/iu.test(line)); + requirePrivateSentinelsAbsent(auditLogLines.join("\n"), webhookSentinels, "audit logs"); + return { + inspections: webhooksAfter, + admittedContext: { + contextId: restarted.context.contextId, + executionId: restarted.context.executionId, + ingress: restarted.context.ingress, + invoker: restarted.context.invoker, + coverageState: restarted.context.coverageState, + }, + auditLogLinesChecked: auditLogLines.length, + }; + }, + }; + }, + }; +} diff --git a/test/e2e/qa-lab/runtime/autonomous-task-lifecycle-receipts.ts b/test/e2e/qa-lab/runtime/autonomous-task-lifecycle-receipts.ts index 33a8ca1fe06f..3857c78b83d2 100644 --- a/test/e2e/qa-lab/runtime/autonomous-task-lifecycle-receipts.ts +++ b/test/e2e/qa-lab/runtime/autonomous-task-lifecycle-receipts.ts @@ -14,9 +14,20 @@ import { type QaGatewayChild, } from "../../../../extensions/qa-lab/src/gateway-child.js"; import { startQaMockOpenAiServer } from "../../../../extensions/qa-lab/src/providers/mock-openai/server.js"; -import type { AuditRunInspectResult } from "../../../../packages/gateway-protocol/src/index.js"; import { formatErrorMessage } from "../../../../src/infra/errors.js"; import { stopQaGatewayFixture } from "../../../helpers/qa-gateway-cleanup.js"; +import { + countExecutionContexts, + createMappedWebhookProof, + hasSqliteColumns, + inspectExecution, + parseAuditInspection, + readCliOwnerRows, + requireOwnerDisplay, + stateDatabasePath, + waitFor, + type ExactOwnerRow, +} from "./autonomous-task-lifecycle-receipts.fixtures.js"; import { createQaScriptEvidenceWriter, type QaScriptEvidenceStatus } from "./script-evidence.js"; const SCENARIO_ID = "autonomous-task-lifecycle-receipts"; @@ -30,29 +41,6 @@ type ProofResult = { durationMs: number; status: QaScriptEvidenceStatus; }; -type ExactOwnerRow = { - context_id: string; - execution_id: string; - run_id: string; - status: string; -}; -type OwnerDisplayProducer = "cron-lifecycle" | "task-lifecycle" | "flow-lifecycle"; - -function hasSqliteColumns(db: DatabaseSync, table: string, columns: readonly string[]): boolean { - const exists = db - .prepare("SELECT 1 FROM sqlite_master WHERE type = 'table' AND name = ?") - .get(table); - if (!exists) { - return false; - } - const present = new Set( - (db.prepare(`PRAGMA table_info(${table})`).all() as Array<{ name: string }>).map( - (row) => row.name, - ), - ); - return columns.every((column) => present.has(column)); -} - function parseOptions(argv: readonly string[]): ProducerOptions { const readValue = (name: string) => { const index = argv.indexOf(name); @@ -68,41 +56,10 @@ function parseOptions(argv: readonly string[]): ProducerOptions { }; } -function parseJson(raw: string, label: string): T { - try { - return JSON.parse(raw) as T; - } catch (error) { - throw new Error(`${label} was not JSON: ${formatErrorMessage(error)}`); - } -} - function sha256(value: string): string { return createHash("sha256").update(value).digest("hex"); } -function stateDatabasePath(gateway: QaGatewayChild): string { - const stateDir = gateway.runtimeEnv.OPENCLAW_STATE_DIR; - if (!stateDir) { - throw new Error("QA Gateway did not expose its isolated state directory"); - } - return path.join(stateDir, "state", "openclaw.sqlite"); -} - -function countExecutionContexts(gateway: QaGatewayChild): number { - const db = new DatabaseSync(stateDatabasePath(gateway), { readOnly: true }); - try { - if (!hasSqliteColumns(db, "execution_identity_contexts", ["context_id"])) { - return 0; - } - const row = db.prepare("SELECT COUNT(*) AS count FROM execution_identity_contexts").get() as { - count: number; - }; - return row.count; - } finally { - db.close(); - } -} - function readCronOwnerRows( gateway: QaGatewayChild, jobId: string, @@ -183,114 +140,21 @@ function readCronOwnerBindingDiagnostic(gateway: QaGatewayChild, jobId: string): } } -function readCliOwnerRows( - gateway: QaGatewayChild, - runId: string, -): { task: ExactOwnerRow } | undefined { - const db = new DatabaseSync(stateDatabasePath(gateway), { readOnly: true }); - try { - if ( - !hasSqliteColumns(db, "execution_identity_contexts", ["context_id", "execution_id"]) || - !hasSqliteColumns(db, "execution_owner_lifecycle_bindings", [ - "owner_kind", - "owner_id", - "context_id", - "execution_id", - ]) || - !hasSqliteColumns(db, "task_runs", ["task_id"]) - ) { - return undefined; - } - const task = db - .prepare( - `SELECT binding.context_id, binding.execution_id, context.run_id, task.status - FROM task_runs AS task - JOIN execution_owner_lifecycle_bindings AS binding - ON binding.owner_kind = 'task' AND binding.owner_id = task.task_id - JOIN execution_identity_contexts AS context - ON context.context_id = binding.context_id - AND context.execution_id = binding.execution_id - WHERE task.runtime = 'cli' AND task.run_id = ? AND task.ended_at IS NOT NULL - LIMIT 1`, - ) - .get(runId) as ExactOwnerRow | undefined; - return task ? { task } : undefined; - } finally { - db.close(); - } -} - -async function waitFor(label: string, read: () => T | undefined): Promise { - const deadline = Date.now() + 30_000; - while (Date.now() < deadline) { - const value = read(); - if (value !== undefined) { - return value; - } - await delay(50); - } - throw new Error(`timed out waiting for ${label}`); -} - -function requireOwnerDisplay(result: AuditRunInspectResult, producer: OwnerDisplayProducer) { - const receipt = result.decisionDisplays.find( - (candidate) => - candidate.provenance.state === "verified" && candidate.provenance.producer === producer, - ); - if ( - !receipt || - receipt.enforcement.coverageState !== "attribution-only" || - receipt.decision.outcome !== "not-applicable" - ) { - throw new Error(`inspection omitted exact attribution-only ${producer} display`); - } - return receipt; -} - -async function inspectExecution(params: { - gateway: QaGatewayChild; - executionId: string; - producers: OwnerDisplayProducer[]; - privateSentinels: string[]; -}) { - const jsonRaw = await params.gateway.runCli([ - "audit", - "--execution", - params.executionId, - "--explain", - "--json", - ]); - const json = parseJson(jsonRaw, "owner lifecycle inspection"); - for (const producer of params.producers) { - requireOwnerDisplay(json, producer); - } - for (const sentinel of params.privateSentinels) { - if (jsonRaw.includes(sentinel)) { - throw new Error(`owner receipt leaked private sentinel ${sentinel}`); - } - } - const human = await params.gateway.runCli([ - "audit", - "--execution", - params.executionId, - "--explain", - ]); - for (const producer of params.producers) { - if (!human.includes(`Display producer: ${producer}`)) { - throw new Error(`human inspection omitted ${producer}`); - } - } - return { json, jsonRaw, human }; -} - async function runProof(options: ProducerOptions): Promise { const mock = await startQaMockOpenAiServer(); const gatewayOwner = createQaGatewayChild(); let gateway: QaGatewayChild | undefined; + const webhook = createMappedWebhookProof(HOOK_TOKEN); try { gateway = await gatewayOwner.start({ repoRoot: options.repoRoot, useRepoCli: true, + command: { + executablePath: process.execPath, + argsPrefix: [path.join(options.repoRoot, "dist/index.js")], + cwd: options.repoRoot, + usePackagedPlugins: true, + }, providerBaseUrl: `${mock.baseUrl}/v1`, providerMode: "mock-openai", transportBaseUrl: "http://127.0.0.1", @@ -305,6 +169,7 @@ async function runProof(options: ProducerOptions): Promise { enabled: true, token: HOOK_TOKEN, mappings: [ + webhook.mapping, { id: "qa-suppressed-source", match: { path: "suppressed" }, @@ -341,6 +206,8 @@ async function runProof(options: ProducerOptions): Promise { throw new Error("pre-admission mapping suppression allocated execution identity"); } + const webhookProof = await webhook.admit(gateway); + const cronSentinel = `PRIVATE-CRON-${randomUUID()}`; const cronJob = (await gateway.call("cron.add", { name: "QA autonomous receipt", @@ -375,7 +242,7 @@ async function runProof(options: ProducerOptions): Promise { producers: ["cron-lifecycle", "task-lifecycle"], privateSentinels: [cronSentinel], }); - const cronCursorPage = parseJson( + const cronCursorPage = parseAuditInspection( await gateway.runCli([ "audit", "--execution", @@ -421,6 +288,7 @@ async function runProof(options: ProducerOptions): Promise { const beforeRestart = JSON.stringify({ cron: cronInspection.json, task: taskInspection.json, + webhooks: webhookProof.inspections, }); await gateway.restartAfterStateMutation(async () => {}); const cronAfter = await inspectExecution({ @@ -435,7 +303,12 @@ async function runProof(options: ProducerOptions): Promise { producers: ["task-lifecycle"], privateSentinels: [taskSentinel], }); - const afterRestart = JSON.stringify({ cron: cronAfter.json, task: taskAfter.json }); + const webhooksAfter = await webhookProof.verifyAfterRestart(gateway); + const afterRestart = JSON.stringify({ + cron: cronAfter.json, + task: taskAfter.json, + webhooks: webhooksAfter.inspections, + }); if (afterRestart !== beforeRestart) { throw new Error("owner lifecycle JSON changed across Gateway replacement"); } @@ -468,6 +341,8 @@ async function runProof(options: ProducerOptions): Promise { `${JSON.stringify( { suppression: { httpStatus: 204, identityAllocation: 0 }, + webhooks: webhookProof.contexts, + webhookAdmittedAfterRestart: webhooksAfter.admittedContext, cron: { contextId: cronRows.cron.context_id, executionId: cronRows.cron.execution_id, @@ -483,7 +358,12 @@ async function runProof(options: ProducerOptions): Promise { cursorCompatibility: { cronPrefixAccepted: true }, genericDuplicateAbsent: true, byteEquivalentAfterRestart: true, - privacy: { cronPromptAbsent: true, taskPromptAbsent: true }, + privacy: { + cronPromptAbsent: true, + taskPromptAbsent: true, + webhookMappingRequestAndBodyAbsent: true, + auditLogLinesChecked: webhooksAfter.auditLogLinesChecked, + }, resultSha256: sha256(afterRestart), }, null, @@ -554,7 +434,7 @@ if (import.meta.url === pathToFileURL(process.argv[1] ?? "").href) { .then((exitCode) => { process.exitCode = exitCode; }) - .catch((error) => { + .catch((error: unknown) => { console.error(formatErrorMessage(error)); process.exitCode = 1; }); diff --git a/test/qa-convex-credential-payload-validation.test.ts b/test/qa-convex-credential-payload-validation.test.ts index 22c6ba831e0a..c4e97d8d03f4 100644 --- a/test/qa-convex-credential-payload-validation.test.ts +++ b/test/qa-convex-credential-payload-validation.test.ts @@ -5,6 +5,34 @@ import { normalizeCredentialPayloadForKind } from "../qa/convex-credential-broke const BUZZ_DRIVER_PRIVATE_KEY = "01".repeat(32); const BUZZ_SUT_PRIVATE_KEY = "02".repeat(32); const BUZZ_DRIVER_NSEC = "nsec1qyqszqgpqyqszqgpqyqszqgpqyqszqgpqyqszqgpqyqszqgpqyqstywftw"; +const TELEGRAM_PRIMARY_ARCHIVE = "YQ=="; +const TELEGRAM_GUEST_ARCHIVE = "Yg=="; + +function buildTelegramTestUserbotPayload() { + return { + schemaVersion: 1, + environment: "test", + groupId: "-1001", + forumGroupId: "-1002", + forumTopicId: 42, + sutToken: "test-token", + sutUsername: "test_bot", + sutBotId: "700000001", + testerUserId: "700000002", + tdlibArchiveBase64: TELEGRAM_PRIMARY_ARCHIVE, + tdlibArchiveSha256: "a".repeat(64), + tdlibVersion: "1.8.67", + participants: [ + { + alias: "guest", + testerUserId: "700000003", + tdlibArchiveBase64: TELEGRAM_GUEST_ARCHIVE, + tdlibArchiveSha256: "b".repeat(64), + tdlibVersion: "1.8.67", + }, + ], + }; +} describe("QA Convex credential payload validation", () => { it("normalizes Buzz credential payloads", () => { @@ -195,6 +223,51 @@ describe("QA Convex credential payload validation", () => { }); }); + it("retains a validated Telegram forum topic and distinct participant sessions", () => { + const normalized = normalizeCredentialPayloadForKind( + "telegram-test-userbot", + buildTelegramTestUserbotPayload(), + ); + + expect({ + forumGroupId: normalized.forumGroupId, + forumTopicId: normalized.forumTopicId, + participants: normalized.participants, + }).toEqual({ + forumGroupId: "-1002", + forumTopicId: 42, + participants: [ + { + alias: "guest", + testerUserId: "700000003", + tdlibArchiveBase64: TELEGRAM_GUEST_ARCHIVE, + tdlibArchiveSha256: "b".repeat(64), + tdlibVersion: "1.8.67", + }, + ], + }); + }); + + it("rejects invalid Telegram forum selectors and duplicate participant authority", () => { + expect(() => + normalizeCredentialPayloadForKind("telegram-test-userbot", { + ...buildTelegramTestUserbotPayload(), + forumTopicId: 0, + }), + ).toThrow(/invalid forumTopicId/u); + expect(() => + normalizeCredentialPayloadForKind("telegram-test-userbot", { + ...buildTelegramTestUserbotPayload(), + participants: [ + { + ...buildTelegramTestUserbotPayload().participants[0], + testerUserId: "700000002", + }, + ], + }), + ).toThrow(/distinct participant identities/u); + }); + it.each([ ["environment", { environment: "production" }], ["bot identity", { sutBotId: "bot" }], diff --git a/test/scripts/codex-npm-plugin-live-audit.test.ts b/test/scripts/codex-npm-plugin-live-audit.test.ts new file mode 100644 index 000000000000..81110ef207a0 --- /dev/null +++ b/test/scripts/codex-npm-plugin-live-audit.test.ts @@ -0,0 +1,273 @@ +// Qualification of the package Codex harness public audit inspection boundary. +import { spawnSync } from "node:child_process"; +import { mkdirSync, readFileSync, writeFileSync } from "node:fs"; +import path from "node:path"; +import { DatabaseSync } from "node:sqlite"; +import { afterEach, describe, expect, it } from "vitest"; +import type { AuditRunInspectResult } from "../../packages/gateway-protocol/src/schema/audit-run.js"; +import { inspectCodexAudit } from "../../scripts/e2e/lib/codex-npm-plugin-live/audit-inspection.mjs"; +import { useAutoCleanupTempDirTracker } from "../helpers/temp-dir.js"; + +const CODEX_NPM_PLUGIN_LIVE_ASSERTIONS_SCRIPT = + "scripts/e2e/lib/codex-npm-plugin-live/assertions.mjs"; +const auditTempDirs = useAutoCleanupTempDirTracker(afterEach); + +function nodeOptionsWithoutExperimentalWarnings(): string { + return [process.env.NODE_OPTIONS, "--disable-warning=ExperimentalWarning"] + .filter(Boolean) + .join(" "); +} + +function writeJson(filePath: string, value: unknown) { + mkdirSync(path.dirname(filePath), { recursive: true }); + writeFileSync(filePath, `${JSON.stringify(value, null, 2)}\n`, "utf8"); +} + +function runCodexNpmPluginLiveConfigure(root: string, auditIdentity = "0") { + return spawnSync(process.execPath, [CODEX_NPM_PLUGIN_LIVE_ASSERTIONS_SCRIPT, "configure"], { + encoding: "utf8", + env: { + ...process.env, + HOME: path.join(root, "home"), + NODE_OPTIONS: nodeOptionsWithoutExperimentalWarnings(), + OPENCLAW_CONFIG_PATH: path.join(root, "state", "openclaw.json"), + OPENCLAW_STATE_DIR: path.join(root, "state"), + OPENCLAW_CODEX_NPM_PLUGIN_AUDIT_IDENTITY: auditIdentity, + }, + }); +} + +function codexAuditFixture() { + const domainRef = `hmac-sha256:v1:${"a".repeat(32)}:${"b".repeat(64)}`; + const selectors = [1, 2, 3].map((index) => ({ + runId: `codex-run-${index}`, + executionId: `codex-execution-${index}`, + contextId: `codex-context-${index}`, + })); + const replies = new Map(); + for (const selector of selectors) { + const result: AuditRunInspectResult = { + schemaVersion: 1, + run: { runId: selector.runId, executionId: selector.executionId, status: "known" }, + identity: { + state: "present", + context: { + schemaVersion: 1, + ...selector, + createdAt: 100, + trustDomain: { + kind: "gateway-cell", + domainRef, + state: "present", + }, + invoker: { state: "absent" }, + ingress: { kind: "local-cli", boundary: "agent-command.local", state: "present" }, + agentPrincipal: { + kind: "agent", + domainRef, + principalRef: "main", + }, + agentDefinition: { definitionRef: "main", state: "present" }, + runtimeInstance: { + kind: "plugin-harness", + runtimeRef: `hmac-sha256:v1:${"a".repeat(32)}:${"c".repeat(64)}`, + state: "present", + }, + applicableGrants: [], + assurance: [ + { + kind: "runtime-binding", + evidenceRef: `hmac-sha256:v1:${"a".repeat(32)}:${"d".repeat(64)}`, + strength: "boundary-verified", + }, + ], + coverageState: "unattributed", + missingEvidence: ["invoker.principal"], + }, + }, + decisionDisplays: [ + { + schemaVersion: 1, + selectorId: `${selector.contextId}:admission`, + occurredAt: 100, + action: { family: "run", operation: "admission" }, + decision: { + outcome: "not-applicable", + reasonCode: "run_admission_identity_not_evaluated", + }, + enforcement: { + coverageState: "unattributed", + grantCount: 0, + policyCount: 0, + contextFieldsUsed: [], + }, + provenance: { state: "verified", producer: "run-admission" }, + missingEvidence: ["invoker.principal"], + remediation: [], + }, + ], + coverage: { state: "unattributed", missingEvidence: ["invoker.principal"] }, + }; + replies.set(`--run ${selector.runId} --explain --limit 50`, structuredClone(result)); + replies.set(`--execution ${selector.executionId} --explain --limit 100`, result); + } + return { selectors, replies }; +} + +describe("Codex package audit qualification", () => { + it.each([ + "missing-context", + "wrong-run", + "wrong-execution", + "wrong-context", + "fallback-runtime", + "raw-runtime-reference", + "invented-invoker", + "missing-admission", + "invented-enforcement", + "private-reply", + "raw-receipts", + "truncated-decisions", + "missing-turn", + ])("rejects %s from the public audit inspector", (failure) => { + const fixture = codexAuditFixture(); + const exact = fixture.replies.get("--execution codex-execution-1 --explain --limit 100"); + if (!exact || exact.identity.state !== "present") { + throw new Error("expected exact fixture identity"); + } + switch (failure) { + case "missing-context": + exact.identity = { + state: "unsupported", + reasonCode: "identity_context_unavailable", + missingEvidence: ["identity.context"], + remediation: [], + }; + break; + case "wrong-run": + exact.identity.context.runId = "another-run"; + break; + case "wrong-execution": + exact.identity.context.executionId = "another-execution"; + break; + case "wrong-context": + exact.identity.context.contextId = "another-context"; + break; + case "fallback-runtime": + exact.identity.context.runtimeInstance.kind = "embedded"; + break; + case "raw-runtime-reference": + exact.identity.context.runtimeInstance.runtimeRef = "hmac-sha256:v1:private-runtime"; + break; + case "invented-invoker": + exact.identity.context.invoker = { + state: "present", + principal: { kind: "person", principalRef: "someone", domainRef: "domain" }, + }; + break; + case "missing-admission": + exact.decisionDisplays = []; + break; + case "invented-enforcement": + exact.decisionDisplays[0]!.enforcement.coverageState = "enforced"; + break; + case "private-reply": + exact.decisionDisplays[0]!.action.summary = "PRIVATE_REPLY_MARKER"; + break; + case "raw-receipts": + Object.assign(exact, { decisions: [] }); + break; + case "truncated-decisions": + exact.nextDecisionCursor = "a:1:1"; + break; + case "missing-turn": + fixture.selectors.pop(); + break; + } + expect(() => + inspectCodexAudit({ + selectors: fixture.selectors, + expectedExecutions: 3, + privateValues: ["PRIVATE_REPLY_MARKER"], + query: (args: string[]) => fixture.replies.get(args.join(" ")), + }), + ).toThrow(); + }); +}); + +describe("Codex audit package CLI boundary", () => { + it("enables audit identity only in the opted-in package fixture and queries the public CLI", () => { + const root = auditTempDirs.make("openclaw-codex-audit-"); + const defaultConfig = runCodexNpmPluginLiveConfigure(root); + expect(defaultConfig.status, defaultConfig.stderr).toBe(0); + expect( + JSON.parse(readFileSync(path.join(root, "state", "openclaw.json"), "utf8")), + ).not.toHaveProperty("logging.audit.executionIdentity"); + const configured = runCodexNpmPluginLiveConfigure(root, "1"); + expect(configured.status, configured.stderr).toBe(0); + expect( + JSON.parse(readFileSync(path.join(root, "state", "openclaw.json"), "utf8")), + ).toMatchObject({ + logging: { audit: { enabled: true, executionIdentity: true } }, + gateway: { mode: "local", bind: "loopback", auth: { mode: "token" } }, + }); + const fixture = codexAuditFixture(); + const databasePath = path.join(root, "state", "state", "openclaw.sqlite"); + mkdirSync(path.dirname(databasePath), { recursive: true }); + const database = new DatabaseSync(databasePath); + try { + database.exec( + "CREATE TABLE execution_identity_contexts (run_id TEXT, execution_id TEXT, context_id TEXT, created_at INTEGER)", + ); + const insert = database.prepare( + "INSERT INTO execution_identity_contexts VALUES (?, ?, ?, ?)", + ); + fixture.selectors.forEach((selector, index) => + insert.run(selector.runId, selector.executionId, selector.contextId, index), + ); + } finally { + database.close(); + } + writeJson(path.join(root, "replies.json"), Object.fromEntries(fixture.replies)); + const cliPath = path.join(root, "openclaw-fixture.cjs"); + writeFileSync( + cliPath, + `const fs = require("node:fs"); +const path = require("node:path"); +const [command, ...args] = process.argv.slice(2); +if (command !== "audit" || args.pop() !== "--json") process.exit(2); +const replies = JSON.parse(fs.readFileSync(path.join(__dirname, "replies.json"), "utf8")); +const reply = replies[args.join(" ")]; +if (!reply) process.exit(3); +fs.appendFileSync(path.join(__dirname, "queries.jsonl"), JSON.stringify(args) + "\\n"); +process.stdout.write(JSON.stringify(reply));\n`, + ); + const result = spawnSync( + process.execPath, + [CODEX_NPM_PLUGIN_LIVE_ASSERTIONS_SCRIPT, "assert-audit", "PRIVATE_REPLY_MARKER"], + { + encoding: "utf8", + env: { + ...process.env, + HOME: path.join(root, "home"), + OPENCLAW_STATE_DIR: path.join(root, "state"), + OPENCLAW_E2E_CLI_BIN: cliPath, + NODE_OPTIONS: nodeOptionsWithoutExperimentalWarnings(), + }, + }, + ); + expect(result.status, result.stderr).toBe(0); + expect(result.stdout).toContain('codex_audit_identity: {"executionCount":3}'); + expect( + readFileSync(path.join(root, "queries.jsonl"), "utf8") + .trim() + .split("\n") + .map((line) => JSON.parse(line)), + ).toEqual( + fixture.selectors.flatMap((selector) => [ + ["--run", selector.runId, "--explain", "--limit", "50"], + ["--execution", selector.executionId, "--explain", "--limit", "100"], + ]), + ); + }); +}); diff --git a/test/scripts/npm-onboard-channel-agent-shell.test.ts b/test/scripts/npm-onboard-channel-agent-shell.test.ts index c47c2d1d786c..1ad7f4b72053 100644 --- a/test/scripts/npm-onboard-channel-agent-shell.test.ts +++ b/test/scripts/npm-onboard-channel-agent-shell.test.ts @@ -22,6 +22,7 @@ type Scenario = { sourcePlugin?: boolean; helpFailure?: "exit" | "timeout"; failProbe?: number; + dirtyState?: boolean; }; function sha256(file: string): string { @@ -82,6 +83,10 @@ function runScenario(scenario: Scenario = {}) { const bundled = scenario.bundled ?? channel === "telegram"; mkdirSync(bin); mkdirSync(home); + mkdirSync(join(home, ".openclaw")); + if (scenario.dirtyState) { + writeFileSync(join(home, ".openclaw", "existing-state"), "unrelated fixture state"); + } mkdirSync(packageRoot); writeFileSync(eventsPath, ""); if (bundled) { @@ -124,6 +129,8 @@ if (help) { } else if (args[0] === "channels" && args[1] === "add" && env.BUNDLED === "0") { if (current && !fs.existsSync(dependencyPath())) fail("external channel needs consent first"); if (!current) installChannelDependency(); +} else if (args[0] === "identity") { + console.log("execution-fixture"); } function dependencyPath() { const dep = { telegram: "grammy", discord: "discord-api-types", slack: "@slack/bolt" }[env.OPENCLAW_NPM_ONBOARD_CHANNEL]; @@ -143,6 +150,13 @@ function installChannelDependency() { if [ "$1" = scripts/e2e/lib/npm-onboard-channel-agent/assertions.mjs ]; then exec "$REAL_NODE" "$FIXTURE_CLI" assertion "\${@:2}" fi +if [ "$1" = scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs ]; then + if [ "$2" = clean-home ]; then exec "$REAL_NODE" "$@"; fi + if [ "$2" = run-id ]; then printf '%s' admitted-run-fixture; exit 0; fi + # Projection/storage assertions have their own real SQLite support tests. + [ "$2" = verify ] || exit 0 + exec "$REAL_NODE" "$FIXTURE_CLI" identity "\${@:2}" +fi exec "$REAL_NODE" "$@" `, { mode: 0o755 }, @@ -156,10 +170,17 @@ exec "$REAL_NODE" "$@" throw new Error("npm onboarding container program not found"); } const testState = ` +OPENCLAW_TEST_STATE_HOME="$HOME" +export OPENCLAW_STATE_DIR="$HOME/.openclaw" +export OPENCLAW_CONFIG_PATH="$OPENCLAW_STATE_DIR/openclaw.json" openclaw_e2e_install_package() { mkdir -p "$HOME/.openclaw"; } openclaw_e2e_package_root() { printf '%s' "$PACKAGE_ROOT"; } +openclaw_e2e_package_entrypoint() { printf '%s/openclaw.mjs' "$PACKAGE_ROOT"; } openclaw_e2e_start_mock_openai() { :; } openclaw_e2e_wait_mock_openai() { :; } +openclaw_e2e_start_gateway() { "$FIXTURE_CLI" gateway-start; printf '%s' fixture-gateway; } +openclaw_e2e_wait_gateway_ready() { :; } +openclaw_e2e_stop_process() { if [ -n "$1" ]; then "$FIXTURE_CLI" gateway-stop; fi; } `; const registryEnv = scenario.registry ? registryFixture(root, scenario) : {}; if (scenario.corruptRegistry) { @@ -209,6 +230,40 @@ openclaw_e2e_wait_mock_openai() { :; } } describe("npm onboarding fixture consent", () => { + it("inspects the admitted local turn with the installed CLI across Gateway restart", () => { + const { result, events, detail } = runScenario(); + expect(result.status, detail).toBe(0); + const optIn = events.findIndex( + (args) => args.join(" ") === "config set logging.audit.executionIdentity true", + ); + const turn = events.findIndex((args) => args[0] === "agent"); + expect(optIn).toBeGreaterThanOrEqual(0); + expect(optIn).toBeLessThan(turn); + expect(events.filter((args) => args[0] === "agent")).toHaveLength(1); + expect( + events + .slice(turn + 1) + .filter((args) => + ["gateway-start", "gateway-stop", "audit"].some((name) => name === args[0]), + ) + .map((args) => args.slice(0, 3)), + ).toEqual([ + ["gateway-start"], + ["audit", "--run", "admitted-run-fixture"], + ["gateway-stop"], + ["gateway-start"], + ["audit", "--execution", "execution-fixture"], + ["gateway-stop"], + ]); + }); + + it("rejects inherited state before installing or admitting a turn", () => { + const { result, events, detail } = runScenario({ dirtyState: true }); + expect(result.status).not.toBe(0); + expect(detail).toContain("package proof inherited existing state"); + expect(events).toEqual([]); + }); + it.each([false, true])("selects the reviewed Codex source with registry=%s", (registry) => { const { result, events, installs, detail } = runScenario({ registry }); expect(result.status, detail).toBe(0); diff --git a/test/scripts/npm-onboard-execution-identity.test.ts b/test/scripts/npm-onboard-execution-identity.test.ts new file mode 100644 index 000000000000..4097d77eb6b8 --- /dev/null +++ b/test/scripts/npm-onboard-execution-identity.test.ts @@ -0,0 +1,169 @@ +import { spawnSync } from "node:child_process"; +import { existsSync, mkdirSync, readdirSync, writeFileSync } from "node:fs"; +import path from "node:path"; +import { DatabaseSync } from "node:sqlite"; +import { afterEach, describe, expect, it } from "vitest"; +import { + assertIdentityProjection, + readIdentityRows, +} from "../../scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs"; +import { useAutoCleanupTempDirTracker } from "../helpers/temp-dir.js"; + +const tempDirs = useAutoCleanupTempDirTracker(afterEach); +const helper = path.resolve("scripts/e2e/lib/npm-onboard-channel-agent/execution-identity.mjs"); +const domainRef = `hmac-sha256:v1:${"a".repeat(32)}:${"b".repeat(64)}`; +const privateMarker = "PRIVATE_PACKAGE_AUDIT_FIXTURE"; + +function projection() { + return { + run: { runId: "admitted-run-1", executionId: "execution-1" }, + identity: { + state: "present", + context: { + schemaVersion: 1, + contextId: "context-1", + executionId: "execution-1", + runId: "admitted-run-1", + createdAt: 123, + trustDomain: { kind: "gateway-cell", domainRef, state: "present" }, + invoker: { state: "absent" }, + ingress: { kind: "local-cli", boundary: "agent-command.local", state: "present" }, + agentPrincipal: { kind: "agent", domainRef, principalRef: "main" }, + agentDefinition: { definitionRef: "main", state: "present" }, + runtimeInstance: { runtimeRef: domainRef, kind: "gateway", state: "present" }, + applicableGrants: [], + assurance: [], + coverageState: "unattributed", + missingEvidence: ["invoker.principal"], + }, + }, + decisionDisplays: [ + { + provenance: { state: "verified", producer: "run-admission" }, + decision: { outcome: "not-applicable", reasonCode: "run_admission_identity_not_evaluated" }, + enforcement: { coverageState: "unattributed" }, + }, + ], + }; +} + +describe("installed-package execution identity proof", () => { + it("does not initialize missing audit state during inspection", () => { + const state = tempDirs.make("openclaw-package-audit-empty-"); + expect(readIdentityRows(state)).toEqual([]); + expect(readdirSync(state)).toEqual([]); + expect(existsSync(path.join(state, "state/openclaw.sqlite"))).toBe(false); + }); + + it("accepts the persisted run id when it differs from the caller session id", () => { + const result = projection(); + expect( + assertIdentityProjection(result, JSON.stringify(result.identity.context), [privateMarker]), + ).toBe("execution-1"); + }); + + it.each([ + [ + "missing identity", + (result: ReturnType) => { + result.identity.state = "unknown"; + }, + ], + [ + "wrong run", + (result: ReturnType) => { + result.run.runId = "other-run"; + }, + ], + [ + "fabricated invoker", + (result: ReturnType) => { + result.identity.context.invoker.state = "present"; + }, + ], + [ + "raw runtime", + (result: ReturnType) => { + result.identity.context.runtimeInstance.runtimeRef = "raw-runtime"; + }, + ], + [ + "false enforcement", + (result: ReturnType) => { + for (const display of result.decisionDisplays) { + display.enforcement.coverageState = "enforced"; + } + }, + ], + [ + "lost admission", + (result: ReturnType) => { + result.decisionDisplays = []; + }, + ], + ] as const)("rejects %s", (_label, mutate) => { + const result = projection(); + mutate(result); + expect(() => + assertIdentityProjection(result, JSON.stringify(result.identity.context), []), + ).toThrow(); + }); + + it.each(["export", "storage", "receipt"])("rejects a private canary in %s", (where) => { + const result = projection(); + const stored = JSON.stringify(result.identity.context); + const exported = + where === "export" + ? { ...result, prompt: privateMarker } + : where === "receipt" + ? { ...result, decisions: [{ body: privateMarker }] } + : result; + expect(() => + assertIdentityProjection(exported, where === "storage" ? stored + privateMarker : stored, [ + privateMarker, + ]), + ).toThrow("execution identity exposed private fixture data"); + }); + + it("rejects identity replacement and duplicate rows after the Gateway restart", () => { + const home = tempDirs.make("openclaw-package-audit-restart-"); + const stateDir = path.join(home, ".openclaw"); + mkdirSync(path.join(stateDir, "state"), { recursive: true }); + const config = path.join(stateDir, "openclaw.json"); + writeFileSync(config, "{}"); + const beforePath = path.join(home, "before.json"); + const afterPath = path.join(home, "after.json"); + const before = projection(); + const db = new DatabaseSync(path.join(stateDir, "state/openclaw.sqlite")); + const verify = () => + spawnSync(process.execPath, [helper, "verify", afterPath, beforePath], { + encoding: "utf8", + env: { + HOME: home, + OPENCLAW_HOME: home, + OPENCLAW_TEST_STATE_HOME: home, + OPENCLAW_STATE_DIR: stateDir, + OPENCLAW_CONFIG_PATH: config, + }, + }); + try { + db.exec("CREATE TABLE execution_identity_contexts (context_json TEXT NOT NULL)"); + const insert = db.prepare("INSERT INTO execution_identity_contexts VALUES (?)"); + insert.run(JSON.stringify(before.identity.context)); + writeFileSync(beforePath, JSON.stringify(before)); + writeFileSync(afterPath, JSON.stringify(before)); + expect(verify().status).toBe(0); + const after = projection(); + after.identity.context.runtimeInstance.runtimeRef = domainRef.replace(/b/gu, "c"); + db.prepare("UPDATE execution_identity_contexts SET context_json = ?").run( + JSON.stringify(after.identity.context), + ); + writeFileSync(afterPath, JSON.stringify(after)); + expect(verify().stderr).toContain("CLI context differs from persisted bytes"); + insert.run(JSON.stringify(after.identity.context)); + expect(verify().stderr).toContain("expected exactly one admitted execution identity"); + } finally { + db.close(); + } + }); +}); diff --git a/test/vitest/vitest.database-worker-core-paths.mjs b/test/vitest/vitest.database-worker-core-paths.mjs index bb39977c7f2a..b3570095d4db 100644 --- a/test/vitest/vitest.database-worker-core-paths.mjs +++ b/test/vitest/vitest.database-worker-core-paths.mjs @@ -227,6 +227,7 @@ export const databaseWorkerCoreTestFiles = [ "src/audit/audit-event-store.message.test.ts", "src/audit/audit-event-writer.test.ts", "src/audit/audit-event-writer.worker.test.ts", + "src/audit/execution-decision-cursors.test.ts", "src/audit/execution-decision-work.test.ts", "src/audit/execution-identity-context.test.ts", "src/agents/tools/sessions-access.test.ts",