openclaw/test/skill-usage.codex.integration.test.ts
Ayaan Zaidi 297956d0a9
fix(skills): restore usage counts and current Workshop inventory (#151048)
## What Problem This Solves

Fixes missing skill-use counts when tool diagnostics are disabled and missing directly created Workshop skills.

## User Impact

Status reports current Workshop files and recorded use. No skills are removed. Missing use is not proof of inactivity; native reads outside OpenClaw tools remain untracked.

## Why This Change Was Made

This state has one writer. Completed activations reach the existing counter; current-file discovery supplies membership and proposals supply known dates. Curator contracts, formatting and handlers each have one owner.

## Compatibility

One connection capability selects full inventory with unknown dates as `null`. Older clients retain numeric-date replies for current skills with known dates. New CLI versions accept older replies with a limited-coverage notice. No database migration or protocol-version bump.

## Consumers

The tool wrapper, Gateway and CLI use the existing owners. Public schema exports and existing Swift response types remain compatible.

## Invalidation

Status rereads configured roots and joins history by file path. Deleted files disappear; moving a skill does not transfer usage by name.

## Evidence

- Real Gateway and native-runtime runs with a local scripted provider: baseline completed two reads but stored no usage; candidate reported `0 → 1 → 2`, retaining `2` after restart. Directly created unused inventory remained visible with zero use and unknown dates.
- Actual `2026.9.4` CLI → candidate Gateway and candidate CLI → `2026.9.4` Gateway passed JSON/text checks with nonempty numeric-date replies. The new CLI showed the legacy-coverage notice.
- 276 focused product tests plus the plugin-boundary regression pass. All 20 native lint/type/protocol checks, architecture, unused exports, line caps, formatting and assertion checks pass.
- Native Swift decoding passed 5 positive and 6 negative cases. Scripted provider tests prove execution and transport, not live model judgment.

<details>
<summary>Reproduction and cleanup</summary>

```sh
node scripts/run-vitest.mjs run --config test/vitest/vitest.tooling.config.ts test/skill-usage.codex.integration.test.ts
node scripts/run-vitest.mjs run src/gateway/server-methods/skills.proposals.test.ts -t 'live Workshop inventory'
```

Cleanup removed duplicate status types, row conversions, inventory fixtures and repeated docs. The bridge suite remains at the cross-plugin integration boundary. The final change is 112 net lines smaller than the first draft; remaining growth is primarily regression coverage and generated compatibility types.
</details>

Co-authored-by: Ayaan Zaidi <hi@obviy.us>
2026-09-18 12:17:09 +05:30

274 lines
9.8 KiB
TypeScript

import fs from "node:fs/promises";
import path from "node:path";
import { Type } from "typebox";
import { afterEach, beforeEach, describe, expect, it, vi } from "vitest";
import { createCodexDynamicToolBridge } from "../extensions/codex/test-api.js";
import { getBeforeToolCallDiagnosticOptions } from "../src/agents/before-tool-call-metadata.js";
import { asToolParamsRecord, type AnyAgentTool } from "../src/agents/tools/common.js";
import {
onDiagnosticEvent,
onInternalDiagnosticEvent,
onTrustedInternalDiagnosticEvent,
resetDiagnosticEventsForTest,
setDiagnosticsEnabledForProcess,
type DiagnosticEventPayload,
waitForDiagnosticEventsDrained,
} from "../src/infra/diagnostic-events.js";
import {
initializeGlobalHookRunner,
resetGlobalHookRunner,
} from "../src/plugins/hook-runner-global.js";
import { createMockPluginRegistry } from "../src/plugins/hooks.test-helpers.js";
import { createEmptyPluginRegistry } from "../src/plugins/registry-empty.js";
import { setActivePluginRegistry } from "../src/plugins/runtime.js";
import { consumeRunSkillUsage } from "../src/skills/runtime/run-usage.js";
import { createCanonicalFixtureSkill } from "../src/skills/test-support/test-helpers.js";
import { registerSkillUsageTracking } from "../src/skills/workshop/curator.js";
import {
closeOpenClawStateDatabaseForTest,
openOpenClawStateDatabase,
} from "../src/state/openclaw-state-db.js";
import {
createOpenClawTestState,
type OpenClawTestState,
} from "../src/test-utils/openclaw-test-state.js";
let testState: OpenClawTestState;
describe("persistent skill usage through registered Codex dynamic tools", () => {
const runId = "skill-usage-run";
const skillName = "daily-brief";
let skillFile: string;
let unregisterUsage: () => void;
let publicEvents: DiagnosticEventPayload[];
let sharedEvents: DiagnosticEventPayload[];
let trustedEvents: DiagnosticEventPayload[];
beforeEach(async () => {
testState = await createOpenClawTestState({
prefix: "openclaw-codex-skill-usage-",
applyEnv: true,
});
resetDiagnosticEventsForTest();
skillFile = await testState.writeText("skills/daily-brief/SKILL.md", "# Daily brief\n");
resetGlobalHookRunner();
setActivePluginRegistry(createEmptyPluginRegistry());
setDiagnosticsEnabledForProcess(false);
publicEvents = [];
sharedEvents = [];
trustedEvents = [];
onDiagnosticEvent((event) => publicEvents.push(event));
onInternalDiagnosticEvent((event) => sharedEvents.push(event));
onTrustedInternalDiagnosticEvent((event) => trustedEvents.push(event));
unregisterUsage = registerSkillUsageTracking({ env: testState.env });
});
afterEach(async () => {
await waitForDiagnosticEventsDrained();
unregisterUsage();
consumeRunSkillUsage(runId);
resetDiagnosticEventsForTest();
resetGlobalHookRunner();
setActivePluginRegistry(createEmptyPluginRegistry());
vi.restoreAllMocks();
closeOpenClawStateDatabaseForTest();
await testState.cleanup();
});
function createBridge(options: { execute?: AnyAgentTool["execute"]; command?: boolean } = {}) {
const execute = vi.fn<AnyAgentTool["execute"]>(
options.execute ??
(async (_callId, args) => {
const filePath = asToolParamsRecord(args).path;
if (typeof filePath !== "string") {
throw new Error("Expected a file path");
}
return {
content: [{ type: "text", text: await fs.readFile(filePath, "utf8") }],
details: {},
};
}),
);
const toolName = options.command ? "daily_brief" : "read";
const bridge = createCodexDynamicToolBridge({
tools: [
{
name: toolName,
label: toolName,
description: "Read a file",
parameters: Type.Object({ path: Type.String() }),
execute,
},
],
signal: new AbortController().signal,
hookContext: {
agentId: "main",
sessionKey: "agent:main:skill-usage",
sessionId: "skill-usage-session",
runId,
workspaceDir: testState.workspaceDir,
loopDetection: { enabled: false },
skillsSnapshot: {
prompt: "",
skills: [{ name: skillName }],
resolvedSkills: [
createCanonicalFixtureSkill({
name: skillName,
description: "Daily brief",
filePath: skillFile,
baseDir: path.dirname(skillFile),
source: "workspace",
}),
],
},
...(options.command
? {
skillCommand: {
commandName: "daily-brief",
skillName,
skillSource: "workspace",
skillFile,
toolName,
},
}
: {}),
},
});
expect(bridge.telemetry.quarantinedTools).toEqual([]);
expect(bridge.availableTools.map((tool) => tool.name)).toEqual([toolName]);
for (const tool of bridge.availableTools) {
expect(getBeforeToolCallDiagnosticOptions(tool)?.emitDiagnostics).toBe(false);
}
const call = (callId: string, filePath = skillFile) =>
bridge.handleToolCall({
threadId: "skill-usage-thread",
turnId: "skill-usage-turn",
callId,
namespace: "openclaw",
tool: toolName,
arguments: { path: filePath },
});
return { call, execute };
}
function usageRows() {
return openOpenClawStateDatabase({ env: testState.env })
.db.prepare(
"SELECT skill_file, skill_name, skill_source, use_count, last_agent_id FROM skill_usage",
)
.all();
}
function expectedUsageRow(count: number) {
return {
skill_file: skillFile,
skill_name: skillName,
skill_source: "workspace",
use_count: count,
last_agent_id: "main",
};
}
it.each([false, true])(
"counts repeated successful reads with process diagnostics=%s",
async (enabled) => {
setDiagnosticsEnabledForProcess(enabled);
const { call, execute } = createBridge();
expect(await call("read-1")).toMatchObject({
success: true,
contentItems: [{ type: "inputText", text: "# Daily brief\n" }],
});
await waitForDiagnosticEventsDrained();
expect(usageRows()).toEqual([expectedUsageRow(1)]);
expect(await call("read-2")).toMatchObject({ success: true });
await waitForDiagnosticEventsDrained();
expect(execute).toHaveBeenCalledTimes(2);
expect(usageRows()).toEqual([expectedUsageRow(2)]);
expect(consumeRunSkillUsage(runId)).toEqual([
{ name: skillName, source: "workspace", activation: "read", skillFile },
]);
expect(consumeRunSkillUsage(runId)).toEqual([]);
expect(publicEvents).toEqual([]);
expect(sharedEvents.map((event) => event.type)).toEqual(
enabled ? ["skill.used", "skill.used"] : [],
);
expect(trustedEvents.map((event) => event.type)).toEqual(["skill.used", "skill.used"]);
expect(JSON.stringify([...sharedEvents, ...trustedEvents])).not.toContain(skillFile);
},
);
it.each(["error", "failed", "blocked", "cancelled", "timed_out"])(
"does not count a structured %s read",
async (status) => {
const { call, execute } = createBridge({
execute: async () => ({
content: [{ type: "text", text: "Read did not complete" }],
details: { status },
}),
});
expect(await call("failed-read")).toMatchObject({ success: false });
await waitForDiagnosticEventsDrained();
expect(execute).toHaveBeenCalledOnce();
expect(consumeRunSkillUsage(runId)).toEqual([]);
expect(usageRows()).toEqual([]);
expect(trustedEvents).toEqual([]);
},
);
it("does not count a thrown read", async () => {
const { call } = createBridge({
execute: async () => {
throw new Error("Read failed");
},
});
expect(await call("thrown-read")).toMatchObject({ success: false });
await waitForDiagnosticEventsDrained();
expect(usageRows()).toEqual([]);
expect(consumeRunSkillUsage(runId)).toEqual([]);
expect(trustedEvents).toEqual([]);
});
it("does not count a read blocked before execution", async () => {
initializeGlobalHookRunner(
createMockPluginRegistry([
{
hookName: "before_tool_call",
handler: async () => ({ block: true, blockReason: "Blocked by test policy" }),
},
]),
);
const { call, execute } = createBridge();
expect(await call("blocked-read")).toMatchObject({ success: false, executionStarted: false });
await waitForDiagnosticEventsDrained();
expect(execute).not.toHaveBeenCalled();
expect(usageRows()).toEqual([]);
expect(consumeRunSkillUsage(runId)).toEqual([]);
expect(trustedEvents).toEqual([]);
});
it.each(["skills/unknown/SKILL.md", "README.md"])(
"does not count reading %s outside the skill snapshot",
async (filePath) => {
const otherFile = await testState.writeText(filePath, "Other file\n");
const { call } = createBridge();
expect(await call("other-read", otherFile)).toMatchObject({ success: true });
await waitForDiagnosticEventsDrained();
expect(usageRows()).toEqual([]);
expect(consumeRunSkillUsage(runId)).toEqual([]);
expect(trustedEvents).toEqual([]);
},
);
it("preserves explicit tool-dispatched skill command activation", async () => {
const { call } = createBridge({ command: true });
expect(await call("skill-command")).toMatchObject({ success: true });
await waitForDiagnosticEventsDrained();
expect(usageRows()).toEqual([expectedUsageRow(1)]);
expect(consumeRunSkillUsage(runId)).toEqual([
{ name: skillName, source: "workspace", activation: "command", skillFile },
]);
expect(trustedEvents).toMatchObject([
{ type: "skill.used", activation: "command", toolName: "daily_brief" },
]);
});
});