mirror of
https://github.com/AgentSeal/codeburn.git
synced 2026-08-21 14:34:32 +00:00
Merge pull request #984 from Enclavet/fix/kiro-parse-oom
fix: OOM on cold parse (V8 SlicedString retention) + kiro projectPath for repo attribution
This commit is contained in:
commit
897020591a
8 changed files with 408 additions and 14 deletions
|
|
@ -2,6 +2,10 @@
|
|||
|
||||
## Unreleased
|
||||
|
||||
### Fixed
|
||||
- **Cold parse no longer retains full message bodies through cached previews.** `flatSlice` skipped its Buffer round-trip for strings already within the bound, but provider adapters pre-truncate user-message previews with `.slice(0, 500)` before the cache-site call — those pre-sliced views are still V8 SlicedStrings pinning their large parent, so the retention that OOM'd cold parses of large histories survived. The round-trip now always runs.
|
||||
- **Kiro sessions carry the real `projectPath`** (CLI meta.cwd, v2 `workspacePaths[0]`, workspace sessions' `workspaceDirectory`), so git-repo attribution can resolve them; previously they were attribution-blind. Bumps the kiro parse version, so the first run after upgrade re-parses kiro history once, and kiro sessions in linked git worktrees now group under the main repo.
|
||||
|
||||
## 0.9.20 - 2026-08-10
|
||||
|
||||
### Added
|
||||
|
|
|
|||
|
|
@ -59,6 +59,7 @@ The stores are disjoint (v2 sessions use `sess_`-prefixed IDs in a separate dire
|
|||
- Token counts are estimated via char count (`CHARS_PER_TOKEN = 4`).
|
||||
- **Credits are the cost source; tokens stay estimated.** Kiro bills in credits ($20/mo for 1,000; overage $0.04/credit). CLI (`metering_usage`), v1 executions (`usageSummary[].usage`), and v2 (`usage_summary.promptTurnSummaries[].usage`) turns record real credits, converted to USD at `USD_PER_KIRO_CREDIT = 0.04` (the public overage rate — the same never-understate approach as Codebuff). Turns without credit data fall back to token-estimated cost (`costIsEstimated: true`); legacy `.chat` and workspace-session records carry no usage data, so they are always token-estimated. Note: an earlier CLI implementation summed credit values directly as dollars, overstating cost 25×. Token *counts* remain char-estimated everywhere (input undercounts: only visible transcript text is seen, not the full resent context; v2's `session_metadata.contextUsage.usagePercentage` × context window is a better input proxy if ever needed). v2 does keep the real `modelId`, so unlike the v1 execution-file path it is not mislabeled `kiro-auto`.
|
||||
- **Cost is frozen at parse time.** Kiro is on the `costUSD` pass-through allowlist in `providerCallToCachedCall` (alongside mistral-vibe, devin, hermes, …), so its credit-based cost survives the session cache instead of being re-priced from estimated tokens — token re-pricing understated/overstated real kiro spend by up to 16× per model. The tradeoff, shared with all allowlisted providers: `codeburn price-override` and `model-alias` do not affect kiro dollar amounts (token *counts* are unaffected). Historical caches from before this change re-parse via the `CACHE_VERSION` bump to 5.
|
||||
- **`projectPath` for git attribution.** The parser now records the session's working directory as `projectPath` (CLI `meta.cwd`, v2 `workspacePaths[0]`, workspace sessions' `workspaceDirectory`), which sync attribution needs to resolve the git repo. The `project-path-v1` parse-version bump re-parses cached kiro history once; sessions in linked git worktrees now group under the main repo.
|
||||
|
||||
## When fixing a bug here
|
||||
|
||||
|
|
|
|||
|
|
@ -24,3 +24,35 @@ export function normalizeContentBlocks<T extends { type?: string; text?: string
|
|||
if (typeof content === 'string') return [{ type: 'text', text: content } as T]
|
||||
return []
|
||||
}
|
||||
|
||||
/// Take a bounded prefix of a string as a FLAT copy.
|
||||
///
|
||||
/// `String.prototype.slice` returns a V8 SlicedString — a view object that
|
||||
/// retains a reference to its ENTIRE parent string. Session files routinely
|
||||
/// carry 100KB+ message strings (agent-injected system prompts, tool
|
||||
/// results); storing a short `.slice()` of each in a long-lived structure
|
||||
/// (the session cache) pins every parent buffer for the life of the process.
|
||||
/// Across thousands of session files this balloons a cold parse of a few GB
|
||||
/// of JSONL into an out-of-memory crash (~5.5GB peak observed), while a warm
|
||||
/// run — whose strings were flattened by the cache's JSON round-trip — needs
|
||||
/// only ~300MB for the same data.
|
||||
///
|
||||
/// Round-tripping through a Buffer forces a fresh flat string with no parent
|
||||
/// reference. This always runs, even when `s` is already within `max`:
|
||||
/// callers may pass an already-sliced view (provider adapters pre-truncate
|
||||
/// with `.slice(0, 500)` before the cache-site call), and that view is
|
||||
/// itself a SlicedString pinning its own large parent.
|
||||
export function flatSlice(s: string, max: number): string {
|
||||
return Buffer.from(s.slice(0, max), 'utf16le').toString('utf16le')
|
||||
}
|
||||
|
||||
/// Force a FLAT copy of a string regardless of length.
|
||||
///
|
||||
/// Companion to `flatSlice` for strings that are ALREADY short but were
|
||||
/// produced as views over a large parent — regex match groups
|
||||
/// (`match[1]` retains the entire subject string) and `trim()` results
|
||||
/// both come back as V8 SlicedStrings. Use this when storing such values
|
||||
/// in long-lived structures; use `flatSlice` when also bounding length.
|
||||
export function flatString(s: string): string {
|
||||
return Buffer.from(s, 'utf16le').toString('utf16le')
|
||||
}
|
||||
|
|
|
|||
|
|
@ -4,7 +4,7 @@ import { basename, dirname, join, resolve, sep } from 'path'
|
|||
import { readSessionLines } from './fs-utils.js'
|
||||
import { calculateCost, calculateLocalModelSavings, getShortModelName, isProxiedPath, getProxyPathsConfigHash, getModelAliasesConfigHash, getPriceOverridesConfigHash, getLocalModelSavingsConfigHash } from './models.js'
|
||||
import { resolveSubagentAttribution, sessionIdentity } from './sessions-report.js'
|
||||
import { normalizeContentBlocks } from './content-utils.js'
|
||||
import { normalizeContentBlocks, flatSlice, flatString } from './content-utils.js'
|
||||
import { discoverAllSessions, getProvider } from './providers/index.js'
|
||||
import { flushCodexCache } from './codex-cache.js'
|
||||
import { antigravityCascadeIdFromPath, flushAntigravityCache, shouldReparseAntigravitySource } from './providers/antigravity.js'
|
||||
|
|
@ -91,7 +91,27 @@ function isCoworkSession(cwd: string, filePath: string): boolean {
|
|||
})
|
||||
}
|
||||
|
||||
// Memoizes resolveCanonicalProjectPath: every ParsedProviderCall with a
|
||||
// projectPath pays the .git-marker directory walk (one lstat per ancestor
|
||||
// level), and a session's calls all share one cwd — without this cache a
|
||||
// cold parse re-walks the same few directories thousands of times
|
||||
// (measured ~+5% cold-parse time for a large kiro store). Filesystem facts
|
||||
// can go stale in a long-lived process (a dir converted to a worktree
|
||||
// mid-run), so the cache is cleared with the session cache.
|
||||
// Stores the Promise, not the resolved value: callers within the same
|
||||
// Promise.all batch would otherwise all miss the cache and each re-walk the
|
||||
// filesystem before the first walk's result lands.
|
||||
const canonicalPathCache = new Map<string, Promise<{ path: string; isWorktree: boolean }>>()
|
||||
|
||||
async function resolveCanonicalProjectPath(cwd: string): Promise<{ path: string; isWorktree: boolean }> {
|
||||
const cached = canonicalPathCache.get(cwd)
|
||||
if (cached) return cached
|
||||
const result = resolveCanonicalProjectPathUncached(cwd)
|
||||
canonicalPathCache.set(cwd, result)
|
||||
return result
|
||||
}
|
||||
|
||||
async function resolveCanonicalProjectPathUncached(cwd: string): Promise<{ path: string; isWorktree: boolean }> {
|
||||
const trimmed = cwd.trim()
|
||||
if (!trimmed) return { path: cwd, isWorktree: false }
|
||||
|
||||
|
|
@ -1261,7 +1281,7 @@ export function collectToolResultMeta(entry: JournalEntry, map: Map<string, Tool
|
|||
export function collectSessionMeta(entry: JournalEntry, meta: SessionMeta): void {
|
||||
if (entry.type === 'ai-title') {
|
||||
const t = (entry as Record<string, unknown>)['aiTitle']
|
||||
if (typeof t === 'string' && t.trim()) meta.title = t.trim().slice(0, 200)
|
||||
if (typeof t === 'string' && t.trim()) meta.title = flatString(t.trim().slice(0, 200))
|
||||
} else if (entry.type === 'pr-link') {
|
||||
const url = (entry as Record<string, unknown>)['prUrl']
|
||||
if (typeof url === 'string' && url && !meta.prLinks.includes(url)) meta.prLinks.push(url)
|
||||
|
|
@ -1900,7 +1920,7 @@ export async function readAgentType(filePath: string): Promise<string | undefine
|
|||
const metaPath = filePath.replace(/\.jsonl$/, '.meta.json')
|
||||
try {
|
||||
const t = (JSON.parse(await readFile(metaPath, 'utf8')) as { agentType?: unknown }).agentType
|
||||
if (typeof t === 'string' && t.trim()) return t.trim().slice(0, 100)
|
||||
if (typeof t === 'string' && t.trim()) return flatString(t.trim().slice(0, 100))
|
||||
} catch { /* missing or unreadable meta */ }
|
||||
// Workflow agents always live under `subagents/workflows/`, so fall back to that
|
||||
// even when the meta sidecar is absent.
|
||||
|
|
@ -2453,7 +2473,7 @@ function parsedTurnToCachedTurn(turn: ParsedTurn): CachedTurn {
|
|||
return {
|
||||
timestamp: turn.timestamp,
|
||||
sessionId: turn.sessionId,
|
||||
userMessage: turn.userMessage.slice(0, 2000),
|
||||
userMessage: flatSlice(turn.userMessage, 2000),
|
||||
calls: turn.assistantCalls.map(apiCallToCachedCall),
|
||||
// Stored per-turn directly (already sorted/deduped in groupIntoTurns), unlike
|
||||
// gitBranch's change-detection dedup, so each turn's refs are self-contained.
|
||||
|
|
@ -2484,7 +2504,7 @@ function providerCallToCachedTurn(call: ParsedProviderCall): CachedTurn {
|
|||
return {
|
||||
timestamp: call.timestamp,
|
||||
sessionId: call.sessionId,
|
||||
userMessage: call.userMessage.slice(0, 2000),
|
||||
userMessage: flatSlice(call.userMessage, 2000),
|
||||
calls: [providerCallToCachedCall(call)],
|
||||
...(prRefs.length ? { prRefs } : {}),
|
||||
}
|
||||
|
|
@ -2507,7 +2527,7 @@ function providerCallsToCachedTurns(calls: ParsedProviderCall[]): CachedTurn[] {
|
|||
turn = {
|
||||
timestamp: call.timestamp,
|
||||
sessionId: call.sessionId,
|
||||
userMessage: call.userMessage.slice(0, 2000),
|
||||
userMessage: flatSlice(call.userMessage, 2000),
|
||||
calls: [],
|
||||
...(prRefs.length ? { prRefs } : {}),
|
||||
}
|
||||
|
|
@ -3253,6 +3273,7 @@ function cacheKey(dateRange?: DateRange, providerFilter?: string): string {
|
|||
|
||||
export function clearSessionCache(): void {
|
||||
sessionCache.clear()
|
||||
canonicalPathCache.clear()
|
||||
}
|
||||
|
||||
function cachePut(key: string, data: ProjectSummary[]) {
|
||||
|
|
|
|||
|
|
@ -5,6 +5,7 @@ import { basename, dirname, extname, join } from 'path'
|
|||
import { homedir } from 'os'
|
||||
|
||||
import { readSessionFile } from '../fs-utils.js'
|
||||
import { flatSlice, flatString } from '../content-utils.js'
|
||||
import { calculateCost } from '../models.js'
|
||||
import { estimateTokensFromChars } from '../token-estimate.js'
|
||||
import type { ToolCall } from '../types.js'
|
||||
|
|
@ -98,7 +99,10 @@ function extractToolNames(content: string): string[] {
|
|||
let match
|
||||
while ((match = regex.exec(content)) !== null) {
|
||||
const name = match[1]!.trim()
|
||||
tools.push(toolNameMap[name] ?? name)
|
||||
// flatString: regex match groups are V8 SlicedStrings that retain the
|
||||
// ENTIRE subject string — storing them in the session cache would pin
|
||||
// every scanned assistant-content buffer. Mapped names are flat literals.
|
||||
tools.push(toolNameMap[name] ?? flatString(name))
|
||||
}
|
||||
return tools
|
||||
}
|
||||
|
|
@ -217,7 +221,7 @@ function parseChatFile(data: KiroChatFile, sessionId: string, project: string, s
|
|||
if (msg.role === 'human') {
|
||||
if (msg.content.startsWith('<identity>')) continue
|
||||
inputChars += msg.content.length
|
||||
pendingUserMessage = msg.content.slice(0, 500)
|
||||
pendingUserMessage = flatSlice(msg.content, 500)
|
||||
}
|
||||
if (msg.role === 'bot') {
|
||||
const msgTools = extractToolNames(msg.content)
|
||||
|
|
@ -296,7 +300,7 @@ function parseModernExecution(data: KiroModernExecution, sourcePath: string, see
|
|||
|
||||
if (directInput) {
|
||||
inputChars += directInput.length
|
||||
pendingUserMessage = directInput.slice(0, 500)
|
||||
pendingUserMessage = flatSlice(directInput, 500)
|
||||
}
|
||||
|
||||
if (directOutput) {
|
||||
|
|
@ -328,7 +332,7 @@ function parseModernExecution(data: KiroModernExecution, sourcePath: string, see
|
|||
if (role === 'human' || role === 'user') {
|
||||
if (!text) continue
|
||||
inputChars += text.length
|
||||
pendingUserMessage = text.slice(0, 500)
|
||||
pendingUserMessage = flatSlice(text, 500)
|
||||
} else if (role === 'bot' || role === 'assistant' || role === 'ai' || role === 'model') {
|
||||
if (text) outputChars += text.length
|
||||
if (text || tools.length > 0) hasOutputActivity = true
|
||||
|
|
@ -506,6 +510,7 @@ function parseCliSession(meta: KiroCliSessionMeta, entries: KiroCliEntry[], seen
|
|||
userMessage: pendingUserMessage,
|
||||
sessionId,
|
||||
project,
|
||||
...(meta.cwd ? { projectPath: meta.cwd } : {}),
|
||||
})
|
||||
turnIndex++
|
||||
}
|
||||
|
|
@ -526,7 +531,7 @@ function parseCliSession(meta: KiroCliSessionMeta, entries: KiroCliEntry[], seen
|
|||
for (const item of content) {
|
||||
const rec = asRecord(item)
|
||||
if (rec && rec['kind'] === 'text' && typeof rec['data'] === 'string') {
|
||||
pendingUserMessage = (rec['data'] as string).slice(0, 500)
|
||||
pendingUserMessage = flatSlice(rec['data'] as string, 500)
|
||||
inputChars += (rec['data'] as string).length
|
||||
}
|
||||
}
|
||||
|
|
@ -605,7 +610,7 @@ async function parseWorkspaceSession(record: Record<string, unknown>, source: Se
|
|||
const text = extractText(msg['content'])
|
||||
if (role === 'user' && text) {
|
||||
inputChars += text.length
|
||||
pendingUserMessage = text.slice(0, 500)
|
||||
pendingUserMessage = flatSlice(text, 500)
|
||||
} else if (role === 'assistant' && !execBacked && text && text !== 'On it.') {
|
||||
// An item carrying an executionId is execution-backed: its content is
|
||||
// counted from the execution file, so counting it here would double-count.
|
||||
|
|
@ -662,6 +667,9 @@ async function parseWorkspaceSession(record: Record<string, unknown>, source: Se
|
|||
deduplicationKey: dedupKey,
|
||||
userMessage: pendingUserMessage,
|
||||
sessionId,
|
||||
...(typeof record['workspaceDirectory'] === 'string' && record['workspaceDirectory']
|
||||
? { projectPath: record['workspaceDirectory'] as string }
|
||||
: {}),
|
||||
})
|
||||
|
||||
return results
|
||||
|
|
@ -774,6 +782,7 @@ async function parseV2Session(source: SessionSource, seenKeys: Set<string>): Pro
|
|||
userMessage: turnUserMessage,
|
||||
sessionId,
|
||||
project: source.project,
|
||||
...(meta.workspacePaths?.[0] ? { projectPath: meta.workspacePaths[0] } : {}),
|
||||
})
|
||||
}
|
||||
}
|
||||
|
|
@ -794,7 +803,7 @@ async function parseV2Session(source: SessionSource, seenKeys: Set<string>): Pro
|
|||
// for the upcoming turn_start.
|
||||
if (inTurn) flushTurn()
|
||||
const text = typeof payload['content'] === 'string' ? payload['content'] as string : extractText(payload['content'])
|
||||
pendingUserMessage = text.slice(0, 500)
|
||||
pendingUserMessage = flatSlice(text, 500)
|
||||
pendingUserChars = text.length
|
||||
} else if (type === 'turn_start') {
|
||||
if (inTurn) flushTurn()
|
||||
|
|
|
|||
|
|
@ -269,7 +269,12 @@ export const PROVIDER_PARSE_VERSIONS: Record<string, string> = {
|
|||
hermes: 'reasoning-output-accounting-v1-est-cost',
|
||||
'lingtai-tui': 'token-ledger-registry-activity-v3',
|
||||
'ibm-bob': 'worktree-project-grouping-v1',
|
||||
kiro: 'ide-parsing-v1-est-cost',
|
||||
// project-path-v1: the parser now records the session's full working
|
||||
// directory as projectPath (CLI meta.cwd, v2 workspacePaths[0], workspace
|
||||
// sessions' workspaceDirectory), which sync attribution needs to resolve
|
||||
// the git repo. Cached entries from before the bump lack projectPath and
|
||||
// would serve attribution-blind sessions forever without a re-parse.
|
||||
kiro: 'ide-parsing-v1-est-cost-project-path-v1',
|
||||
opencode: 'session-model-v1',
|
||||
quickdesk: 'emf-sqlite-v2-est-cost',
|
||||
kimicode: 'wire-usage-v1-est-cost',
|
||||
|
|
|
|||
117
tests/flat-slice.test.ts
Normal file
117
tests/flat-slice.test.ts
Normal file
|
|
@ -0,0 +1,117 @@
|
|||
/**
|
||||
* Tests for flatSlice — the SlicedString-retention fix.
|
||||
*
|
||||
* Background: `String.prototype.slice` returns a V8 SlicedString that
|
||||
* retains its entire parent string. Storing short slices of large session
|
||||
* strings (100KB+ agent prompts) in the long-lived session cache pinned
|
||||
* gigabytes of parent buffers during cold parses, OOMing the default heap
|
||||
* (issue observed at ~5.5GB peak for 3.2GB of kiro session files; ~300MB
|
||||
* after flattening).
|
||||
*/
|
||||
|
||||
import { describe, it, expect } from 'vitest'
|
||||
|
||||
import { flatSlice, flatString } from '../src/content-utils.js'
|
||||
|
||||
describe('flatSlice', () => {
|
||||
it('returns the prefix for strings over the bound', () => {
|
||||
const big = 'x'.repeat(10_000)
|
||||
const out = flatSlice(big, 500)
|
||||
expect(out.length).toBe(500)
|
||||
expect(out).toBe(big.slice(0, 500))
|
||||
})
|
||||
|
||||
it('returns the string itself when within the bound', () => {
|
||||
const small = 'hello world'
|
||||
expect(flatSlice(small, 500)).toBe(small)
|
||||
})
|
||||
|
||||
it('handles multi-byte characters without corruption', () => {
|
||||
// Emoji + CJK near the boundary — Buffer round-trip must not produce
|
||||
// invalid UTF-8 replacement chars for chars fully inside the slice.
|
||||
const s = '🐾'.repeat(300) // each emoji is 2 UTF-16 code units
|
||||
const out = flatSlice(s, 500)
|
||||
expect(out).toBe(s.slice(0, 500))
|
||||
})
|
||||
|
||||
it('preserves a lone surrogate at a mid-pair cut', () => {
|
||||
// A cut landing between the high and low surrogate of a pair leaves a
|
||||
// lone surrogate. utf16le round-trips code units byte-for-byte, so the
|
||||
// lone surrogate survives intact (unlike utf-8, which would replace it
|
||||
// with U+FFFD).
|
||||
const s = 'ab' + '🐾'.repeat(300) // odd offset puts every emoji across even boundaries
|
||||
const out = flatSlice(s, 501) // cuts mid-pair
|
||||
expect(out.length).toBe(501)
|
||||
expect(out.slice(0, 500)).toBe(s.slice(0, 500)) // content before the cut intact
|
||||
expect(out.charCodeAt(500)).toBe(s.charCodeAt(500)) // lone surrogate preserved
|
||||
})
|
||||
|
||||
it('does not retain the parent of an already-sliced view', () => {
|
||||
// The bug this early-return removal fixes: provider adapters pre-truncate
|
||||
// with .slice(0, 500) before the cache-site flatSlice call, so a naive
|
||||
// "already within bound" early return would skip flattening and leave
|
||||
// the SlicedString pinning its 100KB parent.
|
||||
const before = process.memoryUsage().heapUsed
|
||||
const kept: string[] = []
|
||||
for (let i = 0; i < 1000; i++) {
|
||||
const parent = (i % 10).toString().repeat(100_000) + i
|
||||
const preSliced = parent.slice(0, 500)
|
||||
kept.push(flatSlice(preSliced, 2000))
|
||||
}
|
||||
if (typeof global.gc === 'function') global.gc()
|
||||
const after = process.memoryUsage().heapUsed
|
||||
const growthMB = (after - before) / 1048576
|
||||
expect(kept.length).toBe(1000)
|
||||
expect(growthMB).toBeLessThan(50)
|
||||
})
|
||||
|
||||
it('does not retain the parent string (heap growth stays bounded)', () => {
|
||||
// Property test for the retention fix: keep 1000 short prefixes of
|
||||
// 1000 distinct 100KB strings. With plain .slice() each prefix pins its
|
||||
// 100KB parent (~200MB in UTF-16 total). With flatSlice, retained data
|
||||
// is ~1000 × 500 chars ≈ 1MB. Assert heap growth is far below the
|
||||
// retention scenario. Threshold is generous (50MB) to be CI-safe while
|
||||
// still failing decisively if retention returns (>190MB). When the test
|
||||
// runner exposes gc (vitest under --expose-gc), force a collection so
|
||||
// transient parent garbage doesn't inflate the measurement.
|
||||
const before = process.memoryUsage().heapUsed
|
||||
const kept: string[] = []
|
||||
for (let i = 0; i < 1000; i++) {
|
||||
// Distinct content so V8 cannot intern/share the parents.
|
||||
const parent = (i % 10).toString().repeat(100_000)
|
||||
kept.push(flatSlice(parent + i, 500))
|
||||
}
|
||||
if (typeof global.gc === 'function') global.gc()
|
||||
const after = process.memoryUsage().heapUsed
|
||||
const growthMB = (after - before) / 1048576
|
||||
expect(kept.length).toBe(1000)
|
||||
expect(growthMB).toBeLessThan(50)
|
||||
})
|
||||
})
|
||||
|
||||
describe('flatString', () => {
|
||||
it('returns an equal string for any input', () => {
|
||||
expect(flatString('')).toBe('')
|
||||
expect(flatString('hello')).toBe('hello')
|
||||
expect(flatString('🐾 multi-byte ✓')).toBe('🐾 multi-byte ✓')
|
||||
})
|
||||
|
||||
it('does not retain the parent of a regex match group', () => {
|
||||
// match[1] is a SlicedString retaining the entire subject. flatString
|
||||
// must break that link: keep 1000 short match groups of distinct 100KB
|
||||
// subjects and assert bounded heap growth (same thresholds as the
|
||||
// flatSlice retention test).
|
||||
const before = process.memoryUsage().heapUsed
|
||||
const kept: string[] = []
|
||||
for (let i = 0; i < 1000; i++) {
|
||||
const subject = `<name>tool_${i}</name>` + (i % 10).toString().repeat(100_000)
|
||||
const m = /<name>([^<]+)<\/name>/.exec(subject)
|
||||
kept.push(flatString(m![1]!))
|
||||
}
|
||||
if (typeof global.gc === 'function') global.gc()
|
||||
const after = process.memoryUsage().heapUsed
|
||||
const growthMB = (after - before) / 1048576
|
||||
expect(kept.length).toBe(1000)
|
||||
expect(growthMB).toBeLessThan(50)
|
||||
})
|
||||
})
|
||||
205
tests/kiro-projectpath.test.ts
Normal file
205
tests/kiro-projectpath.test.ts
Normal file
|
|
@ -0,0 +1,205 @@
|
|||
/**
|
||||
* Tests for kiro projectPath emission (sync attribution support).
|
||||
*
|
||||
* The kiro provider historically reduced the session's working directory to
|
||||
* `basename(cwd)` for display and discarded the full path. Sync attribution
|
||||
* (`codeburn sync push --attribution`) needs the full path on
|
||||
* `ParsedProviderCall.projectPath` to resolve the git repo — without it,
|
||||
* every kiro session is attribution-blind.
|
||||
*
|
||||
* Also covers the cache side: projectPath is persisted via CachedCall, so
|
||||
* entries cached BEFORE the parser learned to emit it must re-parse. That is
|
||||
* driven by the PROVIDER_PARSE_VERSIONS.kiro bump (project-path-v1); a cache
|
||||
* seeded at the pre-bump fingerprint must be discarded.
|
||||
*/
|
||||
|
||||
import { mkdir, writeFile, rm } from 'node:fs/promises'
|
||||
import { createHash } from 'node:crypto'
|
||||
import { join } from 'node:path'
|
||||
|
||||
import { describe, it, expect, beforeEach, afterAll, vi } from 'vitest'
|
||||
|
||||
import { clearSessionCache, parseAllSessions } from '../src/parser.js'
|
||||
import {
|
||||
CACHE_VERSION,
|
||||
computeEnvFingerprint,
|
||||
fingerprintFile,
|
||||
sessionCachePath,
|
||||
type SessionCache,
|
||||
} from '../src/session-cache.js'
|
||||
|
||||
// The kiro provider reads homedir()/env at call time in discovery; HOME must
|
||||
// point at the test root before ../src/parser.js is evaluated (see the
|
||||
// equivalent note in kiro-cache-invalidation.test.ts).
|
||||
const testRoot = vi.hoisted(() => {
|
||||
const root = `${process.env['TMPDIR'] || '/tmp'}/kiro-projpath-${process.pid}-${Date.now()}`
|
||||
process.env['HOME'] = `${root}/home`
|
||||
process.env['USERPROFILE'] = `${root}/home`
|
||||
return root
|
||||
})
|
||||
|
||||
const HOME = join(testRoot, 'home')
|
||||
const CACHE_DIR = join(testRoot, 'cache')
|
||||
const KIRO_SESSIONS = join(HOME, '.kiro', 'sessions')
|
||||
const CLI_DIR = join(KIRO_SESSIONS, 'cli')
|
||||
|
||||
const CLI_CWD = '/local/home/testuser/workplace/my-project'
|
||||
const V2_WORKSPACE = '/local/home/testuser/workplace/ide-project'
|
||||
|
||||
beforeEach(() => {
|
||||
process.env['HOME'] = HOME
|
||||
process.env['USERPROFILE'] = HOME
|
||||
process.env['CODEBURN_CACHE_DIR'] = CACHE_DIR
|
||||
delete process.env['KIRO_HOME']
|
||||
clearSessionCache()
|
||||
})
|
||||
|
||||
afterAll(async () => {
|
||||
await rm(testRoot, { recursive: true, force: true })
|
||||
})
|
||||
|
||||
/** Write a minimal kiro CLI session: <id>.jsonl entries + companion .json meta. */
|
||||
async function seedCliSession(id: string, cwd: string): Promise<string> {
|
||||
await mkdir(CLI_DIR, { recursive: true })
|
||||
const jsonlPath = join(CLI_DIR, `${id}.jsonl`)
|
||||
const entries = [
|
||||
{ kind: 'Prompt', data: { content: [{ kind: 'text', data: 'add a feature' }] } },
|
||||
{ kind: 'AssistantMessage', data: { content: [{ kind: 'text', data: 'Done — added the feature and tests.' }] } },
|
||||
]
|
||||
await writeFile(jsonlPath, entries.map(e => JSON.stringify(e)).join('\n'))
|
||||
await writeFile(join(CLI_DIR, `${id}.json`), JSON.stringify({
|
||||
session_id: id,
|
||||
cwd,
|
||||
created_at: '2026-08-01T10:00:00Z',
|
||||
updated_at: '2026-08-01T10:05:00Z',
|
||||
session_state: {
|
||||
rts_model_state: { model_info: { model_id: 'auto' } },
|
||||
conversation_metadata: {
|
||||
user_turn_metadatas: [
|
||||
{ end_timestamp: '2026-08-01T10:05:00Z', metering_usage: [] },
|
||||
],
|
||||
},
|
||||
},
|
||||
}))
|
||||
return jsonlPath
|
||||
}
|
||||
|
||||
/** Write a minimal v2 IDE session: sessions/<hash>/sess_<id>/{session.json,messages.jsonl}. */
|
||||
async function seedV2Session(id: string, workspacePath: string): Promise<void> {
|
||||
const sessDir = join(KIRO_SESSIONS, 'f'.repeat(32), `sess_${id}`)
|
||||
await mkdir(sessDir, { recursive: true })
|
||||
await writeFile(join(sessDir, 'session.json'), JSON.stringify({
|
||||
id,
|
||||
modelId: 'auto',
|
||||
workspacePaths: [workspacePath],
|
||||
createdAt: '2026-08-01T11:00:00Z',
|
||||
}))
|
||||
const events = [
|
||||
{ timestamp: '2026-08-01T11:00:00Z', payload: { type: 'user', content: 'fix the bug' } },
|
||||
{ timestamp: '2026-08-01T11:00:01Z', payload: { type: 'turn_start', executionId: 'x1' } },
|
||||
{ timestamp: '2026-08-01T11:00:05Z', payload: { type: 'assistant', content: 'Fixed the bug in handler.ts by checking null first.' } },
|
||||
{ timestamp: '2026-08-01T11:00:06Z', payload: { type: 'turn_end', executionId: 'x1' } },
|
||||
]
|
||||
await writeFile(join(sessDir, 'messages.jsonl'), events.map(e => JSON.stringify(e)).join('\n'))
|
||||
}
|
||||
|
||||
function kiroAgentDir(): string {
|
||||
if (process.platform === 'darwin') {
|
||||
return join(HOME, 'Library', 'Application Support', 'Kiro', 'User', 'globalStorage', 'kiro.kiroagent')
|
||||
}
|
||||
if (process.platform === 'win32') {
|
||||
return join(HOME, 'AppData', 'Roaming', 'Kiro', 'User', 'globalStorage', 'kiro.kiroagent')
|
||||
}
|
||||
return join(HOME, '.config', 'Kiro', 'User', 'globalStorage', 'kiro.kiroagent')
|
||||
}
|
||||
|
||||
/** Write a minimal IDE workspace-session:
|
||||
* <agentDir>/workspace-sessions/<base64(workspacePath), '='→'_'>/<sessionId>.json */
|
||||
async function seedWorkspaceSession(id: string, workspaceDirectory: string): Promise<void> {
|
||||
const encoded = Buffer.from(workspaceDirectory, 'utf-8').toString('base64').replace(/=/g, '_')
|
||||
const dir = join(kiroAgentDir(), 'workspace-sessions', encoded)
|
||||
await mkdir(dir, { recursive: true })
|
||||
await writeFile(join(dir, `${id}.json`), JSON.stringify({
|
||||
sessionId: id,
|
||||
selectedModel: 'auto',
|
||||
workspaceDirectory,
|
||||
history: [
|
||||
{ message: { role: 'user', content: 'refactor the config loader' } },
|
||||
{ message: { role: 'assistant', content: 'Refactored the loader into three small functions with tests.' } },
|
||||
],
|
||||
}))
|
||||
}
|
||||
|
||||
async function kiroCalls() {
|
||||
const projects = await parseAllSessions(undefined, 'kiro')
|
||||
return projects.flatMap(p => p.sessions.map(s => ({ project: p.project, projectPath: p.projectPath, session: s })))
|
||||
}
|
||||
|
||||
describe('kiro projectPath emission', () => {
|
||||
it('CLI session: projectPath is the full meta.cwd, project the basename', async () => {
|
||||
await seedCliSession('cli-001', CLI_CWD)
|
||||
const rows = await kiroCalls()
|
||||
const row = rows.find(r => r.project === 'my-project')
|
||||
expect(row).toBeDefined()
|
||||
expect(row!.projectPath).toBe(CLI_CWD)
|
||||
})
|
||||
|
||||
it('v2 IDE session: projectPath is workspacePaths[0]', async () => {
|
||||
await seedV2Session('v2-001', V2_WORKSPACE)
|
||||
const rows = await kiroCalls()
|
||||
const row = rows.find(r => r.project === 'ide-project')
|
||||
expect(row).toBeDefined()
|
||||
expect(row!.projectPath).toBe(V2_WORKSPACE)
|
||||
})
|
||||
|
||||
it('workspace session: projectPath is workspaceDirectory', async () => {
|
||||
const WS_DIR = '/local/home/testuser/workplace/ws-project'
|
||||
await seedWorkspaceSession('ws-001', WS_DIR)
|
||||
const rows = await kiroCalls()
|
||||
const row = rows.find(r => r.project === 'ws-project')
|
||||
expect(row).toBeDefined()
|
||||
expect(row!.projectPath).toBe(WS_DIR)
|
||||
})
|
||||
})
|
||||
|
||||
describe('kiro projectPath cache invalidation (project-path-v1 bump)', () => {
|
||||
// The fingerprint a cache written by the PREVIOUS release carries: same env
|
||||
// vars, but the parser version before the project-path-v1 bump.
|
||||
function preBumpFingerprint(): string {
|
||||
const parts = [`KIRO_HOME=${process.env['KIRO_HOME'] ?? ''}`, 'parser=ide-parsing-v1-est-cost']
|
||||
return createHash('sha256').update(parts.join('\0')).digest('hex').slice(0, 16)
|
||||
}
|
||||
|
||||
it('the bump changed the env fingerprint', () => {
|
||||
expect(computeEnvFingerprint('kiro')).not.toBe(preBumpFingerprint())
|
||||
})
|
||||
|
||||
it('a pre-bump cache entry (no projectPath) is re-parsed and gains projectPath', async () => {
|
||||
const jsonlPath = await seedCliSession('cli-002', CLI_CWD)
|
||||
|
||||
// Seed a cache exactly as the pre-bump release would have left it:
|
||||
// correct file fingerprint, pre-bump env fingerprint, turns WITHOUT
|
||||
// projectPath on the cached calls.
|
||||
const fp = await fingerprintFile(jsonlPath)
|
||||
if (!fp) throw new Error('failed to fingerprint seeded session file')
|
||||
const cache: SessionCache = {
|
||||
version: CACHE_VERSION,
|
||||
providers: {
|
||||
kiro: {
|
||||
envFingerprint: preBumpFingerprint(),
|
||||
files: {
|
||||
[jsonlPath]: { fingerprint: fp, mcpInventory: [], turns: [] },
|
||||
},
|
||||
},
|
||||
},
|
||||
}
|
||||
await mkdir(CACHE_DIR, { recursive: true })
|
||||
await writeFile(sessionCachePath(), JSON.stringify(cache))
|
||||
clearSessionCache()
|
||||
|
||||
const rows = await kiroCalls()
|
||||
const row = rows.find(r => r.project === 'my-project')
|
||||
expect(row).toBeDefined()
|
||||
expect(row!.projectPath).toBe(CLI_CWD)
|
||||
})
|
||||
})
|
||||
Loading…
Add table
Add a link
Reference in a new issue