normalize.spec.ts 18 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448
  1. import { describe, expect, it } from 'vitest'
  2. import {
  3. type NormalizeContext,
  4. normalizeSessionLog,
  5. normalizeStdout,
  6. scrubRequestHeaders,
  7. scrubSystemPrompts,
  8. scrubToolSchemas,
  9. } from '../src/normalize.ts'
  10. /**
  11. * Unit tests for the pure snapshot normalizers. Live as a *.spec.ts (runs in
  12. * the default unit gate) and import the normalizers directly.
  13. */
  14. const ctx: NormalizeContext = {
  15. sessionIds: ['11111111-2222-3333-4444-555555555555'],
  16. cwd: '/tmp/acp-snap-cwd-abc123',
  17. }
  18. describe('normalizeStdout', () => {
  19. it('rewrites JSON-RPC ids to a stable first-seen sequence', () => {
  20. const raw = [
  21. JSON.stringify({ jsonrpc: '2.0', id: 42, method: 'initialize' }),
  22. JSON.stringify({ jsonrpc: '2.0', id: 42, result: {} }),
  23. JSON.stringify({ jsonrpc: '2.0', id: 99, method: 'session/new' }),
  24. ].join('\n')
  25. const out = normalizeStdout(raw, ctx)
  26. expect(out).toContain('"id":1')
  27. expect(out).toContain('"id":2')
  28. expect(out).not.toContain('42')
  29. expect(out).not.toContain('99')
  30. })
  31. it('scrubs the cwd and session id anywhere they appear', () => {
  32. const raw = JSON.stringify({
  33. jsonrpc: '2.0', method: 'session/update',
  34. params: { sessionId: ctx.sessionIds[0], cwd: ctx.cwd, note: `at ${ctx.cwd}/x` },
  35. })
  36. const out = normalizeStdout(raw, ctx)
  37. expect(out).toContain('{{sessionId}}')
  38. expect(out).toContain('{{cwd}}')
  39. expect(out).not.toContain(ctx.cwd)
  40. expect(out).not.toContain(ctx.sessionIds[0] as string)
  41. })
  42. it('canonicalizes only cwd-rooted path separators', () => {
  43. const windowsCtx: NormalizeContext = {
  44. sessionIds: [],
  45. cwd: String.raw`C:\Users\runner\AppData\Local\Temp\acp-snapshot`,
  46. }
  47. const raw = JSON.stringify({
  48. jsonrpc: '2.0',
  49. method: 'session/update',
  50. params: {
  51. path: `${windowsCtx.cwd}\\nested\\proof.txt`,
  52. regex: String.raw`\d+\w+`,
  53. command: String.raw`printf "\\n"`,
  54. },
  55. })
  56. const frame = JSON.parse(normalizeStdout(raw, windowsCtx)) as {
  57. params: { path: string; regex: string; command: string }
  58. }
  59. expect(frame.params).toEqual({
  60. path: '{{cwd}}/nested/proof.txt',
  61. regex: String.raw`\d+\w+`,
  62. command: String.raw`printf "\\n"`,
  63. })
  64. })
  65. it('canonicalizes generated relative path fields and text markers without rewriting other text', () => {
  66. const raw = JSON.stringify({
  67. path: String.raw`nested\AGENTS.md`,
  68. content: String.raw`<path>.\nested\task.txt</path>
  69. Additional instructions from: nested\AGENTS.md`,
  70. regex: String.raw`\d+\w+`,
  71. })
  72. const frame = JSON.parse(normalizeStdout(raw, { sessionIds: [], cwd: '/unused' })) as {
  73. path: string
  74. content: string
  75. regex: string
  76. }
  77. expect(frame).toEqual({
  78. path: 'nested/AGENTS.md',
  79. content: '<path>./nested/task.txt</path>\nAdditional instructions from: nested/AGENTS.md',
  80. regex: String.raw`\d+\w+`,
  81. })
  82. })
  83. it('can preserve native cwd-rooted separators for a platform golden', () => {
  84. const windowsCtx: NormalizeContext = { sessionIds: [], cwd: String.raw`C:\work\snapshot` }
  85. const raw = JSON.stringify({ path: `${windowsCtx.cwd}\\nested\\proof.txt` })
  86. const frame = JSON.parse(normalizeStdout(raw, windowsCtx, { cwdPathMode: 'native' })) as { path: string }
  87. expect(frame.path).toBe(String.raw`{{cwd}}\nested\proof.txt`)
  88. })
  89. it('scrubs a stray UUID not in the known list', () => {
  90. const raw = JSON.stringify({ jsonrpc: '2.0', method: 'x', params: { id: 'aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee' } })
  91. expect(normalizeStdout(raw, ctx)).toContain('{{sessionId}}')
  92. })
  93. it('leaves notification frames without an id untouched in id-space', () => {
  94. const raw = JSON.stringify({ jsonrpc: '2.0', method: 'session/update', params: {} })
  95. const out = normalizeStdout(raw, ctx)
  96. expect(out).not.toContain('"id"')
  97. })
  98. it('stabilizes the timestamp carried by session title updates', () => {
  99. const raw = JSON.stringify({
  100. jsonrpc: '2.0',
  101. method: 'session/update',
  102. params: {
  103. sessionId: ctx.sessionIds[0],
  104. update: {
  105. sessionUpdate: 'session_info_update',
  106. title: 'Stable title',
  107. updatedAt: '2026-07-20T17:03:13.689Z',
  108. },
  109. },
  110. })
  111. const out = normalizeStdout(raw, ctx)
  112. expect(out).toContain('"updatedAt":"{{updatedAt}}"')
  113. expect(out).not.toContain('2026-07-20T17:03:13.689Z')
  114. })
  115. it('throws on a non-JSON stdout line (the purity check)', () => {
  116. const raw = `${JSON.stringify({ jsonrpc: '2.0', id: 1 })}\noops a log leaked\n`
  117. expect(() => normalizeStdout(raw, ctx)).toThrow()
  118. })
  119. it('ignores blank lines', () => {
  120. const raw = `\n${JSON.stringify({ jsonrpc: '2.0', id: 1, method: 'm' })}\n\n`
  121. expect(() => normalizeStdout(raw, ctx)).not.toThrow()
  122. })
  123. })
  124. describe('normalizeSessionLog', () => {
  125. const header = (over: object) => JSON.stringify({ type: 'session', version: 0, id: 's', createdAt: 123, ...over })
  126. const event = (over: object) => JSON.stringify({ type: 'turn/start', seq: 1, time: 999, data: { turn: 1 }, ...over })
  127. it('zeroes the header createdAt', () => {
  128. const out = normalizeSessionLog(`${header({})}\n`, ctx)
  129. expect(out).toContain('"createdAt":0')
  130. expect(out).not.toContain('123')
  131. })
  132. it('zeroes each event time but keeps seq', () => {
  133. const out = normalizeSessionLog(`${header({})}\n${event({ seq: 7, time: 999 })}\n`, ctx)
  134. expect(out).toContain('"time":0')
  135. expect(out).toContain('"seq":7') // seq is deterministic — NOT scrubbed
  136. expect(out).not.toContain('999')
  137. })
  138. it('scrubs cwd and session id deep inside event data', () => {
  139. const ev = JSON.stringify({
  140. type: 'tool/result', seq: 2, time: 5,
  141. data: { content: [{ type: 'text', text: `wrote ${ctx.cwd}/proof.txt` }] },
  142. })
  143. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  144. expect(out).toContain('{{cwd}}')
  145. expect(out).not.toContain(ctx.cwd)
  146. })
  147. it('scrubs random local spill paths under the snapshot cwd', () => {
  148. const ev = JSON.stringify({
  149. type: 'tool/result', seq: 2, time: 5,
  150. data: {
  151. content: [{
  152. type: 'text',
  153. text: `Full formatted result stored at: ${ctx.cwd}/.spill/session-c22bc3f1d2af/8a7b6c5d4e3f-bash.txt. Use read with offset/limit, or grep this path to search within it.`,
  154. }],
  155. },
  156. })
  157. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  158. expect(out).toContain('{{spillLocator:bash.txt}}')
  159. expect(out).not.toContain('session-c22bc3f1d2af')
  160. expect(out).not.toContain('8a7b6c5d4e3f')
  161. })
  162. it('scrubs macOS /private aliases for local spill paths', () => {
  163. const ev = JSON.stringify({
  164. type: 'tool/result', seq: 2, time: 5,
  165. data: {
  166. content: [{
  167. type: 'text',
  168. text: `Full formatted result stored at: /private${ctx.cwd}/.spill/session-c22bc3f1d2af/8a7b6c5d4e3f-bash.txt. Use read with offset/limit, or grep this path to search within it.`,
  169. }],
  170. },
  171. })
  172. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  173. expect(out).toContain('{{spillLocator:bash.txt}}')
  174. expect(out).not.toContain('/private{{spillLocator')
  175. })
  176. it('scrubs macOS /private prefix on cwd-rooted fs tool result paths', () => {
  177. const ev = JSON.stringify({
  178. type: 'tool/result', seq: 2, time: 5,
  179. data: {
  180. content: [{
  181. type: 'text',
  182. text: `The file /private${ctx.cwd}/config.txt has been updated successfully.`,
  183. }],
  184. },
  185. })
  186. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  187. expect(out).toContain('{{cwd}}/config.txt')
  188. expect(out).not.toContain('/private{{cwd}}')
  189. })
  190. it('scrubs fixed snapshot spill paths', () => {
  191. const ev = JSON.stringify({
  192. type: 'tool/result', seq: 2, time: 5,
  193. data: {
  194. content: [{
  195. type: 'text',
  196. text: 'Full formatted result stored at: /tmp/dsh-acp-snapshot-spill/session-c22bc3f1d2af/8a7b6c5d4e3f-bash.txt. Use read with offset/limit, or grep this path to search within it.',
  197. }],
  198. },
  199. })
  200. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  201. expect(out).toContain('{{spillLocator:bash.txt}}')
  202. expect(out).not.toContain('/tmp/dsh-acp-snapshot-spill')
  203. })
  204. it('scrubs scenario-owned snapshot spill paths', () => {
  205. const ev = JSON.stringify({
  206. type: 'tool/result', seq: 2, time: 5,
  207. data: {
  208. content: [{
  209. type: 'text',
  210. text: 'Full formatted result stored at: /tmp/dsh-acp-snap-012345678/session-c22bc3f1d2af/8a7b6c5d4e3f-bash.txt. Use read with offset/limit, or grep this path to search within it.',
  211. }],
  212. },
  213. })
  214. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  215. expect(out).toContain('{{spillLocator:bash.txt}}')
  216. expect(out).not.toContain('/tmp/dsh-acp-snap-012345678')
  217. })
  218. it('scrubs scenario-owned snapshot spill paths with Windows drive and separators', () => {
  219. const ev = JSON.stringify({
  220. type: 'tool/result', seq: 2, time: 5,
  221. data: {
  222. content: [{
  223. type: 'text',
  224. text: String.raw`Full formatted result stored at: C:\t\dsh-acp-snap-012345678\session-c22bc3f1d2af\8a7b6c5d4e3f-bash.txt. Use read with offset/limit, or grep this path to search within it.`,
  225. }],
  226. },
  227. })
  228. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  229. expect(out).toContain('{{spillLocator:bash.txt}}')
  230. expect(out).not.toContain('C:\\t\\dsh-acp-snap-012345678')
  231. })
  232. it('shares cwd-rooted path handling with stdout normalization', () => {
  233. const windowsCtx: NormalizeContext = { sessionIds: [], cwd: String.raw`C:\work\snapshot` }
  234. const ev = JSON.stringify({
  235. type: 'tool/result', seq: 2, time: 5,
  236. data: { path: `${windowsCtx.cwd}\\nested\\proof.txt` },
  237. })
  238. expect(normalizeSessionLog(`${header({ cwd: windowsCtx.cwd })}\n${ev}\n`, windowsCtx))
  239. .toContain('{{cwd}}/nested/proof.txt')
  240. expect(normalizeSessionLog(`${header({ cwd: windowsCtx.cwd })}\n${ev}\n`, windowsCtx, { cwdPathMode: 'native' }))
  241. .toContain(String.raw`{{cwd}}\\nested\\proof.txt`)
  242. })
  243. it('scrubs the session id in the header', () => {
  244. const out = normalizeSessionLog(`${header({ id: ctx.sessionIds[0] })}\n`, ctx)
  245. expect(out).toContain('{{sessionId}}')
  246. })
  247. it('zeroes a hook/result durationMs (run-to-run noise) but keeps its decision', () => {
  248. const ev = JSON.stringify({
  249. type: 'hook/result', seq: 2, time: 5,
  250. data: { turn: 1, point: 'UserPromptSubmit', handlerId: 'h', decision: 'block', exitCode: 2, durationMs: 37 },
  251. })
  252. const out = normalizeSessionLog(`${header({})}\n${ev}\n`, ctx)
  253. expect(out).toContain('"durationMs":0')
  254. expect(out).not.toContain('37')
  255. expect(out).toContain('"decision":"block"') // the decision is the behavior — kept
  256. })
  257. it('zeroes a packed chunk row\'s time0 and dt gaps but keeps seq0 and payload', () => {
  258. const row = JSON.stringify({
  259. type: 'text-chunks', seq0: 7, time0: 999,
  260. data: { turn: 1, step: 1, index: 0, dt: [212, 27, 0], texts: ['a', 'b', 'c', 'd'] },
  261. })
  262. const out = normalizeSessionLog(`${header({})}\n${row}\n`, ctx)
  263. expect(out).toContain('"time0":0')
  264. expect(out).toContain('"dt":[0,0,0]')
  265. expect(out).toContain('"seq0":7') // seq0 is deterministic, like seq — NOT scrubbed
  266. expect(out).toContain('"texts":["a","b","c","d"]')
  267. expect(out).not.toContain('999')
  268. expect(out).not.toContain('212')
  269. })
  270. it('zeroes time0 even when a malformed row carries no dt array', () => {
  271. const row = JSON.stringify({ type: 'text-chunks', seq0: 1, time0: 999, data: 'not-an-object' })
  272. const out = normalizeSessionLog(`${header({})}\n${row}\n`, ctx)
  273. expect(out).toContain('"time0":0')
  274. expect(out).not.toContain('999')
  275. })
  276. it('leaves a non-hook event durationMs untouched (only hook/result is scrubbed)', () => {
  277. const ev = JSON.stringify({ type: 'tool/result', seq: 2, time: 5, data: { durationMs: 88 } })
  278. const out = normalizeSessionLog(`${header({})}\n${ev}\n`, ctx)
  279. expect(out).toContain('"durationMs":88')
  280. })
  281. it('tolerates records missing the volatile fields it would zero', () => {
  282. const bareHeader = JSON.stringify({ type: 'session', id: 's' })
  283. const timeless = JSON.stringify({ type: 'note', seq: 1 })
  284. const bareHook = JSON.stringify({ type: 'hook/result', seq: 2, time: 5, data: { decision: 'allow' } })
  285. const nullDataHook = JSON.stringify({ type: 'hook/result', seq: 3, time: 6, data: null })
  286. const out = normalizeSessionLog(`${bareHeader}\n${timeless}\n${bareHook}\n${nullDataHook}\n`, ctx)
  287. expect(out).toContain('"type":"note","seq":1')
  288. expect(out).toContain('"decision":"allow"')
  289. expect(out).not.toContain('durationMs')
  290. })
  291. })
  292. describe('scrubRequestHeaders', () => {
  293. const headerLine = JSON.stringify({ type: 'session', version: 0, id: 's', createdAt: 1, cwd: '/w' })
  294. const headerEvent = (header: object) =>
  295. JSON.stringify({ type: 'request/header', seq: 3, time: 9, data: { header, reason: 'initial' } })
  296. it('replaces header system and tools with tokens, keeping config and reason', () => {
  297. const ev = headerEvent({
  298. config: { model: 'm' },
  299. system: 'You are an agent.\nBe brief.',
  300. tools: [{ name: 'read', description: 'Read a file.', parameters: { type: 'object' } }],
  301. })
  302. const out = scrubRequestHeaders(`${headerLine}\n${ev}\n`)
  303. expect(out).toContain('"system":"{{system}}"')
  304. expect(out).toContain('"tools":"{{tools}}"')
  305. expect(out).toContain('"config":{"model":"m"}')
  306. expect(out).toContain('"reason":"initial"')
  307. expect(out).not.toContain('You are an agent')
  308. expect(out).not.toContain('Read a file')
  309. })
  310. it('keeps an absent system/tools absent (presence is behavior)', () => {
  311. const out = scrubRequestHeaders(`${headerLine}\n${headerEvent({ config: { model: 'm' } })}\n`)
  312. expect(out).not.toContain('{{system}}')
  313. expect(out).not.toContain('{{tools}}')
  314. })
  315. it('scrubs a header carrying only one of system/tools, leaving the other absent', () => {
  316. const systemOnly = scrubRequestHeaders(`${headerLine}\n${headerEvent({ system: 'secret prompt' })}\n`)
  317. expect(systemOnly).toContain('"system":"{{system}}"')
  318. expect(systemOnly).not.toContain('{{tools}}')
  319. const toolsOnly = scrubRequestHeaders(`${headerLine}\n${headerEvent({ tools: [{ name: 't' }] })}\n`)
  320. expect(toolsOnly).toContain('"tools":"{{tools}}"')
  321. expect(toolsOnly).not.toContain('{{system}}')
  322. })
  323. it('leaves malformed headers with no scrubbable payload byte-identical', () => {
  324. const headerless = JSON.stringify({ type: 'request/header', seq: 10, time: 9, data: { reason: 'initial' } })
  325. const nullData = JSON.stringify({ type: 'request/header', seq: 11, time: 9, data: null })
  326. const raw = `${headerLine}\n${headerless}\n${nullData}\n`
  327. expect(scrubRequestHeaders(raw)).toBe(raw)
  328. })
  329. it('passes every other line through byte-for-byte and is idempotent', () => {
  330. const other = JSON.stringify({ type: 'assistant/chunk', seq: 4, time: 9, data: { turn: 1, step: 1, chunk: { type: 'text-delta', index: 0, text: 'hi' } } })
  331. const raw = `${headerLine}\n${headerEvent({ config: { model: 'm' }, system: 's', tools: [] })}\n${other}\n`
  332. const once = scrubRequestHeaders(raw)
  333. expect(once.split('\n')[0]).toBe(headerLine)
  334. expect(once.split('\n')[2]).toBe(other)
  335. expect(scrubRequestHeaders(once)).toBe(once)
  336. })
  337. })
  338. describe('scrubSystemPrompts', () => {
  339. it('scrubs only system prompt payloads while keeping tools verbatim', () => {
  340. const header = JSON.stringify({
  341. type: 'request/header', seq: 1, time: 2,
  342. data: {
  343. header: {
  344. system: 'full prompt',
  345. tools: [{ name: 'read', description: 'full schema' }],
  346. },
  347. reason: 'initial',
  348. },
  349. })
  350. const changed = JSON.stringify({
  351. type: 'request/header', seq: 2, time: 3,
  352. data: {
  353. header: {
  354. system: 'new prompt',
  355. tools: [{ name: 'read', description: 'changed schema' }],
  356. },
  357. reason: 'change',
  358. },
  359. })
  360. const toolsOnly = JSON.stringify({
  361. type: 'request/header', seq: 3, time: 4,
  362. data: { header: { tools: [{ name: 'read', description: 'schema only' }] }, reason: 'resume' },
  363. })
  364. const out = scrubSystemPrompts(`${header}\n${changed}\n${toolsOnly}\n`)
  365. expect(out).toContain('"system":"{{system}}"')
  366. expect(out).not.toContain('full prompt')
  367. expect(out).not.toContain('new prompt')
  368. expect(out).toContain('full schema')
  369. expect(out).toContain('changed schema')
  370. expect(out.split('\n')[2]).toBe(toolsOnly)
  371. expect(scrubSystemPrompts(out)).toBe(out)
  372. })
  373. })
  374. describe('scrubToolSchemas', () => {
  375. it('scrubs only tool-schema payloads while keeping prompts verbatim', () => {
  376. const header = JSON.stringify({
  377. type: 'request/header', seq: 1, time: 2,
  378. data: {
  379. header: {
  380. system: 'full prompt',
  381. tools: [{ name: 'read', description: 'full schema', parameters: { type: 'object' } }],
  382. },
  383. reason: 'initial',
  384. },
  385. })
  386. const changed = JSON.stringify({
  387. type: 'request/header', seq: 2, time: 3,
  388. data: {
  389. header: {
  390. system: 'new prompt',
  391. tools: [{ name: 'grep', description: 'new schema' }],
  392. },
  393. reason: 'change',
  394. },
  395. })
  396. const systemOnly = JSON.stringify({
  397. type: 'request/header', seq: 3, time: 4,
  398. data: { header: { system: 'prompt only' }, reason: 'resume' },
  399. })
  400. const out = scrubToolSchemas(`${header}\n${changed}\n${systemOnly}\n`)
  401. expect(out.match(/"tools":"{{tools}}"/g)).toHaveLength(2)
  402. expect(out).not.toContain('full schema')
  403. expect(out).not.toContain('new schema')
  404. expect(out).toContain('full prompt')
  405. expect(out).toContain('new prompt')
  406. expect(out.split('\n')[2]).toBe(systemOnly)
  407. expect(scrubToolSchemas(out)).toBe(out)
  408. })
  409. })