normalize.spec.ts 25 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642
  1. import { describe, expect, it } from 'vitest'
  2. import {
  3. type NormalizeContext,
  4. extractSnapshotSpillPaths,
  5. normalizeSessionLog,
  6. normalizeStdout,
  7. scrubRequestHeaders,
  8. scrubSystemPrompts,
  9. scrubToolSchemas,
  10. tokenizeSessionFixtureCwd,
  11. } from '../src/normalize.ts'
  12. /**
  13. * Unit tests for the pure snapshot normalizers. Live as a *.spec.ts (runs in
  14. * the default unit gate) and import the normalizers directly.
  15. */
  16. const ctx: NormalizeContext = {
  17. sessionIds: ['11111111-2222-3333-4444-555555555555'],
  18. cwd: '/tmp/acp-snap-cwd-abc123',
  19. }
  20. describe('normalizeStdout', () => {
  21. it('rewrites JSON-RPC ids to a stable first-seen sequence', () => {
  22. const raw = [
  23. JSON.stringify({ jsonrpc: '2.0', id: 42, method: 'initialize' }),
  24. JSON.stringify({ jsonrpc: '2.0', id: 42, result: {} }),
  25. JSON.stringify({ jsonrpc: '2.0', id: 99, method: 'session/new' }),
  26. ].join('\n')
  27. const out = normalizeStdout(raw, ctx)
  28. expect(out).toContain('"id":1')
  29. expect(out).toContain('"id":2')
  30. expect(out).not.toContain('42')
  31. expect(out).not.toContain('99')
  32. })
  33. it('scrubs the cwd and session id anywhere they appear', () => {
  34. const raw = JSON.stringify({
  35. jsonrpc: '2.0', method: 'session/update',
  36. params: { sessionId: ctx.sessionIds[0], cwd: ctx.cwd, note: `at ${ctx.cwd}/x` },
  37. })
  38. const out = normalizeStdout(raw, ctx)
  39. expect(out).toContain('{{sessionId}}')
  40. expect(out).toContain('{{cwd}}')
  41. expect(out).not.toContain(ctx.cwd)
  42. expect(out).not.toContain(ctx.sessionIds[0] as string)
  43. })
  44. it('scrubs cwd at file URI and chained-punctuation boundaries', () => {
  45. const raw = JSON.stringify({
  46. jsonrpc: '2.0',
  47. method: 'session/update',
  48. params: {
  49. uri: `file://${ctx.cwd}/proof.txt`,
  50. punctuated: `${ctx.cwd}.,`,
  51. dottedSegment: `${ctx.cwd}.backup`,
  52. dashedSegment: `${ctx.cwd}-backup`,
  53. },
  54. })
  55. const frame = JSON.parse(normalizeStdout(raw, ctx)) as {
  56. params: Record<string, string>
  57. }
  58. expect(frame.params).toEqual({
  59. uri: 'file://{{cwd}}/proof.txt',
  60. punctuated: '{{cwd}}.,',
  61. dottedSegment: `${ctx.cwd}.backup`,
  62. dashedSegment: `${ctx.cwd}-backup`,
  63. })
  64. })
  65. it('scrubs every filesystem spelling of the cwd longest-first', () => {
  66. const longCwd = String.raw`C:\Users\runneradmin\AppData\Local\Temp\acp-snapshot`
  67. const aliasedCtx: NormalizeContext = {
  68. sessionIds: [],
  69. cwd: String.raw`C:\Users\RUNNER~1\AppData\Local\Temp\acp-snapshot`,
  70. cwdAliases: [
  71. longCwd,
  72. String.raw`C:\Users\runneradmin\AppData\Local\Temp\acp`,
  73. ],
  74. }
  75. const raw = JSON.stringify({
  76. cwd: longCwd,
  77. path: `${longCwd}\\nested\\proof.txt`,
  78. })
  79. const frame = JSON.parse(normalizeStdout(raw, aliasedCtx)) as { cwd: string; path: string }
  80. expect(frame).toEqual({ cwd: '{{cwd}}', path: '{{cwd}}/nested/proof.txt' })
  81. })
  82. it('canonicalizes only cwd-rooted path separators', () => {
  83. const windowsCtx: NormalizeContext = {
  84. sessionIds: [],
  85. cwd: String.raw`C:\Users\runner\AppData\Local\Temp\acp-snapshot`,
  86. }
  87. const raw = JSON.stringify({
  88. jsonrpc: '2.0',
  89. method: 'session/update',
  90. params: {
  91. path: `${windowsCtx.cwd}\\nested\\proof.txt`,
  92. regex: String.raw`\d+\w+`,
  93. command: String.raw`printf "\\n"`,
  94. },
  95. })
  96. const frame = JSON.parse(normalizeStdout(raw, windowsCtx)) as {
  97. params: { path: string; regex: string; command: string }
  98. }
  99. expect(frame.params).toEqual({
  100. path: '{{cwd}}/nested/proof.txt',
  101. regex: String.raw`\d+\w+`,
  102. command: String.raw`printf "\\n"`,
  103. })
  104. })
  105. it('canonicalizes generated relative path fields and text markers without rewriting other text', () => {
  106. const raw = JSON.stringify({
  107. path: String.raw`nested\AGENTS.md`,
  108. content: String.raw`<path>.\nested\task.txt</path>
  109. Additional instructions from: nested\AGENTS.md`,
  110. regex: String.raw`\d+\w+`,
  111. })
  112. const frame = JSON.parse(normalizeStdout(raw, { sessionIds: [], cwd: '/unused' })) as {
  113. path: string
  114. content: string
  115. regex: string
  116. }
  117. expect(frame).toEqual({
  118. path: 'nested/AGENTS.md',
  119. content: '<path>./nested/task.txt</path>\nAdditional instructions from: nested/AGENTS.md',
  120. regex: String.raw`\d+\w+`,
  121. })
  122. })
  123. it('can preserve native cwd-rooted separators for a platform golden', () => {
  124. const windowsCtx: NormalizeContext = { sessionIds: [], cwd: String.raw`C:\work\snapshot` }
  125. const raw = JSON.stringify({ path: `${windowsCtx.cwd}\\nested\\proof.txt` })
  126. const frame = JSON.parse(normalizeStdout(raw, windowsCtx, { cwdPathMode: 'native' })) as { path: string }
  127. expect(frame.path).toBe(String.raw`{{cwd}}\nested\proof.txt`)
  128. })
  129. it('scrubs a stray UUID not in the known list', () => {
  130. const raw = JSON.stringify({ jsonrpc: '2.0', method: 'x', params: { id: 'aaaaaaaa-bbbb-cccc-dddd-eeeeeeeeeeee' } })
  131. expect(normalizeStdout(raw, ctx)).toContain('{{sessionId}}')
  132. })
  133. it('leaves notification frames without an id untouched in id-space', () => {
  134. const raw = JSON.stringify({ jsonrpc: '2.0', method: 'session/update', params: {} })
  135. const out = normalizeStdout(raw, ctx)
  136. expect(out).not.toContain('"id"')
  137. })
  138. it('stabilizes only the top-level event timestamp and spill byte count in event-read text', () => {
  139. const raw = JSON.stringify({
  140. jsonrpc: '2.0',
  141. method: 'session/update',
  142. params: {
  143. update: {
  144. sessionUpdate: 'tool_call_update',
  145. content: [{
  146. type: 'content',
  147. content: {
  148. type: 'text',
  149. text: 'Session prior — title\nTarget event seq 4:\n```json\n{\n "seq": 4,\n "time": 1784876275593,\n "data": {\n "time": 31337,\n "note": "model-visible"\n }\n}\n```\n\nAfter:\n "time": 424242,\n neighbor semantic text\n\n(Omitted 39387 bytes. Full formatted result stored at: /tmp/result.txt.)',
  150. },
  151. }],
  152. },
  153. },
  154. })
  155. const out = normalizeStdout(raw, ctx)
  156. expect(out).toContain('\\"time\\": {{eventTime}}')
  157. expect(out).toContain('\\"time\\": 31337')
  158. expect(out).toContain('\\"time\\": 424242')
  159. expect(out).toContain('Omitted {{eventOmittedBytes}} bytes')
  160. expect(out).not.toContain('1784876275593')
  161. expect(out).not.toContain('39387')
  162. })
  163. it('preserves event-like timestamps in unrelated output text', () => {
  164. const raw = JSON.stringify({
  165. jsonrpc: '2.0',
  166. method: 'session/update',
  167. params: {
  168. update: {
  169. sessionUpdate: 'tool_call_update',
  170. content: [{
  171. type: 'content',
  172. content: {
  173. type: 'text',
  174. text: 'bash output:\n```json\n{\n "time": 1784876275593,\n "data": {}\n}\n```\n\n(Omitted 39387 bytes. Full formatted result stored at: /tmp/result.txt.)',
  175. },
  176. }],
  177. },
  178. },
  179. })
  180. const out = normalizeStdout(raw, ctx)
  181. expect(out).toContain('1784876275593')
  182. expect(out).toContain('39387')
  183. expect(out).not.toContain('{{eventTime}}')
  184. expect(out).not.toContain('{{eventOmittedBytes}}')
  185. })
  186. it('throws on a non-JSON stdout line (the purity check)', () => {
  187. const raw = `${JSON.stringify({ jsonrpc: '2.0', id: 1 })}\noops a log leaked\n`
  188. expect(() => normalizeStdout(raw, ctx)).toThrow()
  189. })
  190. it('ignores blank lines', () => {
  191. const raw = `\n${JSON.stringify({ jsonrpc: '2.0', id: 1, method: 'm' })}\n\n`
  192. expect(() => normalizeStdout(raw, ctx)).not.toThrow()
  193. })
  194. })
  195. describe('normalizeSessionLog', () => {
  196. const header = (over: object) => JSON.stringify({ type: 'session', version: 0, id: 's', createdAt: 123, ...over })
  197. const event = (over: object) => JSON.stringify({ type: 'turn/start', seq: 1, time: 999, data: { turn: 1 }, ...over })
  198. it('zeroes the header createdAt', () => {
  199. const out = normalizeSessionLog(`${header({})}\n`, ctx)
  200. expect(out).toContain('"createdAt":0')
  201. expect(out).not.toContain('123')
  202. })
  203. it('zeroes each event time but keeps seq', () => {
  204. const out = normalizeSessionLog(`${header({})}\n${event({ seq: 7, time: 999 })}\n`, ctx)
  205. expect(out).toContain('"time":0')
  206. expect(out).toContain('"seq":7') // seq is deterministic — NOT scrubbed
  207. expect(out).not.toContain('999')
  208. })
  209. it('scrubs cwd and session id deep inside event data', () => {
  210. const ev = JSON.stringify({
  211. type: 'tool/result', seq: 2, time: 5,
  212. data: { content: [{ type: 'text', text: `wrote ${ctx.cwd}/proof.txt` }] },
  213. })
  214. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  215. expect(out).toContain('{{cwd}}')
  216. expect(out).not.toContain(ctx.cwd)
  217. })
  218. it('scrubs cwd at file URI and chained-punctuation boundaries in event data', () => {
  219. const ev = JSON.stringify({
  220. type: 'tool/result',
  221. seq: 2,
  222. time: 5,
  223. data: {
  224. uri: `file://${ctx.cwd}/proof.txt`,
  225. punctuated: `${ctx.cwd}.,`,
  226. },
  227. })
  228. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  229. expect(out).toContain('file://{{cwd}}/proof.txt')
  230. expect(out).toContain('{{cwd}}.,')
  231. expect(out).not.toContain(`file://${ctx.cwd}`)
  232. })
  233. it('scrubs random local spill paths under the snapshot cwd', () => {
  234. const ev = JSON.stringify({
  235. type: 'tool/result', seq: 2, time: 5,
  236. data: {
  237. content: [{
  238. type: 'text',
  239. text: `Full formatted result stored at: ${ctx.cwd}/.spill/session-c22bc3f1d2af/8a7b6c5d4e3f-bash.txt. Use read with offset/limit, or grep this path to search within it.`,
  240. }],
  241. },
  242. })
  243. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  244. expect(out).toContain('{{spillLocator:bash.txt}}')
  245. expect(out).not.toContain('session-c22bc3f1d2af')
  246. expect(out).not.toContain('8a7b6c5d4e3f')
  247. })
  248. it('scrubs macOS /private aliases for local spill paths', () => {
  249. const ev = JSON.stringify({
  250. type: 'tool/result', seq: 2, time: 5,
  251. data: {
  252. content: [{
  253. type: 'text',
  254. text: `Full formatted result stored at: /private${ctx.cwd}/.spill/session-c22bc3f1d2af/8a7b6c5d4e3f-bash.txt. Use read with offset/limit, or grep this path to search within it.`,
  255. }],
  256. },
  257. })
  258. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  259. expect(out).toContain('{{spillLocator:bash.txt}}')
  260. expect(out).not.toContain('/private{{spillLocator')
  261. })
  262. it('scrubs macOS /private prefix on cwd-rooted fs tool result paths', () => {
  263. const ev = JSON.stringify({
  264. type: 'tool/result', seq: 2, time: 5,
  265. data: {
  266. content: [{
  267. type: 'text',
  268. text: `The file /private${ctx.cwd}/config.txt has been updated successfully.`,
  269. }],
  270. },
  271. })
  272. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  273. expect(out).toContain('{{cwd}}/config.txt')
  274. expect(out).not.toContain('/private{{cwd}}')
  275. })
  276. it('scrubs fixed snapshot spill paths', () => {
  277. const ev = JSON.stringify({
  278. type: 'tool/result', seq: 2, time: 5,
  279. data: {
  280. content: [{
  281. type: 'text',
  282. text: 'Full formatted result stored at: /tmp/dsh-acp-snapshot-spill/session-c22bc3f1d2af/8a7b6c5d4e3f-bash.txt. Use read with offset/limit, or grep this path to search within it.',
  283. }],
  284. },
  285. })
  286. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  287. expect(out).toContain('{{spillLocator:bash.txt}}')
  288. expect(out).not.toContain('/tmp/dsh-acp-snapshot-spill')
  289. })
  290. it('scrubs scenario-owned snapshot spill paths', () => {
  291. const ev = JSON.stringify({
  292. type: 'tool/result', seq: 2, time: 5,
  293. data: {
  294. content: [{
  295. type: 'text',
  296. text: 'Full formatted result stored at: /tmp/dsh-acp-snap-012345678/session-c22bc3f1d2af/8a7b6c5d4e3f-bash.txt. Use read with offset/limit, or grep this path to search within it.',
  297. }],
  298. },
  299. })
  300. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  301. expect(out).toContain('{{spillLocator:bash.txt}}')
  302. expect(out).not.toContain('/tmp/dsh-acp-snap-012345678')
  303. })
  304. it('scrubs scenario-owned snapshot spill paths with Windows drive and separators', () => {
  305. const ev = JSON.stringify({
  306. type: 'tool/result', seq: 2, time: 5,
  307. data: {
  308. content: [{
  309. type: 'text',
  310. text: String.raw`Full formatted result stored at: C:\t\dsh-acp-snap-012345678\session-c22bc3f1d2af\8a7b6c5d4e3f-bash.txt. Use read with offset/limit, or grep this path to search within it.`,
  311. }],
  312. },
  313. })
  314. const out = normalizeSessionLog(`${header({ cwd: ctx.cwd })}\n${ev}\n`, ctx)
  315. expect(out).toContain('{{spillLocator:bash.txt}}')
  316. expect(out).not.toContain('C:\\t\\dsh-acp-snap-012345678')
  317. })
  318. it('shares cwd-rooted path handling with stdout normalization', () => {
  319. const windowsCtx: NormalizeContext = { sessionIds: [], cwd: String.raw`C:\work\snapshot` }
  320. const ev = JSON.stringify({
  321. type: 'tool/result', seq: 2, time: 5,
  322. data: { path: `${windowsCtx.cwd}\\nested\\proof.txt` },
  323. })
  324. expect(normalizeSessionLog(`${header({ cwd: windowsCtx.cwd })}\n${ev}\n`, windowsCtx))
  325. .toContain('{{cwd}}/nested/proof.txt')
  326. expect(normalizeSessionLog(`${header({ cwd: windowsCtx.cwd })}\n${ev}\n`, windowsCtx, { cwdPathMode: 'native' }))
  327. .toContain(String.raw`{{cwd}}\\nested\\proof.txt`)
  328. })
  329. it('scrubs the session id in the header', () => {
  330. const out = normalizeSessionLog(`${header({ id: ctx.sessionIds[0] })}\n`, ctx)
  331. expect(out).toContain('{{sessionId}}')
  332. })
  333. it('zeroes a hook/result durationMs (run-to-run noise) but keeps its decision', () => {
  334. const ev = JSON.stringify({
  335. type: 'hook/result', seq: 2, time: 5,
  336. data: { turn: 1, point: 'UserPromptSubmit', handlerId: 'h', decision: 'block', exitCode: 2, durationMs: 37 },
  337. })
  338. const out = normalizeSessionLog(`${header({})}\n${ev}\n`, ctx)
  339. expect(out).toContain('"durationMs":0')
  340. expect(out).not.toContain('37')
  341. expect(out).toContain('"decision":"block"') // the decision is the behavior — kept
  342. })
  343. it('zeroes a packed chunk row\'s time0 and dt gaps but keeps seq0 and payload', () => {
  344. const row = JSON.stringify({
  345. type: 'text-chunks', seq0: 7, time0: 999,
  346. data: { turn: 1, step: 1, index: 0, dt: [212, 27, 0], texts: ['a', 'b', 'c', 'd'] },
  347. })
  348. const out = normalizeSessionLog(`${header({})}\n${row}\n`, ctx)
  349. expect(out).toContain('"time0":0')
  350. expect(out).toContain('"dt":[0,0,0]')
  351. expect(out).toContain('"seq0":7') // seq0 is deterministic, like seq — NOT scrubbed
  352. expect(out).toContain('"texts":["a","b","c","d"]')
  353. expect(out).not.toContain('999')
  354. expect(out).not.toContain('212')
  355. })
  356. it('zeroes time0 even when a malformed row carries no dt array', () => {
  357. const row = JSON.stringify({ type: 'text-chunks', seq0: 1, time0: 999, data: 'not-an-object' })
  358. const out = normalizeSessionLog(`${header({})}\n${row}\n`, ctx)
  359. expect(out).toContain('"time0":0')
  360. expect(out).not.toContain('999')
  361. })
  362. it('leaves a non-hook event durationMs untouched (only hook/result is scrubbed)', () => {
  363. const ev = JSON.stringify({ type: 'tool/result', seq: 2, time: 5, data: { durationMs: 88 } })
  364. const out = normalizeSessionLog(`${header({})}\n${ev}\n`, ctx)
  365. expect(out).toContain('"durationMs":88')
  366. })
  367. it('tolerates records missing the volatile fields it would zero', () => {
  368. const bareHeader = JSON.stringify({ type: 'session', id: 's' })
  369. const timeless = JSON.stringify({ type: 'note', seq: 1 })
  370. const bareHook = JSON.stringify({ type: 'hook/result', seq: 2, time: 5, data: { decision: 'allow' } })
  371. const nullDataHook = JSON.stringify({ type: 'hook/result', seq: 3, time: 6, data: null })
  372. const out = normalizeSessionLog(`${bareHeader}\n${timeless}\n${bareHook}\n${nullDataHook}\n`, ctx)
  373. expect(out).toContain('"type":"note","seq":1')
  374. expect(out).toContain('"decision":"allow"')
  375. expect(out).not.toContain('durationMs')
  376. })
  377. })
  378. describe('tokenizeSessionFixtureCwd', () => {
  379. it.each([
  380. {
  381. name: 'macOS',
  382. context: {
  383. sessionIds: [],
  384. cwd: '/var/folders/2g/snapshot/T/acp-snap-cwd-abc123',
  385. cwdAliases: ['/private/var/folders/2g/snapshot/T/acp-snap-cwd-abc123'],
  386. },
  387. reportedCwd: '/private/var/folders/2g/snapshot/T/acp-snap-cwd-abc123',
  388. },
  389. {
  390. name: 'Linux',
  391. context: {
  392. sessionIds: [],
  393. cwd: '/tmp/acp-snap-cwd-abc123',
  394. },
  395. reportedCwd: '/tmp/acp-snap-cwd-abc123',
  396. },
  397. {
  398. name: 'Windows',
  399. context: {
  400. sessionIds: [],
  401. cwd: String.raw`C:\Users\runner\AppData\Local\Temp\acp-snap-cwd-abc123`,
  402. },
  403. reportedCwd: String.raw`C:\Users\runner\AppData\Local\Temp\acp-snap-cwd-abc123`,
  404. },
  405. ])('stores $name temporary workspaces with one portable root token', ({ context, reportedCwd }) => {
  406. const raw = [
  407. JSON.stringify({ type: 'session', id: 's', createdAt: 1, cwd: context.cwd }),
  408. JSON.stringify({
  409. type: 'tool/result',
  410. seq: 1,
  411. time: 2,
  412. data: {
  413. content: [{
  414. type: 'text',
  415. text: `wrote ${reportedCwd}/proof.txt. alias /different/root/acp-snap-cwd-abc123/alias.txt. cwd ${context.cwd}. Next; kept ${context.cwd}-backup, ${context.cwd}.backup, and /tmp/authored.txt`,
  416. }],
  417. },
  418. }),
  419. '',
  420. ].join('\n')
  421. const out = tokenizeSessionFixtureCwd(raw)
  422. const result = JSON.parse(out.split('\n')[1] as string) as {
  423. data: { content: { text: string }[] }
  424. }
  425. const resultText = (result.data.content[0] as { text: string }).text
  426. expect(out).toContain('"cwd":"{{cwd}}"')
  427. expect(resultText).toContain('wrote {{cwd}}/proof.txt')
  428. expect(resultText).toContain('alias {{cwd}}/alias.txt')
  429. expect(resultText).toContain('cwd {{cwd}}. Next')
  430. expect(resultText).toContain(`${context.cwd}-backup`)
  431. expect(resultText).toContain(`${context.cwd}.backup`)
  432. expect(resultText).toContain('/tmp/authored.txt')
  433. expect(resultText).not.toContain(`${reportedCwd}/proof.txt`)
  434. expect(tokenizeSessionFixtureCwd(out)).toBe(out)
  435. })
  436. it('collapses a residual macOS realpath prefix around an existing cwd token', () => {
  437. const raw = [
  438. JSON.stringify({ type: 'session', id: 's', createdAt: 1, cwd: '{{cwd}}' }),
  439. JSON.stringify({
  440. type: 'tool/result',
  441. seq: 1,
  442. time: 2,
  443. data: { content: [{ type: 'text', text: 'wrote /private{{cwd}}/proof.txt' }] },
  444. }),
  445. '',
  446. ].join('\n')
  447. const out = tokenizeSessionFixtureCwd(raw)
  448. expect(out).toContain('wrote {{cwd}}/proof.txt')
  449. expect(out).not.toContain('/private{{cwd}}')
  450. expect(tokenizeSessionFixtureCwd(out)).toBe(out)
  451. })
  452. it('rejects a log without a session cwd', () => {
  453. expect(() => tokenizeSessionFixtureCwd('')).toThrow(
  454. 'acp-snapshot: cannot tokenize a cwd without a basename',
  455. )
  456. })
  457. })
  458. describe('extractSnapshotSpillPaths', () => {
  459. it('maps each spill filename to its full matched path, last match wins per name', () => {
  460. const log = [
  461. 'Full formatted result stored at: /tmp/dsh-acp-snapshot-spill/session-c22bc3f1d2af/8a7b6c5d4e3f-bash.txt. Use read with offset/limit, or grep this path to search within it.',
  462. 'stale copy at /tmp/dsh-acp-snap-012345678/session-aaaaaaaaaaaa/bbbbbbbbbbbb-grep.txt then',
  463. 'fresh copy at /tmp/dsh-acp-snap-012345678/session-cccccccccccc/dddddddddddd-grep.txt then',
  464. ].join('\n')
  465. expect(extractSnapshotSpillPaths(log)).toEqual(new Map([
  466. ['bash.txt', '/tmp/dsh-acp-snapshot-spill/session-c22bc3f1d2af/8a7b6c5d4e3f-bash.txt'],
  467. ['grep.txt', '/tmp/dsh-acp-snap-012345678/session-cccccccccccc/dddddddddddd-grep.txt'],
  468. ]))
  469. })
  470. it('returns an empty map when the log carries no snapshot spill paths', () => {
  471. expect(extractSnapshotSpillPaths('no spill paths here, only /tmp/other.txt\n')).toEqual(new Map())
  472. })
  473. })
  474. describe('scrubRequestHeaders', () => {
  475. const headerLine = JSON.stringify({ type: 'session', version: 0, id: 's', createdAt: 1, cwd: '/w' })
  476. const headerEvent = (header: object) =>
  477. JSON.stringify({ type: 'request/header', seq: 3, time: 9, data: { header, reason: 'initial' } })
  478. it('replaces header system and tools with tokens, keeping config and reason', () => {
  479. const ev = headerEvent({
  480. config: { model: 'm' },
  481. system: 'You are an agent.\nBe brief.',
  482. tools: [{ name: 'read', description: 'Read a file.', parameters: { type: 'object' } }],
  483. })
  484. const out = scrubRequestHeaders(`${headerLine}\n${ev}\n`)
  485. expect(out).toContain('"system":"{{system}}"')
  486. expect(out).toContain('"tools":"{{tools}}"')
  487. expect(out).toContain('"config":{"model":"m"}')
  488. expect(out).toContain('"reason":"initial"')
  489. expect(out).not.toContain('You are an agent')
  490. expect(out).not.toContain('Read a file')
  491. })
  492. it('keeps an absent system/tools absent (presence is behavior)', () => {
  493. const out = scrubRequestHeaders(`${headerLine}\n${headerEvent({ config: { model: 'm' } })}\n`)
  494. expect(out).not.toContain('{{system}}')
  495. expect(out).not.toContain('{{tools}}')
  496. })
  497. it('scrubs a header carrying only one of system/tools, leaving the other absent', () => {
  498. const systemOnly = scrubRequestHeaders(`${headerLine}\n${headerEvent({ system: 'secret prompt' })}\n`)
  499. expect(systemOnly).toContain('"system":"{{system}}"')
  500. expect(systemOnly).not.toContain('{{tools}}')
  501. const toolsOnly = scrubRequestHeaders(`${headerLine}\n${headerEvent({ tools: [{ name: 't' }] })}\n`)
  502. expect(toolsOnly).toContain('"tools":"{{tools}}"')
  503. expect(toolsOnly).not.toContain('{{system}}')
  504. })
  505. it('leaves malformed headers with no scrubbable payload byte-identical', () => {
  506. const headerless = JSON.stringify({ type: 'request/header', seq: 10, time: 9, data: { reason: 'initial' } })
  507. const nullData = JSON.stringify({ type: 'request/header', seq: 11, time: 9, data: null })
  508. const raw = `${headerLine}\n${headerless}\n${nullData}\n`
  509. expect(scrubRequestHeaders(raw)).toBe(raw)
  510. })
  511. it('passes every other line through byte-for-byte and is idempotent', () => {
  512. const other = JSON.stringify({ type: 'assistant/chunk', seq: 4, time: 9, data: { turn: 1, step: 1, chunk: { type: 'text-delta', index: 0, text: 'hi' } } })
  513. const raw = `${headerLine}\n${headerEvent({ config: { model: 'm' }, system: 's', tools: [] })}\n${other}\n`
  514. const once = scrubRequestHeaders(raw)
  515. expect(once.split('\n')[0]).toBe(headerLine)
  516. expect(once.split('\n')[2]).toBe(other)
  517. expect(scrubRequestHeaders(once)).toBe(once)
  518. })
  519. })
  520. describe('scrubSystemPrompts', () => {
  521. it('scrubs only system prompt payloads while keeping tools verbatim', () => {
  522. const header = JSON.stringify({
  523. type: 'request/header', seq: 1, time: 2,
  524. data: {
  525. header: {
  526. system: 'full prompt',
  527. tools: [{ name: 'read', description: 'full schema' }],
  528. },
  529. reason: 'initial',
  530. },
  531. })
  532. const changed = JSON.stringify({
  533. type: 'request/header', seq: 2, time: 3,
  534. data: {
  535. header: {
  536. system: 'new prompt',
  537. tools: [{ name: 'read', description: 'changed schema' }],
  538. },
  539. reason: 'change',
  540. },
  541. })
  542. const toolsOnly = JSON.stringify({
  543. type: 'request/header', seq: 3, time: 4,
  544. data: { header: { tools: [{ name: 'read', description: 'schema only' }] }, reason: 'resume' },
  545. })
  546. const out = scrubSystemPrompts(`${header}\n${changed}\n${toolsOnly}\n`)
  547. expect(out).toContain('"system":"{{system}}"')
  548. expect(out).not.toContain('full prompt')
  549. expect(out).not.toContain('new prompt')
  550. expect(out).toContain('full schema')
  551. expect(out).toContain('changed schema')
  552. expect(out.split('\n')[2]).toBe(toolsOnly)
  553. expect(scrubSystemPrompts(out)).toBe(out)
  554. })
  555. })
  556. describe('scrubToolSchemas', () => {
  557. it('scrubs only tool-schema payloads while keeping prompts verbatim', () => {
  558. const header = JSON.stringify({
  559. type: 'request/header', seq: 1, time: 2,
  560. data: {
  561. header: {
  562. system: 'full prompt',
  563. tools: [{ name: 'read', description: 'full schema', parameters: { type: 'object' } }],
  564. },
  565. reason: 'initial',
  566. },
  567. })
  568. const changed = JSON.stringify({
  569. type: 'request/header', seq: 2, time: 3,
  570. data: {
  571. header: {
  572. system: 'new prompt',
  573. tools: [{ name: 'grep', description: 'new schema' }],
  574. },
  575. reason: 'change',
  576. },
  577. })
  578. const systemOnly = JSON.stringify({
  579. type: 'request/header', seq: 3, time: 4,
  580. data: { header: { system: 'prompt only' }, reason: 'resume' },
  581. })
  582. const out = scrubToolSchemas(`${header}\n${changed}\n${systemOnly}\n`)
  583. expect(out.match(/"tools":"{{tools}}"/g)).toHaveLength(2)
  584. expect(out).not.toContain('full schema')
  585. expect(out).not.toContain('new schema')
  586. expect(out).toContain('full prompt')
  587. expect(out).toContain('new prompt')
  588. expect(out.split('\n')[2]).toBe(systemOnly)
  589. expect(scrubToolSchemas(out)).toBe(out)
  590. })
  591. })