conversation-fold.bench.client.ts 3.0 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778
  1. /** Required performance budget for the compiled cold Client conversation fold. */
  2. import { join } from 'node:path'
  3. import { describe, expect, it } from 'vitest'
  4. import {
  5. runBuiltBenchmarkWorker,
  6. type BuiltBenchmarkWorkerRun,
  7. } from '../support/built-worker.ts'
  8. import {
  9. ciTimeBudget,
  10. PERFORMANCE_BUDGET_HEADROOM,
  11. } from '../support/calibration.ts'
  12. import type { ConversationFoldWorkerReport } from './conversation-fold.worker.client.ts'
  13. /** Replies in the folded window; each carries one reasoning block and one text block. */
  14. const TURNS = 200
  15. /** Text deltas per reply in the large workload; each reply adds one quarter as many reasoning deltas. */
  16. const LARGE_DELTAS = 2_000
  17. /** Text deltas per reply in the small workload used as the scaling reference. */
  18. const SMALL_DELTAS = 100
  19. /** Fresh object graphs measured in one compiled worker; the fastest sample removes scheduler delay. */
  20. const ATTEMPTS = 3
  21. /** A stuck fold worker is reaped before the outer benchmark deadline. */
  22. const WORKER_TIMEOUT_MS = 60_000
  23. /**
  24. * The large window contains 500,000 streamed deltas compacted into 1,600
  25. * stream records. The budget separates the record-proportional fold from the
  26. * per-delta replay that needs hundreds of milliseconds for the same window.
  27. */
  28. const EXPECTED_LARGE_FOLD_MS = 16
  29. const LARGE_FOLD_BUDGET_MS = ciTimeBudget(EXPECTED_LARGE_FOLD_MS)
  30. /**
  31. * Both windows contain equal event and compact-record counts. A fold over
  32. * records plus joined text measures about 2.5×; replaying every delta measures
  33. * about 11× as the delta count grows 20×.
  34. */
  35. const EXPECTED_DELTA_SCALING = 2.5
  36. const MAX_DELTA_SCALING = EXPECTED_DELTA_SCALING * PERFORMANCE_BUDGET_HEADROOM
  37. const WORKER = join(
  38. import.meta.dirname,
  39. '..',
  40. '.dsh-build',
  41. 'conversation-fold',
  42. 'conversation-fold.worker.js',
  43. )
  44. function requireReport(
  45. run: BuiltBenchmarkWorkerRun<ConversationFoldWorkerReport>,
  46. ): ConversationFoldWorkerReport {
  47. if (run.report !== undefined) return run.report
  48. const stderrLines = run.stderr.trim().split('\n')
  49. throw new Error(
  50. `conversation-fold worker failed: exit=${String(run.exitCode)}, signal=${String(run.signal)}, `
  51. + `timedOut=${String(run.timedOut)}\n${stderrLines.slice(-10).join('\n')}`,
  52. )
  53. }
  54. describe('cold Chat fold of a large v2 history window', () => {
  55. it(`folds ${String(TURNS)} replies with ${String(LARGE_DELTAS)} deltas each within ${String(LARGE_FOLD_BUDGET_MS)} ms and scales with compact records`, async () => {
  56. const report = requireReport(await runBuiltBenchmarkWorker<ConversationFoldWorkerReport>({
  57. worker: WORKER,
  58. args: [String(TURNS), String(SMALL_DELTAS), String(LARGE_DELTAS), String(ATTEMPTS)],
  59. timeoutMs: WORKER_TIMEOUT_MS,
  60. }))
  61. console.log(JSON.stringify({
  62. benchmark: 'conversation-fold/large-window',
  63. ...report,
  64. budgetMs: LARGE_FOLD_BUDGET_MS,
  65. maxScaling: MAX_DELTA_SCALING,
  66. }))
  67. expect(report.chatNodes).toBeGreaterThan(0)
  68. expect(report.largeFoldMs).toBeLessThanOrEqual(LARGE_FOLD_BUDGET_MS)
  69. expect(report.scaling).toBeLessThanOrEqual(MAX_DELTA_SCALING)
  70. })
  71. })