1
0

smoke-python-runtime.py 42 KB

12345678910111213141516171819202122232425262728293031323334353637383940414243444546474849505152535455565758596061626364656667686970717273747576777879808182838485868788899091929394959697989910010110210310410510610710810911011111211311411511611711811912012112212312412512612712812913013113213313413513613713813914014114214314414514614714814915015115215315415515615715815916016116216316416516616716816917017117217317417517617717817918018118218318418518618718818919019119219319419519619719819920020120220320420520620720820921021121221321421521621721821922022122222322422522622722822923023123223323423523623723823924024124224324424524624724824925025125225325425525625725825926026126226326426526626726826927027127227327427527627727827928028128228328428528628728828929029129229329429529629729829930030130230330430530630730830931031131231331431531631731831932032132232332432532632732832933033133233333433533633733833934034134234334434534634734834935035135235335435535635735835936036136236336436536636736836937037137237337437537637737837938038138238338438538638738838939039139239339439539639739839940040140240340440540640740840941041141241341441541641741841942042142242342442542642742842943043143243343443543643743843944044144244344444544644744844945045145245345445545645745845946046146246346446546646746846947047147247347447547647747847948048148248348448548648748848949049149249349449549649749849950050150250350450550650750850951051151251351451551651751851952052152252352452552652752852953053153253353453553653753853954054154254354454554654754854955055155255355455555655755855956056156256356456556656756856957057157257357457557657757857958058158258358458558658758858959059159259359459559659759859960060160260360460560660760860961061161261361461561661761861962062162262362462562662762862963063163263363463563663763863964064164264364464564664764864965065165265365465565665765865966066166266366466566666766866967067167267367467567667767867968068168268368468568668768868969069169269369469569669769869970070170270370470570670770870971071171271371471571671771871972072172272372472572672772872973073173273373473573673773873974074174274374474574674774874975075175275375475575675775875976076176276376476576676776876977077177277377477577677777877978078178278378478578678778878979079179279379479579679779879980080180280380480580680780880981081181281381481581681781881982082182282382482582682782882983083183283383483583683783883984084184284384484584684784884985085185285385485585685785885986086186286386486586686786886987087187287387487587687787887988088188288388488588688788888989089189289389489589689789889990090190290390490590690790890991091191291391491591691791891992092192292392492592692792892993093193293393493593693793893994094194294394494594694794894995095195295395495595695795895996096196296396496596696796896997097197297397497597697797897998098198298398498598698798898999099199299399499599699799899910001001100210031004100510061007100810091010101110121013101410151016101710181019102010211022102310241025102610271028102910301031103210331034103510361037103810391040104110421043104410451046104710481049105010511052105310541055105610571058105910601061
  1. #!/usr/bin/env python3
  2. """Keyless full-turn and snapshot smoke for the Python SDK runtime."""
  3. from __future__ import annotations
  4. import argparse
  5. import difflib
  6. import json
  7. import os
  8. import queue
  9. import subprocess
  10. import tempfile
  11. import threading
  12. import time
  13. from http.server import BaseHTTPRequestHandler, ThreadingHTTPServer
  14. from pathlib import Path
  15. from typing import TYPE_CHECKING, Callable
  16. if TYPE_CHECKING:
  17. from deepseek_harness import RunResult
  18. EXPECTED_TEXT = "runtime smoke ok"
  19. CODE_PROMPT = "Use run_code to compute the packaged worker smoke value."
  20. CODE_WORKER_TEXT = "code worker smoke ok"
  21. WORKFLOW_PROMPT = "Use workflow to compute the packaged worker smoke value without agents."
  22. WORKFLOW_WORKER_TEXT = "workflow worker smoke ok"
  23. MINIMAL_PROMPT = "Exercise the packaged minimal agent's persistent Bash and string-replacement editor."
  24. MINIMAL_TEXT = "minimal agent smoke ok"
  25. MINIMAL_EDITOR_PATH_PREFIX = "Editor path: "
  26. MINIMAL_SYSTEM_PROMPT = "You are a helpful software engineer assistant."
  27. FS_SEARCH_PROMPT = "Exercise the packaged filesystem search tools."
  28. FS_SEARCH_TEXT = "filesystem search smoke ok"
  29. FS_SEARCH_MARKER = "PACKAGED_FS_SEARCH_OK"
  30. MINIMAL_CORDIS = (
  31. Path(__file__).resolve().parent.parent / "examples" / "jsonrpc-agent" / "minimal.cordis.yml"
  32. )
  33. MINIMAL_BASH_COMMAND = (
  34. "counter=$(( ${counter:-0} + 1 )); export counter; "
  35. "printf 'COUNT=%s CWD=%s\\n' \"$counter\" \"$PWD\"; "
  36. "if [ \"$counter\" -eq 1 ]; then cd /tmp; fi"
  37. )
  38. SNAPSHOT_PROMPT = "Run the advanced packaged-runtime snapshot scenario."
  39. SNAPSHOT_SESSION_ID = "advanced-executable"
  40. SNAPSHOT_DIRECT_CHILD_PROMPT = "Reply with exactly DIRECT_CHILD_OK and nothing else."
  41. SNAPSHOT_WORKFLOW_CHILD_PROMPT = "Reply with exactly WORKFLOW_CHILD_OK and nothing else."
  42. SNAPSHOT_FINAL_TEXT = "ADVANCED_EXECUTABLE_OK"
  43. SNAPSHOT_PLUGIN_CODE = """\
  44. return (ctx) => {
  45. harness.registerTool(ctx, harness.defineTool({
  46. name: 'snapshot_double',
  47. description: 'Double a number for executable snapshot verification.',
  48. parameters: { value: { type: 'number', required: true } },
  49. output: {
  50. schema: { type: 'number' },
  51. render(_args, value) {
  52. return [{ type: 'text', text: String(value) }]
  53. }
  54. },
  55. async execute(args) {
  56. return args.value * 2
  57. }
  58. }))
  59. }
  60. """
  61. SNAPSHOT_WORKFLOW_SCRIPT = (
  62. "phase('Delegate')\n"
  63. f"const reply = await agent('{SNAPSHOT_WORKFLOW_CHILD_PROMPT}', {{ label: 'workflow-child' }})\n"
  64. "return { reply }"
  65. )
  66. SNAPSHOT_DIRECTORY = (
  67. Path(__file__).resolve().parent / "snapshots" / "python-sdk-single-exe" / "advanced"
  68. )
  69. SNAPSHOT_FILENAMES = ("result.json", "session.jsonl", "session.1.jsonl", "session.2.jsonl")
  70. CUSTOM_CORDIS = """\
  71. - id: sdk-jsonrpc-server
  72. name: '@deepseek-ai/dsh-sdk-jsonrpc-server'
  73. - id: agent-core
  74. name: '@deepseek-ai/dsh-agent-spine-demo'
  75. config:
  76. workspaceContext: false
  77. skills:
  78. enabled: false
  79. toolBash: false
  80. tools:
  81. mode: both
  82. - id: sessions
  83. name: '@deepseek-ai/dsh-session-persistence-jsonl'
  84. config:
  85. root: !!js process.env.DSH_SESSION_ROOT
  86. compression: 'none'
  87. - id: code-runtime
  88. name: '@deepseek-ai/dsh-code-runtime-worker-thread'
  89. - id: subagents
  90. name: '@deepseek-ai/dsh-subagent'
  91. - id: subagent-spawn-in-process
  92. name: '@deepseek-ai/dsh-subagent-spawn-in-process'
  93. config:
  94. providerName: spawn
  95. - id: subagent-tool
  96. name: '@deepseek-ai/dsh-tool-subagent'
  97. config:
  98. provider: spawn
  99. - id: workflow-engine
  100. name: '@deepseek-ai/dsh-workflow-worker-thread'
  101. config:
  102. provider: spawn
  103. - id: workflow-tool
  104. name: '@deepseek-ai/dsh-tool-workflow'
  105. - id: cordis-host-runner
  106. name: '@deepseek-ai/dsh-cordis-host-runner'
  107. - id: cordis-tool
  108. name: '@deepseek-ai/dsh-tool-cordis'
  109. """
  110. FS_SEARCH_CORDIS = """\
  111. - id: sdk-jsonrpc-server
  112. name: '@deepseek-ai/dsh-sdk-jsonrpc-server'
  113. - id: agent-core
  114. name: '@deepseek-ai/dsh-agent-spine-demo'
  115. config:
  116. workspaceContext: false
  117. skills:
  118. enabled: false
  119. toolBash: false
  120. toolJobs: false
  121. - id: sessions
  122. name: '@deepseek-ai/dsh-session-persistence-jsonl'
  123. config:
  124. root: !!js process.env.DSH_SESSION_ROOT
  125. compression: 'none'
  126. - id: subprocess
  127. name: '@deepseek-ai/dsh-subprocess-local'
  128. - id: fs-search
  129. name: '@deepseek-ai/dsh-tool-fs-search'
  130. config:
  131. sampleOverCapGlobResults: false
  132. """
  133. class MockModelHandler(BaseHTTPRequestHandler):
  134. """Return deterministic text, worker, and orchestration completions."""
  135. requests: list[dict[str, object]] = []
  136. def do_POST(self) -> None:
  137. content_length = int(self.headers.get("content-length", "0"))
  138. body = json.loads(self.rfile.read(content_length))
  139. self.requests.append(body)
  140. self.send_response(200)
  141. self.send_header("content-type", "text/event-stream")
  142. self.end_headers()
  143. chunks = completion_chunks(body)
  144. for chunk in chunks:
  145. self.wfile.write(f"data: {json.dumps(chunk)}\n\n".encode())
  146. self.wfile.write(b"data: [DONE]\n\n")
  147. self.wfile.flush()
  148. def log_message(self, _format: str, *_args: object) -> None:
  149. return
  150. def completion_chunks(body: dict[str, object]) -> list[dict[str, object]]:
  151. """Choose the next deterministic model response from request history."""
  152. messages = body.get("messages")
  153. if not isinstance(messages, list) or not messages:
  154. raise AssertionError(f"model request has no messages: {body}")
  155. latest = messages[-1]
  156. if not isinstance(latest, dict):
  157. raise AssertionError(f"model request has an invalid latest message: {body}")
  158. if latest.get("role") == "tool":
  159. call_id, tool_name = latest_tool_call(messages)
  160. tool_text = message_text(latest.get("content"))
  161. fs_search = fs_search_tool_followup(call_id, tool_name, tool_text)
  162. if fs_search is not None:
  163. return fs_search
  164. minimal = minimal_tool_followup(body, call_id, tool_name, tool_text)
  165. if minimal is not None:
  166. return minimal
  167. advanced = advanced_tool_followup(body, call_id, tool_name, tool_text)
  168. if advanced is not None:
  169. return advanced
  170. if "42" not in tool_text:
  171. raise AssertionError(f"{tool_name} worker returned no expected value: {latest}")
  172. if tool_name == "run_code":
  173. return text_chunks(CODE_WORKER_TEXT)
  174. if tool_name == "workflow":
  175. return text_chunks(WORKFLOW_WORKER_TEXT)
  176. raise AssertionError(f"unexpected tool follow-up: {tool_name}")
  177. user_prompts = [
  178. message_text(message.get("content"))
  179. for message in reversed(messages)
  180. if isinstance(message, dict) and message.get("role") == "user"
  181. ]
  182. minimal_prompt = next(
  183. (
  184. prompt
  185. for prompt in user_prompts
  186. if prompt.startswith(f"{MINIMAL_PROMPT}\n{MINIMAL_EDITOR_PATH_PREFIX}")
  187. ),
  188. None,
  189. )
  190. if minimal_prompt is not None:
  191. names = advertised_tool_names(body)
  192. if names != {"bash", "str_replace_editor"}:
  193. raise AssertionError(f"minimal agent smoke advertised unexpected tools: {names}")
  194. system_prompts = [
  195. message_text(message.get("content"))
  196. for message in messages
  197. if isinstance(message, dict) and message.get("role") == "system"
  198. ]
  199. if system_prompts != [MINIMAL_SYSTEM_PROMPT]:
  200. raise AssertionError(f"minimal agent smoke assembled unexpected system prompts: {system_prompts}")
  201. return tool_call_chunks(
  202. "minimal-bash-1",
  203. "bash",
  204. {"command": MINIMAL_BASH_COMMAND},
  205. )
  206. scenario_prompts = {
  207. SNAPSHOT_DIRECT_CHILD_PROMPT,
  208. SNAPSHOT_WORKFLOW_CHILD_PROMPT,
  209. SNAPSHOT_PROMPT,
  210. CODE_PROMPT,
  211. WORKFLOW_PROMPT,
  212. FS_SEARCH_PROMPT,
  213. }
  214. prompt = next(
  215. (candidate for candidate in user_prompts if candidate in scenario_prompts),
  216. message_text(latest.get("content")),
  217. )
  218. if prompt == SNAPSHOT_DIRECT_CHILD_PROMPT:
  219. return text_chunks("DIRECT_CHILD_OK")
  220. if prompt == SNAPSHOT_WORKFLOW_CHILD_PROMPT:
  221. return text_chunks("WORKFLOW_CHILD_OK")
  222. if prompt == SNAPSHOT_PROMPT:
  223. assert_advertised_tool(body, "cordis_define")
  224. return tool_call_chunks(
  225. "advanced-define",
  226. "cordis_define",
  227. {
  228. "plugin": {"kind": "new", "idPrefix": "snap"},
  229. "name": "Snapshot Double",
  230. "purpose": "Expose a deterministic doubling tool for executable snapshot verification.",
  231. "code": {"host": SNAPSHOT_PLUGIN_CODE},
  232. },
  233. )
  234. if prompt == CODE_PROMPT:
  235. assert_advertised_tool(body, "run_code")
  236. return tool_call_chunks(
  237. "call-code-worker",
  238. "run_code",
  239. {"code": "return 6 * 7", "description": "Compute the smoke value"},
  240. )
  241. if prompt == WORKFLOW_PROMPT:
  242. assert_advertised_tool(body, "workflow")
  243. return tool_call_chunks(
  244. "call-workflow-worker",
  245. "workflow",
  246. {
  247. "script": "return 6 * 7",
  248. "meta": {
  249. "name": "pkg-worker-smoke",
  250. "description": "exercise the packaged workflow worker",
  251. },
  252. },
  253. )
  254. if prompt == FS_SEARCH_PROMPT:
  255. assert_advertised_tool(body, "grep")
  256. assert_advertised_tool(body, "glob")
  257. return tool_call_chunks(
  258. "fs-search-grep",
  259. "grep",
  260. {"pattern": FS_SEARCH_MARKER, "path": "."},
  261. )
  262. return text_chunks(EXPECTED_TEXT)
  263. def fs_search_tool_followup(
  264. call_id: str,
  265. tool_name: str,
  266. tool_text: str,
  267. ) -> list[dict[str, object]] | None:
  268. """Exercise both ripgrep-backed tools through the packaged executable."""
  269. if not call_id.startswith("fs-search-"):
  270. return None
  271. if call_id == "fs-search-grep" and tool_name == "grep":
  272. if "needle.txt" not in tool_text or FS_SEARCH_MARKER not in tool_text:
  273. raise AssertionError(f"packaged grep returned no marker: {tool_text}")
  274. return tool_call_chunks(
  275. "fs-search-glob",
  276. "glob",
  277. {"pattern": "**/*.txt"},
  278. )
  279. if call_id == "fs-search-glob" and tool_name == "glob":
  280. if "needle.txt" not in tool_text:
  281. raise AssertionError(f"packaged glob returned no fixture path: {tool_text}")
  282. return text_chunks(FS_SEARCH_TEXT)
  283. raise AssertionError(f"unexpected filesystem-search follow-up: {call_id} {tool_name}: {tool_text}")
  284. def minimal_tool_followup(
  285. body: dict[str, object],
  286. call_id: str,
  287. tool_name: str,
  288. tool_text: str,
  289. ) -> list[dict[str, object]] | None:
  290. """Verify the checked-in minimal composition's PTY and editor."""
  291. if not call_id.startswith("minimal-"):
  292. return None
  293. if call_id == "minimal-bash-1" and tool_name == "bash":
  294. if "COUNT=1" not in tool_text:
  295. raise AssertionError(f"first persistent bash call lost its output: {tool_text}")
  296. return tool_call_chunks(
  297. "minimal-bash-2",
  298. "bash",
  299. {"command": MINIMAL_BASH_COMMAND},
  300. )
  301. if call_id == "minimal-bash-2" and tool_name == "bash":
  302. if "COUNT=2 CWD=/tmp" not in tool_text:
  303. raise AssertionError(f"persistent bash did not retain state: {tool_text}")
  304. messages = body.get("messages")
  305. if not isinstance(messages, list):
  306. raise AssertionError("persistent editor smoke request has no messages")
  307. editor_path = next(
  308. (
  309. text.split(MINIMAL_EDITOR_PATH_PREFIX, 1)[1].strip()
  310. for message in messages
  311. if isinstance(message, dict) and message.get("role") == "user"
  312. for text in [message_text(message.get("content"))]
  313. if MINIMAL_EDITOR_PATH_PREFIX in text
  314. ),
  315. None,
  316. )
  317. if editor_path is None:
  318. raise AssertionError("persistent editor smoke prompt has no editor path")
  319. return tool_call_chunks(
  320. "minimal-editor",
  321. "str_replace_editor",
  322. {
  323. "command": "create",
  324. "path": editor_path,
  325. "file_text": "created by packaged editor\n",
  326. },
  327. )
  328. if call_id == "minimal-editor" and tool_name == "str_replace_editor":
  329. if "New file created successfully" not in tool_text:
  330. raise AssertionError(f"packaged editor did not create its file: {tool_text}")
  331. return text_chunks(MINIMAL_TEXT)
  332. raise AssertionError(f"unexpected minimal-agent follow-up: {call_id} {tool_name}: {tool_text}")
  333. def advanced_tool_followup(
  334. body: dict[str, object],
  335. call_id: str,
  336. tool_name: str,
  337. tool_text: str,
  338. ) -> list[dict[str, object]] | None:
  339. """Advance the executable snapshot's deterministic parent tool chain."""
  340. if not call_id.startswith("advanced-"):
  341. return None
  342. if call_id == "advanced-define" and tool_name == "cordis_define":
  343. if "Defined snap-1/pkg-1 (Snapshot Double)" not in tool_text:
  344. raise AssertionError(f"cordis_define returned no dynamic Package ids: {tool_text}")
  345. if "snapshot_double" in advertised_tool_names(body):
  346. raise AssertionError("snapshot_double was advertised before cordis_run")
  347. assert_advertised_tool(body, "cordis_run")
  348. return tool_call_chunks(
  349. "advanced-run",
  350. "cordis_run",
  351. {"pluginId": "snap-1", "packageId": "pkg-1", "mode": "run"},
  352. )
  353. if call_id == "advanced-run" and tool_name == "cordis_run":
  354. if "snap-1/pkg-1 is running (run-1)" not in tool_text:
  355. raise AssertionError(f"cordis_run returned no running Package ids: {tool_text}")
  356. assert_advertised_tool(body, "run_code")
  357. assert_advertised_tool(body, "snapshot_double")
  358. return tool_call_chunks(
  359. "advanced-code",
  360. "run_code",
  361. {
  362. "code": "return await tools.snapshot_double({ value: 21 })",
  363. "description": "Run the temporary Plugin tool",
  364. },
  365. )
  366. if call_id == "advanced-code" and tool_name == "run_code":
  367. if "42" not in tool_text:
  368. raise AssertionError(f"run_code returned no dynamic-tool value: {tool_text}")
  369. assert_advertised_tool(body, "subagent")
  370. return tool_call_chunks(
  371. "advanced-direct-child",
  372. "subagent",
  373. {
  374. "description": "Check direct child",
  375. "prompt": SNAPSHOT_DIRECT_CHILD_PROMPT,
  376. },
  377. )
  378. if call_id == "advanced-direct-child" and tool_name == "subagent":
  379. if "DIRECT_CHILD_OK" not in tool_text:
  380. raise AssertionError(f"subagent returned no expected child value: {tool_text}")
  381. assert_advertised_tool(body, "workflow")
  382. return tool_call_chunks(
  383. "advanced-workflow",
  384. "workflow",
  385. {
  386. "script": SNAPSHOT_WORKFLOW_SCRIPT,
  387. "meta": {
  388. "name": "advanced-exe-snapshot",
  389. "description": "exercise one packaged workflow child",
  390. },
  391. },
  392. )
  393. if call_id == "advanced-workflow" and tool_name == "workflow":
  394. if "WORKFLOW_CHILD_OK" not in tool_text:
  395. raise AssertionError(f"workflow returned no expected child value: {tool_text}")
  396. assert_advertised_tool(body, "cordis_undefine")
  397. return tool_call_chunks(
  398. "advanced-undefine",
  399. "cordis_undefine",
  400. {"pluginId": "snap-1"},
  401. )
  402. if call_id == "advanced-undefine" and tool_name == "cordis_undefine":
  403. if "Removed dynamic Plugin snap-1 and all of its Packages." not in tool_text:
  404. raise AssertionError(f"cordis_undefine returned no removal result: {tool_text}")
  405. if "snapshot_double" in advertised_tool_names(body):
  406. raise AssertionError("snapshot_double remained advertised after cordis_undefine")
  407. return text_chunks(SNAPSHOT_FINAL_TEXT)
  408. raise AssertionError(f"unexpected advanced tool follow-up: {call_id} {tool_name}: {tool_text}")
  409. def text_chunks(text: str) -> list[dict[str, object]]:
  410. """Build a complete streaming text response."""
  411. return [
  412. {"choices": [{"delta": {"role": "assistant", "content": None, "reasoning_content": ""}}]},
  413. {"choices": [{"delta": {"content": text}}]},
  414. {
  415. "choices": [{"delta": {"content": ""}, "finish_reason": "stop"}],
  416. "usage": {"prompt_tokens": 3, "completion_tokens": 3},
  417. },
  418. ]
  419. def tool_call_chunks(call_id: str, name: str, arguments: dict[str, object]) -> list[dict[str, object]]:
  420. """Build a complete streaming function-call response."""
  421. return [
  422. {"choices": [{"delta": {"role": "assistant", "content": None, "reasoning_content": ""}}]},
  423. {
  424. "choices": [{
  425. "delta": {
  426. "tool_calls": [{
  427. "index": 0,
  428. "id": call_id,
  429. "type": "function",
  430. "function": {"name": name, "arguments": json.dumps(arguments)},
  431. }],
  432. },
  433. }],
  434. },
  435. {
  436. "choices": [{"delta": {"content": ""}, "finish_reason": "tool_calls"}],
  437. "usage": {"prompt_tokens": 3, "completion_tokens": 3},
  438. },
  439. ]
  440. def latest_tool_call(messages: list[object]) -> tuple[str, str]:
  441. """Find the assistant call id and name paired with the latest tool result."""
  442. for message in reversed(messages[:-1]):
  443. if not isinstance(message, dict):
  444. continue
  445. calls = message.get("tool_calls")
  446. if not isinstance(calls, list):
  447. continue
  448. for call in reversed(calls):
  449. if not isinstance(call, dict):
  450. continue
  451. function = call.get("function")
  452. call_id = call.get("id")
  453. if (
  454. isinstance(call_id, str)
  455. and isinstance(function, dict)
  456. and isinstance(function.get("name"), str)
  457. ):
  458. return call_id, function["name"]
  459. raise AssertionError(f"tool result has no preceding assistant tool call: {messages}")
  460. def message_text(content: object) -> str:
  461. """Read OpenAI text content in either string or block-list form."""
  462. if isinstance(content, str):
  463. return content
  464. if isinstance(content, list):
  465. return "".join(
  466. block.get("text", "")
  467. for block in content
  468. if isinstance(block, dict) and isinstance(block.get("text"), str)
  469. )
  470. return ""
  471. def advertised_tool_names(body: dict[str, object]) -> set[str]:
  472. """Return the model-facing tool names advertised on one request."""
  473. tools = body.get("tools")
  474. if not isinstance(tools, list):
  475. raise AssertionError(f"model request advertised no tools: {body}")
  476. names: set[str] = set()
  477. for tool in tools:
  478. if not isinstance(tool, dict):
  479. continue
  480. function = tool.get("function")
  481. if isinstance(function, dict) and isinstance(function.get("name"), str):
  482. names.add(function["name"])
  483. return names
  484. def assert_advertised_tool(body: dict[str, object], expected: str) -> None:
  485. """Require the packaged deployment to expose the requested tool."""
  486. names = advertised_tool_names(body)
  487. if expected not in names:
  488. raise AssertionError(f"model request did not advertise {expected}: {names}")
  489. class MockModel:
  490. def __enter__(self) -> "MockModel":
  491. MockModelHandler.requests.clear()
  492. self.server = ThreadingHTTPServer(("127.0.0.1", 0), MockModelHandler)
  493. self.thread = threading.Thread(target=self.server.serve_forever, daemon=True)
  494. self.thread.start()
  495. host, port = self.server.server_address
  496. self.url = f"http://{host}:{port}"
  497. return self
  498. def __exit__(self, _exc_type: object, _exc: object, _tb: object) -> None:
  499. self.server.shutdown()
  500. self.server.server_close()
  501. self.thread.join(timeout=5)
  502. def main() -> None:
  503. parser = argparse.ArgumentParser(description=__doc__)
  504. parser.add_argument(
  505. "--scenario",
  506. choices=("all", "sdk-default", "sdk-custom", "sdk-minimal", "sdk-fs-search", "sdk-snapshot", "direct"),
  507. default="all",
  508. )
  509. parser.add_argument("--exe", type=Path)
  510. parser.add_argument("--update-snapshots", action="store_true")
  511. args = parser.parse_args()
  512. if args.scenario in {"all", "sdk-custom", "sdk-minimal", "sdk-fs-search", "sdk-snapshot", "direct"} and args.exe is None:
  513. parser.error("--exe is required for custom, minimal, snapshot, and direct scenarios")
  514. if args.update_snapshots and args.scenario not in {"all", "sdk-snapshot"}:
  515. parser.error("--update-snapshots requires --scenario sdk-snapshot or all")
  516. if args.exe is not None and not args.exe.is_file():
  517. parser.error(f"runtime executable does not exist: {args.exe}")
  518. with MockModel() as model:
  519. if args.scenario in {"all", "sdk-default"}:
  520. smoke_sdk_default(model.url)
  521. if args.scenario in {"all", "sdk-custom"}:
  522. assert args.exe is not None
  523. smoke_sdk_custom(model.url, args.exe.resolve())
  524. if args.scenario in {"all", "sdk-minimal"}:
  525. assert args.exe is not None
  526. smoke_sdk_minimal(model.url, args.exe.resolve())
  527. if args.scenario in {"all", "sdk-fs-search"}:
  528. assert args.exe is not None
  529. smoke_sdk_fs_search(model.url, args.exe.resolve())
  530. if args.scenario in {"all", "sdk-snapshot"}:
  531. assert args.exe is not None
  532. smoke_sdk_snapshot(model.url, args.exe.resolve(), args.update_snapshots)
  533. if args.scenario in {"all", "direct"}:
  534. assert args.exe is not None
  535. smoke_direct(model.url, args.exe.resolve())
  536. if not MockModelHandler.requests:
  537. raise AssertionError("mock model endpoint received no requests")
  538. print(f"smoke-python-runtime: {args.scenario} passed")
  539. def smoke_sdk_default(base_url: str) -> None:
  540. from deepseek_harness import DeepSeekHarness
  541. with tempfile.TemporaryDirectory(prefix="dsh-sdk-default-") as temporary:
  542. root = Path(temporary).resolve()
  543. sessions = root / "sessions"
  544. with DeepSeekHarness(
  545. provider="deepseek-official",
  546. model="smoke-model",
  547. cwd=str(root),
  548. session_root=str(sessions),
  549. api_key="sk-keyless-smoke",
  550. base_url=base_url,
  551. request_timeout_seconds=60,
  552. ) as harness:
  553. result = harness.run("reply with the smoke text", session_id="default-smoke")
  554. assert result.final_response == EXPECTED_TEXT, result.final_response
  555. assert_zstd_session_log(sessions)
  556. def smoke_sdk_custom(base_url: str, executable: Path) -> None:
  557. from deepseek_harness import DeepSeekHarness
  558. with tempfile.TemporaryDirectory(prefix="dsh-sdk-custom-") as temporary:
  559. root = Path(temporary).resolve()
  560. sessions = root / "sessions"
  561. cordis = root / "cordis.yml"
  562. cordis.write_text(CUSTOM_CORDIS)
  563. with DeepSeekHarness(
  564. provider="deepseek-official",
  565. model="smoke-model",
  566. cwd=str(root),
  567. session_root=str(sessions),
  568. cordis=str(cordis),
  569. runtime_bin=str(executable),
  570. api_key="sk-keyless-smoke",
  571. base_url=base_url,
  572. request_timeout_seconds=60,
  573. ) as harness:
  574. text_result = harness.run("reply with the smoke text", session_id="custom-smoke")
  575. code_result = harness.run(CODE_PROMPT, session_id="custom-smoke")
  576. workflow_result = harness.run(WORKFLOW_PROMPT, session_id="custom-smoke")
  577. assert text_result.final_response == EXPECTED_TEXT, text_result.final_response
  578. assert code_result.final_response == CODE_WORKER_TEXT, code_result.final_response
  579. assert workflow_result.final_response == WORKFLOW_WORKER_TEXT, workflow_result.final_response
  580. assert_session_log(sessions, root, EXPECTED_TEXT, CODE_WORKER_TEXT, WORKFLOW_WORKER_TEXT)
  581. def smoke_sdk_minimal(base_url: str, executable: Path) -> None:
  582. """Exercise the checked-in minimal composition through the packaged executable."""
  583. from deepseek_harness import DeepSeekHarness
  584. with tempfile.TemporaryDirectory(prefix="dsh-sdk-minimal-") as temporary:
  585. root = Path(temporary).resolve()
  586. editor_path = root / "created.txt"
  587. prompt = f"{MINIMAL_PROMPT}\n{MINIMAL_EDITOR_PATH_PREFIX}{editor_path}"
  588. sessions = root / "sessions"
  589. with DeepSeekHarness(
  590. provider="deepseek-official",
  591. model="smoke-model",
  592. cwd=str(root),
  593. session_root=str(sessions),
  594. cordis=str(MINIMAL_CORDIS),
  595. runtime_bin=str(executable),
  596. api_key="sk-keyless-smoke",
  597. base_url=base_url,
  598. request_timeout_seconds=60,
  599. ) as harness:
  600. result = harness.run(prompt, session_id="minimal-agent-smoke")
  601. event_text = json.dumps(result.events)
  602. if MINIMAL_TEXT not in event_text:
  603. raise AssertionError(f"minimal agent run emitted no final response: {result.events}")
  604. if editor_path.read_text() != "created by packaged editor\n":
  605. raise AssertionError(f"packaged editor wrote unexpected content: {editor_path.read_text()!r}")
  606. assert_session_log(sessions, root, MINIMAL_TEXT, "COUNT=1", "COUNT=2 CWD=/tmp")
  607. def smoke_sdk_fs_search(base_url: str, executable: Path) -> None:
  608. """Exercise real grep and glob spawns through the packaged executable."""
  609. from deepseek_harness import DeepSeekHarness
  610. with tempfile.TemporaryDirectory(prefix="dsh-sdk-fs-search-") as temporary:
  611. root = Path(temporary).resolve()
  612. (root / "needle.txt").write_text(f"{FS_SEARCH_MARKER}\n")
  613. sessions = root / "sessions"
  614. cordis = root / "cordis.yml"
  615. cordis.write_text(FS_SEARCH_CORDIS)
  616. with DeepSeekHarness(
  617. provider="deepseek-official",
  618. model="smoke-model",
  619. cwd=str(root),
  620. session_root=str(sessions),
  621. cordis=str(cordis),
  622. runtime_bin=str(executable),
  623. api_key="sk-keyless-smoke",
  624. base_url=base_url,
  625. request_timeout_seconds=60,
  626. ) as harness:
  627. result = harness.run(FS_SEARCH_PROMPT, session_id="fs-search-smoke")
  628. assert result.final_response == FS_SEARCH_TEXT, result.final_response
  629. assert_session_log(sessions, root, FS_SEARCH_TEXT, FS_SEARCH_MARKER, "needle.txt")
  630. def smoke_sdk_snapshot(base_url: str, executable: Path, update_snapshots: bool) -> None:
  631. """Drive and compare the advanced SDK/executable behavioral snapshot."""
  632. from deepseek_harness import DeepSeekHarness
  633. with tempfile.TemporaryDirectory(prefix="dsh-sdk-snapshot-") as temporary:
  634. root = Path(temporary).resolve()
  635. sessions = root / "sessions"
  636. cordis = root / "cordis.yml"
  637. cordis.write_text(CUSTOM_CORDIS)
  638. with DeepSeekHarness(
  639. provider="deepseek-official",
  640. model="smoke-model",
  641. cwd=str(root),
  642. session_root=str(sessions),
  643. cordis=str(cordis),
  644. runtime_bin=str(executable),
  645. api_key="sk-keyless-smoke",
  646. base_url=base_url,
  647. request_timeout_seconds=60,
  648. ) as harness:
  649. result = harness.run(SNAPSHOT_PROMPT, session_id=SNAPSHOT_SESSION_ID)
  650. assert result.final_response == SNAPSHOT_FINAL_TEXT, result.final_response
  651. methods = [notification.method for notification in result.notifications]
  652. if methods.count("subagent.started") != 2 or methods.count("subagent.finished") != 2:
  653. raise AssertionError(f"advanced snapshot emitted unexpected subagent lifecycle: {methods}")
  654. if not any(event.get("type") == "tool/code-dispatch" for event in result.events):
  655. raise AssertionError("advanced snapshot emitted no tool/code-dispatch event")
  656. logs = read_session_logs(sessions)
  657. child_ids = snapshot_child_ids(result)
  658. expected_ids = {SNAPSHOT_SESSION_ID, *child_ids}
  659. if set(logs) != expected_ids:
  660. raise AssertionError(f"advanced snapshot expected parent plus two child logs: {sorted(logs)}")
  661. if "DIRECT_CHILD_OK" not in render_jsonl(logs[child_ids[0]]):
  662. raise AssertionError("first advanced child log has no direct-subagent result")
  663. if "WORKFLOW_CHILD_OK" not in render_jsonl(logs[child_ids[1]]):
  664. raise AssertionError("second advanced child log has no workflow-subagent result")
  665. files = build_snapshot_files(result, logs, child_ids, root)
  666. compare_snapshot_files(files, update_snapshots)
  667. def smoke_direct(base_url: str, executable: Path) -> None:
  668. with tempfile.TemporaryDirectory(prefix="dsh-direct-") as temporary:
  669. root = Path(temporary).resolve()
  670. sessions = root / "sessions"
  671. cordis = root / "cordis.yml"
  672. cordis.write_text(CUSTOM_CORDIS)
  673. environment = {
  674. **os.environ,
  675. "DSH_CORDIS_CONFIG": str(cordis),
  676. "DSH_SESSION_ROOT": str(sessions),
  677. "DSH_CWD": str(root),
  678. "DEEPSEEK_API_KEY": "sk-keyless-smoke",
  679. "DEEPSEEK_BASE_URL": base_url,
  680. }
  681. peer = RuntimePeer([str(executable)], root, environment)
  682. try:
  683. peer.send({"jsonrpc": "2.0", "id": "initialize", "method": "initialize", "params": {"cwd": str(root), "provider": "deepseek-official", "model": "smoke-model"}})
  684. peer.read_until(lambda message: message.get("id") == "initialize")
  685. peer.send({
  686. "jsonrpc": "2.0",
  687. "id": "prompt",
  688. "method": "session/prompt",
  689. "params": {"sessionId": "direct-smoke", "contentBlocks": [{"type": "text", "text": "reply with the smoke text"}]},
  690. })
  691. messages = peer.read_until(lambda message: message.get("id") == "prompt")
  692. if not any(is_idle_notification(message) for message in messages):
  693. messages.extend(peer.read_until(is_idle_notification))
  694. event_text = json.dumps(messages)
  695. if EXPECTED_TEXT not in event_text:
  696. raise AssertionError(f"direct runtime emitted no final response: {messages}")
  697. peer.send({"jsonrpc": "2.0", "id": "shutdown", "method": "shutdown"})
  698. peer.read_until(lambda message: message.get("id") == "shutdown")
  699. finally:
  700. peer.close()
  701. assert_session_log(sessions, root, EXPECTED_TEXT)
  702. def is_idle_notification(message: dict[str, object]) -> bool:
  703. """Return whether a JSON-RPC notification marks a session idle."""
  704. params = message.get("params")
  705. return (
  706. message.get("method") == "session.status"
  707. and isinstance(params, dict)
  708. and params.get("status") == "idle"
  709. )
  710. class RuntimePeer:
  711. def __init__(self, argv: list[str], cwd: Path, environment: dict[str, str]) -> None:
  712. self.process = subprocess.Popen(
  713. argv,
  714. cwd=cwd,
  715. env=environment,
  716. stdin=subprocess.PIPE,
  717. stdout=subprocess.PIPE,
  718. stderr=subprocess.PIPE,
  719. text=True,
  720. encoding="utf-8",
  721. bufsize=1,
  722. )
  723. self.stdout: queue.Queue[str | None] = queue.Queue()
  724. self.stderr: list[str] = []
  725. threading.Thread(target=self._read_stdout, daemon=True).start()
  726. threading.Thread(target=self._read_stderr, daemon=True).start()
  727. def send(self, message: dict[str, object]) -> None:
  728. if self.process.stdin is None:
  729. raise RuntimeError("runtime stdin is unavailable")
  730. self.process.stdin.write(json.dumps(message) + "\n")
  731. self.process.stdin.flush()
  732. def read_until(self, predicate: Callable[[dict[str, object]], bool]) -> list[dict[str, object]]:
  733. deadline = time.monotonic() + 60
  734. messages: list[dict[str, object]] = []
  735. while time.monotonic() < deadline:
  736. try:
  737. line = self.stdout.get(timeout=min(0.25, deadline - time.monotonic()))
  738. except queue.Empty:
  739. continue
  740. if line is None:
  741. raise RuntimeError(f"runtime exited before expected message; stderr: {''.join(self.stderr)}")
  742. try:
  743. message = json.loads(line)
  744. except json.JSONDecodeError:
  745. continue
  746. messages.append(message)
  747. if predicate(message):
  748. return messages
  749. raise TimeoutError(f"runtime timed out; messages={messages}; stderr={''.join(self.stderr)}")
  750. def close(self) -> None:
  751. if self.process.stdin is not None and not self.process.stdin.closed:
  752. self.process.stdin.close()
  753. try:
  754. self.process.wait(timeout=10)
  755. except subprocess.TimeoutExpired:
  756. self.process.kill()
  757. self.process.wait()
  758. if self.process.returncode not in {0, -15}:
  759. raise RuntimeError(f"runtime exited {self.process.returncode}; stderr: {''.join(self.stderr)}")
  760. def _read_stdout(self) -> None:
  761. assert self.process.stdout is not None
  762. for line in self.process.stdout:
  763. self.stdout.put(line)
  764. self.stdout.put(None)
  765. def _read_stderr(self) -> None:
  766. assert self.process.stderr is not None
  767. self.stderr.extend(self.process.stderr)
  768. def assert_session_log(sessions: Path, cwd: Path, *expected_texts: str) -> None:
  769. logs = list(sessions.rglob("*.jsonl"))
  770. if len(logs) != 1:
  771. raise AssertionError(f"expected one JSONL session log under {sessions}, found {logs}")
  772. lines = logs[0].read_text().splitlines()
  773. header = json.loads(lines[0])
  774. if header.get("cwd") != str(cwd):
  775. raise AssertionError(f"session header cwd is not absolute/canonical: {header}")
  776. rendered = "\n".join(lines)
  777. for expected in expected_texts:
  778. if expected not in rendered:
  779. raise AssertionError(f"session log has no {expected!r} response: {logs[0]}")
  780. def assert_zstd_session_log(sessions: Path) -> None:
  781. logs = list(sessions.rglob("*.jsonl.zstd"))
  782. if len(logs) != 1:
  783. raise AssertionError(f"expected one Zstandard JSONL session log under {sessions}, found {logs}")
  784. if not logs[0].read_bytes().startswith(bytes.fromhex("28b52ffd")):
  785. raise AssertionError(f"session log has no Zstandard magic: {logs[0]}")
  786. def read_session_logs(sessions: Path) -> dict[str, list[dict[str, object]]]:
  787. """Parse every persisted JSONL session into a map keyed by header id."""
  788. logs: dict[str, list[dict[str, object]]] = {}
  789. for path in sorted(sessions.rglob("*.jsonl")):
  790. records = [
  791. json.loads(line)
  792. for line in path.read_text(encoding="utf-8").splitlines()
  793. if line
  794. ]
  795. if not records or records[0].get("type") != "session":
  796. raise AssertionError(f"session log has no header: {path}")
  797. session_id = records[0].get("id")
  798. if not isinstance(session_id, str):
  799. raise AssertionError(f"session log header has no string id: {path}")
  800. if session_id in logs:
  801. raise AssertionError(f"duplicate persisted session id: {session_id}")
  802. logs[session_id] = records
  803. return logs
  804. def snapshot_child_ids(result: "RunResult") -> list[str]:
  805. """Return the two child session ids in their SDK notification order."""
  806. child_ids: list[str] = []
  807. for notification in result.notifications:
  808. if notification.method != "subagent.started":
  809. continue
  810. payload = notification.payload
  811. if payload.get("parentSessionId") != SNAPSHOT_SESSION_ID:
  812. continue
  813. child_id = payload.get("childSessionId")
  814. if isinstance(child_id, str) and child_id not in child_ids:
  815. child_ids.append(child_id)
  816. if len(child_ids) != 2:
  817. raise AssertionError(f"advanced snapshot expected two child session ids: {child_ids}")
  818. return child_ids
  819. def build_snapshot_files(
  820. result: "RunResult",
  821. logs: dict[str, list[dict[str, object]]],
  822. child_ids: list[str],
  823. cwd: Path,
  824. ) -> dict[str, str]:
  825. """Render the SDK result and three persisted logs into stable expected outputs."""
  826. replacements = [(str(cwd), "{{cwd}}"), (SNAPSHOT_SESSION_ID, "{{parent}}")]
  827. replacements.append((snapshot_workflow_run_id(result), "{{workflow-run}}"))
  828. for index, child_id in enumerate(child_ids, start=1):
  829. replacements.append((child_id, f"{{{{child-{index}}}}}"))
  830. agent_id = snapshot_agent_id(result, child_id)
  831. replacements.append((agent_id, f"{{{{agent-{index}}}}}"))
  832. replacements.sort(key=lambda pair: len(pair[0]), reverse=True)
  833. result_value = {
  834. "session_id": result.session_id,
  835. "final_response": result.final_response,
  836. "events": result.events,
  837. "notifications": [
  838. {"method": notification.method, "payload": notification.payload}
  839. for notification in result.notifications
  840. ],
  841. "session_root": result.session_root,
  842. }
  843. normalized_result = normalize_snapshot_value(result_value, replacements)
  844. files = {
  845. "result.json": json.dumps(normalized_result, indent=2, ensure_ascii=False) + "\n",
  846. "session.jsonl": render_jsonl(
  847. [normalize_snapshot_value(record, replacements) for record in logs[SNAPSHOT_SESSION_ID]]
  848. ),
  849. }
  850. for index, child_id in enumerate(child_ids, start=1):
  851. files[f"session.{index}.jsonl"] = render_jsonl(
  852. [normalize_snapshot_value(record, replacements) for record in logs[child_id]]
  853. )
  854. if tuple(files) != SNAPSHOT_FILENAMES:
  855. raise AssertionError(f"advanced snapshot file set drifted: {tuple(files)}")
  856. return files
  857. def snapshot_workflow_run_id(result: "RunResult") -> str:
  858. """Return the one workflow run id emitted by the advanced scenario."""
  859. run_ids: set[str] = set()
  860. for event in result.events:
  861. event_type = event.get("type")
  862. data = event.get("data")
  863. if not isinstance(event_type, str) or not event_type.startswith("tool-workflow/"):
  864. continue
  865. if isinstance(data, dict) and isinstance(data.get("runId"), str):
  866. run_ids.add(data["runId"])
  867. if len(run_ids) != 1:
  868. raise AssertionError(f"advanced snapshot expected one workflow run id: {sorted(run_ids)}")
  869. return next(iter(run_ids))
  870. def snapshot_agent_id(result: "RunResult", child_id: str) -> str:
  871. """Find the successful subagent id paired with one child session."""
  872. for notification in result.notifications:
  873. if notification.method != "subagent.finished":
  874. continue
  875. payload = notification.payload
  876. if payload.get("childSessionId") != child_id:
  877. continue
  878. if payload.get("provider") != "spawn" or payload.get("status") != "ok":
  879. raise AssertionError(f"advanced child did not finish successfully: {payload}")
  880. agent_id = payload.get("agentId")
  881. if isinstance(agent_id, str):
  882. return agent_id
  883. raise AssertionError(f"advanced snapshot has no finished agent for child {child_id}")
  884. def normalize_snapshot_value(
  885. value: object,
  886. replacements: list[tuple[str, str]],
  887. ) -> object:
  888. """Scrub volatile values and bulky request headers without losing behavior."""
  889. if isinstance(value, str):
  890. normalized = value
  891. for actual, token in replacements:
  892. normalized = normalized.replace(actual, token)
  893. return normalized
  894. if isinstance(value, list):
  895. return [normalize_snapshot_value(item, replacements) for item in value]
  896. if not isinstance(value, dict):
  897. return value
  898. normalized = {
  899. key: normalize_snapshot_value(item, replacements)
  900. for key, item in value.items()
  901. }
  902. if normalized.get("type") == "session" and "createdAt" in normalized:
  903. normalized["createdAt"] = 0
  904. if "seq" in normalized and "time" in normalized:
  905. normalized["time"] = 0
  906. if isinstance(normalized.get("id"), str) and normalized.get("role") in ("assistant", "user"):
  907. normalized["id"] = "{{messageId}}"
  908. scrub_snapshot_header(normalized)
  909. return normalized
  910. def scrub_snapshot_header(value: dict[object, object]) -> None:
  911. """Tokenize full request-header bulk while retaining tool names."""
  912. data = value.get("data")
  913. if not isinstance(data, dict):
  914. return
  915. if value.get("type") == "request/header":
  916. header = data.get("header")
  917. if not isinstance(header, dict):
  918. return
  919. if "system" in header:
  920. header["system"] = "{{system}}"
  921. tools = header.get("tools")
  922. if isinstance(tools, list):
  923. header["tools"] = [
  924. tool.get("name") if isinstance(tool, dict) else "{{tools}}"
  925. for tool in tools
  926. ]
  927. def render_jsonl(records: list[object]) -> str:
  928. """Render parsed JSON values as compact, newline-terminated JSONL."""
  929. return "".join(
  930. json.dumps(record, ensure_ascii=False, separators=(",", ":")) + "\n"
  931. for record in records
  932. )
  933. def compare_snapshot_files(files: dict[str, str], update: bool) -> None:
  934. """Write or exactly compare the advanced executable snapshot files."""
  935. if update:
  936. SNAPSHOT_DIRECTORY.mkdir(parents=True, exist_ok=True)
  937. for name, content in files.items():
  938. (SNAPSHOT_DIRECTORY / name).write_text(content, encoding="utf-8")
  939. print(f"smoke-python-runtime: updated snapshots in {SNAPSHOT_DIRECTORY}")
  940. existing = {
  941. path.name
  942. for path in SNAPSHOT_DIRECTORY.iterdir()
  943. if path.is_file()
  944. } if SNAPSHOT_DIRECTORY.is_dir() else set()
  945. expected = set(SNAPSHOT_FILENAMES)
  946. if existing != expected:
  947. raise AssertionError(
  948. "advanced snapshot files differ: "
  949. f"missing={sorted(expected - existing)}, unexpected={sorted(existing - expected)}"
  950. )
  951. for name, actual in files.items():
  952. expected_text = (SNAPSHOT_DIRECTORY / name).read_text(encoding="utf-8")
  953. if actual == expected_text:
  954. continue
  955. diff = "".join(difflib.unified_diff(
  956. expected_text.splitlines(keepends=True),
  957. actual.splitlines(keepends=True),
  958. fromfile=f"expected/{name}",
  959. tofile=f"actual/{name}",
  960. ))
  961. raise AssertionError(
  962. f"advanced executable snapshot mismatch in {name}; "
  963. "rerun with --update-snapshots after reviewing the behavior\n"
  964. f"{diff}"
  965. )
  966. if __name__ == "__main__":
  967. main()