1
0

test-subagent-driven-development-integration.sh 9.3 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314
  1. #!/usr/bin/env bash
  2. # Integration Test: subagent-driven-development workflow
  3. # Actually executes a plan and verifies the new workflow behaviors
  4. set -euo pipefail
  5. SCRIPT_DIR="$(cd "$(dirname "$0")" && pwd)"
  6. source "$SCRIPT_DIR/test-helpers.sh"
  7. echo "========================================"
  8. echo " Integration Test: subagent-driven-development"
  9. echo "========================================"
  10. echo ""
  11. echo "This test executes a real plan using the skill and verifies:"
  12. echo " 1. Plan is read once (not per task)"
  13. echo " 2. Full task text provided to subagents"
  14. echo " 3. Subagents perform self-review"
  15. echo " 4. Spec compliance review before code quality"
  16. echo " 5. Review loops when issues found"
  17. echo " 6. Spec reviewer reads code independently"
  18. echo ""
  19. echo "WARNING: This test may take 10-30 minutes to complete."
  20. echo ""
  21. # Create test project
  22. TEST_PROJECT=$(create_test_project)
  23. echo "Test project: $TEST_PROJECT"
  24. # Trap to cleanup
  25. trap "cleanup_test_project $TEST_PROJECT" EXIT
  26. # Set up minimal Node.js project
  27. cd "$TEST_PROJECT"
  28. cat > package.json <<'EOF'
  29. {
  30. "name": "test-project",
  31. "version": "1.0.0",
  32. "type": "module",
  33. "scripts": {
  34. "test": "node --test"
  35. }
  36. }
  37. EOF
  38. mkdir -p src test docs/superpowers/plans
  39. # Create a simple implementation plan
  40. cat > docs/superpowers/plans/implementation-plan.md <<'EOF'
  41. # Test Implementation Plan
  42. This is a minimal plan to test the subagent-driven-development workflow.
  43. ## Task 1: Create Add Function
  44. Create a function that adds two numbers.
  45. **File:** `src/math.js`
  46. **Requirements:**
  47. - Function named `add`
  48. - Takes two parameters: `a` and `b`
  49. - Returns the sum of `a` and `b`
  50. - Export the function
  51. **Implementation:**
  52. ```javascript
  53. export function add(a, b) {
  54. return a + b;
  55. }
  56. ```
  57. **Tests:** Create `test/math.test.js` that verifies:
  58. - `add(2, 3)` returns `5`
  59. - `add(0, 0)` returns `0`
  60. - `add(-1, 1)` returns `0`
  61. **Verification:** `npm test`
  62. ## Task 2: Create Multiply Function
  63. Create a function that multiplies two numbers.
  64. **File:** `src/math.js` (add to existing file)
  65. **Requirements:**
  66. - Function named `multiply`
  67. - Takes two parameters: `a` and `b`
  68. - Returns the product of `a` and `b`
  69. - Export the function
  70. - DO NOT add any extra features (like power, divide, etc.)
  71. **Implementation:**
  72. ```javascript
  73. export function multiply(a, b) {
  74. return a * b;
  75. }
  76. ```
  77. **Tests:** Add to `test/math.test.js`:
  78. - `multiply(2, 3)` returns `6`
  79. - `multiply(0, 5)` returns `0`
  80. - `multiply(-2, 3)` returns `-6`
  81. **Verification:** `npm test`
  82. EOF
  83. # Initialize git repo
  84. git init --quiet
  85. git config user.email "test@test.com"
  86. git config user.name "Test User"
  87. git add .
  88. git commit -m "Initial commit" --quiet
  89. echo ""
  90. echo "Project setup complete. Starting execution..."
  91. echo ""
  92. # Run Claude with subagent-driven-development
  93. # Capture full output to analyze
  94. OUTPUT_FILE="$TEST_PROJECT/claude-output.txt"
  95. # Create prompt file
  96. cat > "$TEST_PROJECT/prompt.txt" <<'EOF'
  97. I want you to execute the implementation plan at docs/superpowers/plans/implementation-plan.md using the subagent-driven-development skill.
  98. IMPORTANT: Follow the skill exactly. I will be verifying that you:
  99. 1. Read the plan once at the beginning
  100. 2. Provide full task text to subagents (don't make them read files)
  101. 3. Ensure subagents do self-review before reporting
  102. 4. Run spec compliance review before code quality review
  103. 5. Use review loops when issues are found
  104. Begin now. Execute the plan.
  105. EOF
  106. # Note: We use a longer timeout since this is integration testing
  107. # Use --allowed-tools to enable tool usage in headless mode
  108. # IMPORTANT: Run from superpowers directory so local dev skills are available
  109. PROMPT="Change to directory $TEST_PROJECT and then execute the implementation plan at docs/superpowers/plans/implementation-plan.md using the subagent-driven-development skill.
  110. IMPORTANT: Follow the skill exactly. I will be verifying that you:
  111. 1. Read the plan once at the beginning
  112. 2. Provide full task text to subagents (don't make them read files)
  113. 3. Ensure subagents do self-review before reporting
  114. 4. Run spec compliance review before code quality review
  115. 5. Use review loops when issues are found
  116. Begin now. Execute the plan."
  117. echo "Running Claude (output will be shown below and saved to $OUTPUT_FILE)..."
  118. echo "================================================================================"
  119. cd "$SCRIPT_DIR/../.." && timeout 1800 claude -p "$PROMPT" --allowed-tools=all --add-dir "$TEST_PROJECT" --permission-mode bypassPermissions 2>&1 | tee "$OUTPUT_FILE" || {
  120. echo ""
  121. echo "================================================================================"
  122. echo "EXECUTION FAILED (exit code: $?)"
  123. exit 1
  124. }
  125. echo "================================================================================"
  126. echo ""
  127. echo "Execution complete. Analyzing results..."
  128. echo ""
  129. # Find the session transcript
  130. # Session files are in ~/.claude/projects/-<working-dir>/<session-id>.jsonl
  131. WORKING_DIR_ESCAPED=$(echo "$SCRIPT_DIR/../.." | sed 's/\//-/g' | sed 's/^-//')
  132. SESSION_DIR="$HOME/.claude/projects/$WORKING_DIR_ESCAPED"
  133. # Find the most recent session file (created during this test run)
  134. SESSION_FILE=$(find "$SESSION_DIR" -name "*.jsonl" -type f -mmin -60 2>/dev/null | sort -r | head -1)
  135. if [ -z "$SESSION_FILE" ]; then
  136. echo "ERROR: Could not find session transcript file"
  137. echo "Looked in: $SESSION_DIR"
  138. exit 1
  139. fi
  140. echo "Analyzing session transcript: $(basename "$SESSION_FILE")"
  141. echo ""
  142. # Verification tests
  143. FAILED=0
  144. echo "=== Verification Tests ==="
  145. echo ""
  146. # Test 1: Skill was invoked
  147. echo "Test 1: Skill tool invoked..."
  148. if grep -q '"name":"Skill".*"skill":"superpowers:subagent-driven-development"' "$SESSION_FILE"; then
  149. echo " [PASS] subagent-driven-development skill was invoked"
  150. else
  151. echo " [FAIL] Skill was not invoked"
  152. FAILED=$((FAILED + 1))
  153. fi
  154. echo ""
  155. # Test 2: Subagents were used (Task tool)
  156. echo "Test 2: Subagents dispatched..."
  157. task_count=$(grep -c '"name":"Task"' "$SESSION_FILE" || echo "0")
  158. if [ "$task_count" -ge 2 ]; then
  159. echo " [PASS] $task_count subagents dispatched"
  160. else
  161. echo " [FAIL] Only $task_count subagent(s) dispatched (expected >= 2)"
  162. FAILED=$((FAILED + 1))
  163. fi
  164. echo ""
  165. # Test 3: TodoWrite was used for tracking
  166. echo "Test 3: Task tracking..."
  167. todo_count=$(grep -c '"name":"TodoWrite"' "$SESSION_FILE" || echo "0")
  168. if [ "$todo_count" -ge 1 ]; then
  169. echo " [PASS] TodoWrite used $todo_count time(s) for task tracking"
  170. else
  171. echo " [FAIL] TodoWrite not used"
  172. FAILED=$((FAILED + 1))
  173. fi
  174. echo ""
  175. # Test 6: Implementation actually works
  176. echo "Test 6: Implementation verification..."
  177. if [ -f "$TEST_PROJECT/src/math.js" ]; then
  178. echo " [PASS] src/math.js created"
  179. if grep -q "export function add" "$TEST_PROJECT/src/math.js"; then
  180. echo " [PASS] add function exists"
  181. else
  182. echo " [FAIL] add function missing"
  183. FAILED=$((FAILED + 1))
  184. fi
  185. if grep -q "export function multiply" "$TEST_PROJECT/src/math.js"; then
  186. echo " [PASS] multiply function exists"
  187. else
  188. echo " [FAIL] multiply function missing"
  189. FAILED=$((FAILED + 1))
  190. fi
  191. else
  192. echo " [FAIL] src/math.js not created"
  193. FAILED=$((FAILED + 1))
  194. fi
  195. if [ -f "$TEST_PROJECT/test/math.test.js" ]; then
  196. echo " [PASS] test/math.test.js created"
  197. else
  198. echo " [FAIL] test/math.test.js not created"
  199. FAILED=$((FAILED + 1))
  200. fi
  201. # Try running tests
  202. if cd "$TEST_PROJECT" && npm test > test-output.txt 2>&1; then
  203. echo " [PASS] Tests pass"
  204. else
  205. echo " [FAIL] Tests failed"
  206. cat test-output.txt
  207. FAILED=$((FAILED + 1))
  208. fi
  209. echo ""
  210. # Test 7: Git commits show proper workflow
  211. echo "Test 7: Git commit history..."
  212. commit_count=$(git -C "$TEST_PROJECT" log --oneline | wc -l)
  213. if [ "$commit_count" -gt 2 ]; then # Initial + at least 2 task commits
  214. echo " [PASS] Multiple commits created ($commit_count total)"
  215. else
  216. echo " [FAIL] Too few commits ($commit_count, expected >2)"
  217. FAILED=$((FAILED + 1))
  218. fi
  219. echo ""
  220. # Test 8: Check for extra features (spec compliance should catch)
  221. echo "Test 8: No extra features added (spec compliance)..."
  222. if grep -q "export function divide\|export function power\|export function subtract" "$TEST_PROJECT/src/math.js" 2>/dev/null; then
  223. echo " [WARN] Extra features found (spec review should have caught this)"
  224. # Not failing on this as it tests reviewer effectiveness
  225. else
  226. echo " [PASS] No extra features added"
  227. fi
  228. echo ""
  229. # Token Usage Analysis
  230. echo "========================================="
  231. echo " Token Usage Analysis"
  232. echo "========================================="
  233. echo ""
  234. python3 "$SCRIPT_DIR/analyze-token-usage.py" "$SESSION_FILE"
  235. echo ""
  236. # Summary
  237. echo "========================================"
  238. echo " Test Summary"
  239. echo "========================================"
  240. echo ""
  241. if [ $FAILED -eq 0 ]; then
  242. echo "STATUS: PASSED"
  243. echo "All verification tests passed!"
  244. echo ""
  245. echo "The subagent-driven-development skill correctly:"
  246. echo " ✓ Reads plan once at start"
  247. echo " ✓ Provides full task text to subagents"
  248. echo " ✓ Enforces self-review"
  249. echo " ✓ Runs spec compliance before code quality"
  250. echo " ✓ Spec reviewer verifies independently"
  251. echo " ✓ Produces working implementation"
  252. exit 0
  253. else
  254. echo "STATUS: FAILED"
  255. echo "Failed $FAILED verification tests"
  256. echo ""
  257. echo "Output saved to: $OUTPUT_FILE"
  258. echo ""
  259. echo "Review the output to see what went wrong."
  260. exit 1
  261. fi