characterise-loops.py 14 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379
  1. #!/usr/bin/env python3
  2. """Characterisation tests for loop execution.
  3. Runs a set of workflows that exercise the loop walk and writes a normalised
  4. summary of what each one did. The point is NOT that the recorded behaviour is
  5. correct - it is to notice when a change to the engine alters any of it, so
  6. every difference has to be looked at and justified rather than discovered
  7. later in production.
  8. scripts/characterise-loops.py before.json # before your change
  9. ... rebuild the runner and restart it ...
  10. scripts/characterise-loops.py after.json # after it
  11. diff before.json after.json
  12. This exists because the node suite cannot catch this class of change. When the
  13. two graph walks were unified (2026-08-13), four behaviours differed between a
  14. node inside a loop and the same node outside one; three of the four were
  15. invisible to all 92 node tests, which passed green throughout. What caught them
  16. was diffing these baselines.
  17. Needs the webserver and a runner up, and logs in as admin/admin. Every workflow
  18. it creates is named "zz char: ..." and is deleted again, including on failure.
  19. """
  20. import json, sys, time, urllib.request
  21. API = "http://localhost:8090/api/v1"
  22. def req(method, path, body=None, token=None):
  23. data = json.dumps(body).encode() if body is not None else None
  24. r = urllib.request.Request(API + path, data=data, method=method,
  25. headers={"Content-Type": "application/json"})
  26. if token:
  27. r.add_header("Authorization", "Bearer " + token)
  28. return json.loads(urllib.request.urlopen(r).read())
  29. TOKEN = req("POST", "/auth/login", {"username": "admin", "password": "admin"})["accessToken"]
  30. def conn(src, tgt, src_out="main", tgt_in="data"):
  31. return {"sourceNodeId": src, "sourceOutput": src_out,
  32. "targetNodeId": tgt, "targetInput": tgt_in}
  33. def seed(items, node_id="seed"):
  34. return {"id": node_id, "type": "code", "name": node_id, "position": {"x": 0, "y": 0},
  35. "config": {"code": "return { items: %s }" % json.dumps(items)}}
  36. def loop_node(node_id="loop", field="result.items", cont=True, x=240):
  37. return {"id": node_id, "type": "loop", "name": node_id, "position": {"x": x, "y": 0},
  38. "config": {"inputField": field, "continueOnError": cont}}
  39. def setf(node_id, name, value, x=480, disabled=False):
  40. n = {"id": node_id, "type": "set-fields", "name": node_id, "position": {"x": x, "y": 0},
  41. "config": {"fields": [{"name": name, "value": value, "type": "string"}]}}
  42. if disabled:
  43. n["disabled"] = True
  44. return n
  45. CASES = {}
  46. def case(name):
  47. def deco(fn):
  48. CASES[name] = fn
  49. return fn
  50. return deco
  51. @case("simple_three_items_two_body_nodes")
  52. def _():
  53. return {
  54. "nodes": [
  55. seed([{"n": "a"}, {"n": "b"}, {"n": "c"}]),
  56. loop_node(),
  57. setf("first", "seen", "{{ loop.item.n }}"),
  58. # The item must still be readable on the SECOND body node.
  59. setf("second", "alsoSeen", "{{ loop.item.n }}-{{ loop.index }}", x=720),
  60. setf("after", "done", "yes", x=480),
  61. ],
  62. "connections": [
  63. conn("seed", "loop"), conn("loop", "first", "loop"),
  64. conn("first", "second"), conn("loop", "after", "done"),
  65. ],
  66. }
  67. @case("nested_loop_two_by_two")
  68. def _():
  69. return {
  70. "nodes": [
  71. seed([{"n": "x"}, {"n": "y"}]),
  72. loop_node("outer"),
  73. {"id": "inner_seed", "type": "code", "name": "inner_seed",
  74. "position": {"x": 480, "y": 0},
  75. "config": {"code": "return { sub: [{m:'1'}, {m:'2'}] }"}},
  76. loop_node("inner", "result.sub", x=720),
  77. setf("leaf", "pair", "{{ loop.item.m }}", x=960),
  78. setf("after", "done", "yes", x=480),
  79. ],
  80. "connections": [
  81. conn("seed", "outer"), conn("outer", "inner_seed", "loop"),
  82. conn("inner_seed", "inner"), conn("inner", "leaf", "loop"),
  83. conn("outer", "after", "done"),
  84. ],
  85. }
  86. @case("branch_gating_in_body")
  87. def _():
  88. return {
  89. "nodes": [
  90. seed([{"n": "a"}, {"n": "b"}]),
  91. loop_node(),
  92. setf("mark", "seen", "{{ loop.item.n }}"),
  93. {"id": "gate", "type": "if-condition", "name": "gate", "position": {"x": 720, "y": 0},
  94. "config": {"conditions": [{"field": "data.seen", "operator": "equals", "value": "a"}]}},
  95. setf("on_true", "took", "true", x=960),
  96. setf("on_false", "took", "false", x=960),
  97. setf("after", "done", "yes", x=480),
  98. ],
  99. "connections": [
  100. conn("seed", "loop"), conn("loop", "mark", "loop"), conn("mark", "gate"),
  101. conn("gate", "on_true", "true"), conn("gate", "on_false", "false"),
  102. conn("loop", "after", "done"),
  103. ],
  104. }
  105. @case("back_edge_true_port_only")
  106. def _():
  107. return {
  108. "nodes": [
  109. seed([{"n": "a"}, {"n": "b"}, {"n": "c"}]),
  110. loop_node(),
  111. setf("mark", "seen", "{{ loop.item.n }}"),
  112. {"id": "gate", "type": "if-condition", "name": "gate", "position": {"x": 720, "y": 0},
  113. "config": {"conditions": [{"field": "data.seen", "operator": "not_equals", "value": "a"}]}},
  114. setf("after", "done", "yes", x=480),
  115. ],
  116. "connections": [
  117. conn("seed", "loop"), conn("loop", "mark", "loop"), conn("mark", "gate"),
  118. conn("gate", "loop", "true"), conn("loop", "after", "done"),
  119. ],
  120. }
  121. @case("disabled_node_in_body_is_transparent")
  122. def _():
  123. return {
  124. "nodes": [
  125. seed([{"n": "a"}, {"n": "b"}]),
  126. loop_node(),
  127. setf("mark", "seen", "{{ loop.item.n }}"),
  128. setf("off", "ignored", "x", x=720, disabled=True),
  129. setf("tail", "reached", "yes", x=960),
  130. setf("after", "done", "yes", x=480),
  131. ],
  132. "connections": [
  133. conn("seed", "loop"), conn("loop", "mark", "loop"),
  134. conn("mark", "off"), conn("off", "tail"), conn("loop", "after", "done"),
  135. ],
  136. }
  137. @case("body_failure_continue_on_error_true")
  138. def _():
  139. return {
  140. "nodes": [
  141. seed([{"n": "a"}, {"n": "b"}, {"n": "c"}]),
  142. loop_node(cont=True),
  143. # NOTE: a code node's body cannot see `loop` - it is available to
  144. # expressions, not as a variable in the sandbox - so this throws on
  145. # every item, not only on 'b'. Left as it is because a baseline only
  146. # has to be stable and this one exercises "every item failed, run
  147. # completed anyway"; do not read the case name as a promise that
  148. # exactly one item fails.
  149. {"id": "boom", "type": "code", "name": "boom", "position": {"x": 480, "y": 0},
  150. "config": {"code": "if (input.item.n === 'b') { throw new Error('planned'); } return { ok: input.item.n }"}},
  151. setf("after", "done", "yes", x=480),
  152. ],
  153. "connections": [
  154. conn("seed", "loop"), conn("loop", "boom", "loop"), conn("loop", "after", "done"),
  155. ],
  156. }
  157. @case("body_failure_continue_on_error_false")
  158. def _():
  159. return {
  160. "nodes": [
  161. seed([{"n": "a"}, {"n": "b"}, {"n": "c"}]),
  162. loop_node(cont=False),
  163. {"id": "boom", "type": "code", "name": "boom", "position": {"x": 480, "y": 0},
  164. "config": {"code": "if (input.item.n === 'b') { throw new Error('planned'); } return { ok: input.item.n }"}},
  165. setf("after", "done", "yes", x=480),
  166. ],
  167. "connections": [
  168. conn("seed", "loop"), conn("loop", "boom", "loop"), conn("loop", "after", "done"),
  169. ],
  170. }
  171. # continue_on_error is a per-item policy, not a per-run one. A loop that
  172. # tolerates failures should still fail the run when NOTHING succeeded - a dead
  173. # dependency made every item fail for hours while the run reported success.
  174. # A code node reaches the loop item through `input` - `input.item`, `input.index`,
  175. # `input.loop`. There is no bare `loop` global; using one throws "'loop' is not
  176. # defined" on EVERY item, which is how two cases here spent their life asserting
  177. # "one item fails" while actually testing "all of them do".
  178. @case("tolerated_all_items_fail_fails_the_run")
  179. def _():
  180. return {
  181. "nodes": [
  182. seed([{"n": "a"}, {"n": "b"}, {"n": "c"}]),
  183. loop_node(cont=True),
  184. {"id": "boom", "type": "code", "name": "boom", "position": {"x": 480, "y": 0},
  185. "config": {"code": "throw new Error('every item fails');"}},
  186. setf("after", "done", "yes", x=480),
  187. ],
  188. "connections": [
  189. conn("seed", "loop"), conn("loop", "boom", "loop"), conn("loop", "after", "done"),
  190. ],
  191. }
  192. # The other half of the same rule: tolerating SOME failures must be unchanged.
  193. @case("tolerated_some_items_fail_still_succeeds")
  194. def _():
  195. return {
  196. "nodes": [
  197. seed([{"n": "a"}, {"n": "b"}, {"n": "c"}]),
  198. loop_node(cont=True),
  199. {"id": "boom", "type": "code", "name": "boom", "position": {"x": 480, "y": 0},
  200. "config": {"code": "if (input.item.n === 'b') { throw new Error('planned'); } return { ok: input.item.n }"}},
  201. setf("after", "done", "yes", x=480),
  202. ],
  203. "connections": [
  204. conn("seed", "loop"), conn("loop", "boom", "loop"), conn("loop", "after", "done"),
  205. ],
  206. }
  207. def _halt_in_loop(mode):
  208. """A loop of three whose middle item trips a stop-and-error."""
  209. return {
  210. "nodes": [
  211. seed([{"n": "a"}, {"n": "b"}, {"n": "c"}]),
  212. loop_node(),
  213. setf("mark", "seen", "{{ loop.item.n }}"),
  214. {"id": "gate", "type": "if-condition", "name": "gate", "position": {"x": 720, "y": 0},
  215. "config": {"conditions": [{"field": "data.seen", "operator": "equals", "value": "b"}]}},
  216. {"id": "halt", "type": "stop-and-error", "name": "halt",
  217. "position": {"x": 960, "y": -80},
  218. "config": {"mode": mode, "message": "item b says stop"}},
  219. setf("keep", "kept", "yes", x=960),
  220. setf("after", "done", "yes", x=480),
  221. ],
  222. "connections": [
  223. conn("seed", "loop"), conn("loop", "mark", "loop"), conn("mark", "gate"),
  224. conn("gate", "halt", "true"), conn("gate", "keep", "false"),
  225. conn("loop", "after", "done"),
  226. ],
  227. }
  228. # The three ways a body node can end something, which are three different
  229. # things: fail the whole run, end the whole run cleanly, and drop this one item
  230. # and carry on. error mode used to do none of them inside a loop - it threw, and
  231. # the loop's Continue On Error caught it - so all three are pinned here.
  232. @case("halt_in_loop_error_fails_the_run")
  233. def _():
  234. return _halt_in_loop("error")
  235. @case("halt_in_loop_stop_ends_the_run")
  236. def _():
  237. return _halt_in_loop("stop")
  238. @case("halt_in_loop_skip_drops_one_item")
  239. def _():
  240. return _halt_in_loop("skip")
  241. @case("empty_item_list")
  242. def _():
  243. return {
  244. "nodes": [
  245. seed([]),
  246. loop_node(),
  247. setf("mark", "seen", "{{ loop.item.n }}"),
  248. setf("after", "done", "yes", x=480),
  249. ],
  250. "connections": [
  251. conn("seed", "loop"), conn("loop", "mark", "loop"), conn("loop", "after", "done"),
  252. ],
  253. }
  254. def run_case(name, spec):
  255. wf = req("POST", "/workflows", {"name": "zz char: " + name, "active": False,
  256. "nodes": spec["nodes"],
  257. "connections": spec["connections"]}, TOKEN)
  258. wf_id = wf.get("_id") or wf.get("id")
  259. try:
  260. ex = req("POST", f"/workflows/{wf_id}/execute", {}, TOKEN)
  261. exec_id = ex.get("executionId")
  262. detail = {}
  263. for _ in range(40):
  264. time.sleep(1.5)
  265. detail = req("GET", "/executions/" + exec_id, token=TOKEN)
  266. if detail.get("status") not in ("running", "pending"):
  267. break
  268. def drop_timings(value):
  269. """executionTime is how long a node took, not what it did.
  270. It wobbles between 0 and 1 ms run to run, and it appears nested
  271. inside collected loop results too - so left in, four of thirteen
  272. cases differ on every single diff and a real change hides among
  273. them. Timing belongs in a benchmark, not a characterisation.
  274. """
  275. if isinstance(value, dict):
  276. return {k: drop_timings(v) for k, v in value.items()
  277. if k != "executionTime"}
  278. if isinstance(value, list):
  279. return [drop_timings(v) for v in value]
  280. return value
  281. records = []
  282. for rec in detail.get("nodeExecutions", []):
  283. out = rec.get("output")
  284. # Only the fields a workflow author would see, and drop the engine's
  285. # internal loop bookkeeping, which is noisy and not behaviour.
  286. if isinstance(out, dict):
  287. out = {k: v for k, v in out.items() if not k.startswith("_")}
  288. out = drop_timings(out)
  289. records.append({
  290. "node": rec.get("nodeId"),
  291. "status": rec.get("status"),
  292. "iteration": rec.get("loopIteration"),
  293. "loopNode": rec.get("loopNodeId"),
  294. "error": (rec.get("error") or "").split("\n")[0][:110],
  295. "output": out,
  296. })
  297. # Sorted so a run-to-run ordering wobble is not mistaken for a change.
  298. records.sort(key=lambda r: (r["node"], str(r["iteration"])))
  299. return {"status": detail.get("status"),
  300. "error": (detail.get("error") or "").split("\n")[0][:110],
  301. "stopped": detail.get("stopped"),
  302. "stopReason": detail.get("stopReason"),
  303. "toleratedErrorCount": detail.get("toleratedErrorCount"),
  304. "records": records}
  305. finally:
  306. urllib.request.urlopen(urllib.request.Request(
  307. API + "/workflows/" + wf_id, method="DELETE",
  308. headers={"Authorization": "Bearer " + TOKEN}))
  309. out = {}
  310. for name in sorted(CASES):
  311. print("running", name, flush=True)
  312. try:
  313. out[name] = run_case(name, CASES[name]())
  314. except Exception as e:
  315. out[name] = {"harness_error": repr(e)}
  316. path = sys.argv[1] if len(sys.argv) > 1 else "loop_characterisation.json"
  317. with open(path, "w") as f:
  318. json.dump(out, f, indent=2, sort_keys=True)
  319. print("\nwrote", path)
  320. for name in sorted(out):
  321. r = out[name]
  322. print(" %-42s %s" % (name, r.get("status") or r.get("harness_error")))