Просмотр исходного кода

feat: end a loop iteration by not wiring it back, and add an sdcpp health node

WIRING A LOOP BACK

A loop body can now be closed explicitly - the last body node connects back to
the loop node - and a branch that does NOT lead back ends the iteration. The
loop still finishes normally: its done output fires with the results collected
so far and the nodes after the loop run. So an If inside the body with only its
false branch wired back is how a loop stops early, which is what the connection
already looks like on the canvas.

This replaces a Break Loop node that was written and then removed: a node that
did invisibly what an edge can say visibly, and worse, it carried a condition
option whose expression came through as undefined rather than false when it
failed to resolve, so a break meant for the third item fired on the first.

The rule only applies when at least one body node is wired back. A loop that was
never closed keeps iterating everything, because that is how every workflow
written before this behaved and turning them all into one-item loops silently
would have been the worst possible outcome. That nearly happened: the first
version counted any edge targeting the loop node, which includes the incoming
edge carrying the loop its data - so every loop looked explicitly wired back and
two existing fixtures stopped after one item. A back edge now has to come from a
node inside the body.

Only a node that actually ran counts. A node on a branch that was not taken is
skipped, and a skipped loop-back edge is exactly the case that should end the
iteration.

SD.CPP HEALTH NODE

Checks reachability and what the server has loaded, over /health - the one
endpoint sdcpp-restapi leaves unauthenticated, so it needs no credential.

It does not throw by default. A health check exists to report that a server is
not answering, so returning reachable false to branch on is more useful than
failing the run; failIfUnreachable turns that around. A refused connection and
an HTTP 500 are reported differently - a missing server and a broken one call
for different responses.

requireModelLoaded and requireUpscalerLoaded exist because a reachable server
with an empty model slot rejects generation jobs, so a check that only pings
would go green immediately before the work it guards fails.

Verified against the live server: reachable reports healthy with architecture
SD 1.x, an unreachable address reports reachable false with the connection
error, an explicit loop-back stops after 2 of 5 items with the nodes after the
loop still running, and loops with no back edge still process every item. Full
suite 48/48.
fszontagh 1 месяц назад
Родитель
Сommit
2a14309fb1

+ 231 - 0
nodes/sdcpp/sdcpp-health.js

@@ -0,0 +1,231 @@
+/**
+ * @node sdcpp-health
+ * @name SD.cpp Health
+ * @category sdcpp
+ * @version 1.0.0
+ * @description Check whether an sdcpp-restapi server is reachable and what it currently has loaded
+ * @icon heart-pulse
+ */
+
+const configSchema = {
+    type: 'object',
+    properties: {
+        serverUrl: {
+            type: 'string',
+            title: 'Server URL',
+            description: 'Base address of the sdcpp-restapi server, such as http://localhost:8077',
+            default: 'http://localhost:8077'
+        },
+        failIfUnreachable: {
+            type: 'boolean',
+            title: 'Fail If Unreachable',
+            description: 'Throw when the server cannot be reached. Leave off to return reachable false and branch on it, which is the point of a health check',
+            default: false
+        },
+        requireModelLoaded: {
+            type: 'boolean',
+            title: 'Require A Loaded Model',
+            description: 'Treat a reachable server with no model in its slot as unhealthy. A generation job would be rejected in that state, so a check that only pings the server would pass right before the real work fails',
+            default: false
+        },
+        requireUpscalerLoaded: {
+            type: 'boolean',
+            title: 'Require A Loaded Upscaler',
+            description: 'Treat a reachable server with no upscaler loaded as unhealthy. Only upscale jobs need one',
+            default: false
+        },
+        timeout: {
+            type: 'number',
+            title: 'Timeout (ms)',
+            description: 'A health check should give up quickly - it exists to tell you the server is not answering',
+            default: 5000
+        }
+    },
+    required: []
+};
+
+const inputSchema = {
+    type: 'object',
+    properties: {
+        data: { type: 'any' }
+    }
+};
+
+const outputSchema = {
+    type: 'object',
+    properties: {
+        healthy: { type: 'boolean', description: 'Reachable and meeting whichever requirements were asked for' },
+        reachable: { type: 'boolean', description: 'The server answered at all' },
+        status: { type: 'string', description: 'The status the server reports about itself' },
+        error: { type: 'string', description: 'Why the check failed, empty when healthy' },
+        responseMs: { type: 'number', description: 'How long the server took to answer' },
+        modelLoaded: { type: 'boolean' },
+        modelName: { type: 'string' },
+        modelType: { type: 'string' },
+        architecture: { type: 'string', description: 'Architecture the server detected, which decides generation defaults' },
+        modelLoading: { type: 'boolean', description: 'True while a model is still being read from disk' },
+        loadingModelName: { type: 'string' },
+        loadingProgress: { type: 'number', description: 'Load progress from 0 to 100, or -1 when nothing is loading' },
+        upscalerLoaded: { type: 'boolean' },
+        upscalerName: { type: 'string' },
+        loadedComponents: { type: 'object' },
+        memory: { type: 'object', description: 'Memory snapshot the server reports' },
+        features: { type: 'object', description: 'Feature flags, including whether authentication is required' },
+        version: { type: 'string' },
+        gitCommit: { type: 'string' },
+        lastError: { type: 'string', description: 'The last error the server itself recorded, if any' }
+    }
+};
+
+function normalizeServer(url) {
+    const value = String(url || '').trim();
+    if (!value) {
+        throw new Error('SD.cpp: a server URL is required, such as http://localhost:8077');
+    }
+    return value.replace(/\/+$/, '');
+}
+
+function unhealthy(server, reason, responseMs) {
+    return {
+        healthy: false,
+        reachable: false,
+        status: '',
+        error: reason,
+        responseMs: responseMs,
+        modelLoaded: false,
+        modelName: '',
+        modelType: '',
+        architecture: '',
+        modelLoading: false,
+        loadingModelName: '',
+        loadingProgress: -1,
+        upscalerLoaded: false,
+        upscalerName: '',
+        loadedComponents: {},
+        memory: {},
+        features: {},
+        version: '',
+        gitCommit: '',
+        lastError: ''
+    };
+}
+
+async function execute(config, input, context) {
+    const server = normalizeServer(config.serverUrl);
+    const timeout = config.timeout > 0 ? config.timeout : 5000;
+
+    const startedAt = Date.now();
+
+    // /health is the one endpoint sdcpp-restapi leaves unauthenticated, which
+    // is what makes this usable as a reachability probe with no credential.
+    let response;
+    try {
+        response = smartbotic.http.request({
+            method: 'GET',
+            url: server + '/health',
+            timeout: timeout
+        });
+    } catch (e) {
+        // A refused connection or a DNS failure throws rather than returning a
+        // status. That is the single most useful thing a health check can
+        // report, so it must not escape as a node error unless asked for.
+        const result = unhealthy(server, 'Could not reach ' + server + ': ' + (e.message || e),
+                                 Date.now() - startedAt);
+        if (config.failIfUnreachable) {
+            throw new Error('SD.cpp health: ' + result.error);
+        }
+        smartbotic.log.warn('SD.cpp health: ' + result.error);
+        return result;
+    }
+
+    const responseMs = Date.now() - startedAt;
+
+    if (!response || response.status < 200 || response.status >= 300) {
+        const status = response ? response.status : 0;
+        const result = unhealthy(server, 'Server answered with HTTP ' + status, responseMs);
+        // It answered, so it is reachable - just not well. Keeping those apart
+        // matters: a 500 is a broken server, a refused connection is a missing
+        // one, and they call for different responses.
+        result.reachable = status > 0;
+        if (config.failIfUnreachable) {
+            throw new Error('SD.cpp health: ' + result.error);
+        }
+        smartbotic.log.warn('SD.cpp health: ' + result.error);
+        return result;
+    }
+
+    let health = response.data;
+    if (typeof health === 'string') {
+        try {
+            health = JSON.parse(health);
+        } catch (e) {
+            const result = unhealthy(server, 'The server answered with something that is not JSON',
+                                     responseMs);
+            result.reachable = true;
+            if (config.failIfUnreachable) {
+                throw new Error('SD.cpp health: ' + result.error);
+            }
+            return result;
+        }
+    }
+    health = health || {};
+
+    const loadingStep = health.loading_step;
+    const loadingTotal = health.loading_total_steps;
+
+    const result = {
+        healthy: true,
+        reachable: true,
+        status: health.status || '',
+        error: '',
+        responseMs: responseMs,
+        modelLoaded: health.model_loaded === true,
+        modelName: health.model_name || '',
+        modelType: health.model_type || '',
+        architecture: health.model_architecture || '',
+        modelLoading: health.model_loading === true,
+        loadingModelName: health.loading_model_name || '',
+        loadingProgress: (loadingTotal > 0 && loadingStep !== undefined)
+            ? Math.round((loadingStep / loadingTotal) * 100)
+            : -1,
+        upscalerLoaded: health.upscaler_loaded === true,
+        upscalerName: health.upscaler_name || '',
+        loadedComponents: health.loaded_components || {},
+        memory: health.memory || {},
+        features: health.features || {},
+        version: health.version || '',
+        gitCommit: health.git_commit || '',
+        lastError: health.last_error ? String(health.last_error) : ''
+    };
+
+    // A server that answers but has nothing loaded will reject a generation
+    // job. Without these checks a health node would go green immediately before
+    // the work it was guarding fails.
+    const problems = [];
+    if (config.requireModelLoaded && !result.modelLoaded) {
+        problems.push(result.modelLoading
+            ? 'a model is still loading (' + result.loadingModelName + ')'
+            : 'no model is loaded');
+    }
+    if (config.requireUpscalerLoaded && !result.upscalerLoaded) {
+        problems.push('no upscaler is loaded');
+    }
+
+    if (problems.length > 0) {
+        result.healthy = false;
+        result.error = 'Server is reachable but ' + problems.join(' and ');
+        if (config.failIfUnreachable) {
+            throw new Error('SD.cpp health: ' + result.error);
+        }
+        smartbotic.log.warn('SD.cpp health: ' + result.error);
+        return result;
+    }
+
+    smartbotic.log.info('SD.cpp health: ' + server + ' healthy in ' + responseMs + 'ms' +
+        (result.modelLoaded ? ', model ' + result.modelName + ' (' + result.architecture + ')'
+                            : ', no model loaded'));
+
+    return result;
+}
+
+module.exports = { configSchema, inputSchema, outputSchema, execute };

+ 53 - 0
src/runner/workflow_engine.cpp

@@ -1887,11 +1887,44 @@ bool WorkflowEngine::executeLoopBody(
     // loop and the per-item loop can unwind.
     // loop and the per-item loop can unwind.
     bool stopped_in_body = false;
     bool stopped_in_body = false;
 
 
+    // Which body nodes are wired back into the loop node. Drawing that edge is
+    // how a workflow says "carry on with the next item", and leaving a branch
+    // without one is how it says "stop here" - the loop still finishes and
+    // everything after it still runs.
+    //
+    // The rule only applies when at least one such edge exists. A loop whose
+    // body was never wired back keeps iterating everything, because that is
+    // what every workflow written before this behaved like, and silently
+    // turning those into one-item loops would break them all at once.
+    // The source has to be a node INSIDE the body. Every loop also has an
+    // incoming edge carrying its data, and that edge targets the loop node too -
+    // counting it would make every loop in existence look explicitly wired back
+    // and stop them all after one item.
+    std::unordered_set<std::string> back_edge_sources;
+    for (const auto& conn : workflow.connections) {
+        if (conn.target_node_id == loop_node_id &&
+            conn.source_node_id != loop_node_id &&
+            body_nodes_set.contains(conn.source_node_id)) {
+            back_edge_sources.insert(conn.source_node_id);
+        }
+    }
+    const bool explicit_loop_back = !back_edge_sources.empty();
+    if (explicit_loop_back) {
+        LOG_DEBUG("Loop {} is wired back explicitly by {} node(s); an iteration continues "
+                  "only when one of them runs", loop_node_id, back_edge_sources.size());
+    }
+
+    // Set when an iteration ended without reaching a node wired back to the
+    // loop, so the remaining items are left alone.
+    bool left_loop_early = false;
+
 
 
     // Execute body for each item
     // Execute body for each item
     for (size_t i = 0; i < ctx.items.size(); ++i) {
     for (size_t i = 0; i < ctx.items.size(); ++i) {
         ctx.current_index = i;
         ctx.current_index = i;
 
 
+        bool reached_back_edge = false;
+
         // The main walk checks for cancellation between nodes; without the same
         // The main walk checks for cancellation between nodes; without the same
         // check here a loop was uncancellable, and a loop is exactly where a run
         // check here a loop was uncancellable, and a loop is exactly where a run
         // spends its time - ten thousand items, or a body that waits on a slow
         // spends its time - ten thousand items, or a body that waits on a slow
@@ -2160,6 +2193,14 @@ bool WorkflowEngine::executeLoopBody(
 
 
             rejectPauseInLoopBody(body_result);
             rejectPauseInLoopBody(body_result);
 
 
+            // Only a node that actually ran counts. A node on a branch that was
+            // not taken is skipped, and a skipped loop-back edge is exactly the
+            // case that should end the iteration.
+            if (body_result.status == NodeStatus::Completed &&
+                back_edge_sources.contains(body_node_id)) {
+                reached_back_edge = true;
+            }
+
             iteration_results[body_node_id] = body_result;
             iteration_results[body_node_id] = body_result;
 
 
             // Store in main results with iteration suffix
             // Store in main results with iteration suffix
@@ -2247,11 +2288,23 @@ bool WorkflowEngine::executeLoopBody(
             break;
             break;
         }
         }
 
 
+        if (explicit_loop_back && !reached_back_edge) {
+            LOG_INFO("Loop {} stopped after item {} - the path taken did not lead back to "
+                     "the loop", loop_node_id, i);
+            left_loop_early = true;
+        }
+
         // A stop ends the remaining items too. Results collected so far are
         // A stop ends the remaining items too. Results collected so far are
         // kept - they were produced before anyone asked to stop.
         // kept - they were produced before anyone asked to stop.
         if (stopped_in_body) {
         if (stopped_in_body) {
             break;
             break;
         }
         }
+
+        // A break ends the remaining items and lets the loop complete, so the
+        // done output carries the results gathered up to this point.
+        if (left_loop_early) {
+            break;
+        }
     }
     }
 
 
     // Copy results back (ctx was passed by const ref, but we used a mutable copy)
     // Copy results back (ctx was passed by const ref, but we used a mutable copy)

+ 34 - 0
tests/nodes/loop-explicit-loopback.json

@@ -0,0 +1,34 @@
+{
+  "name": "verify-loop-explicit-loopback",
+  "nodes": [
+    {"id": "n1", "name": "Trigger", "type": "click-trigger", "position": {"x": 0, "y": 0}, "config": {}},
+    {"id": "src", "name": "Src", "type": "code", "position": {"x": 0, "y": 100},
+     "config": {"code": "return { items: ['a', 'b', 'c', 'd', 'e'] };"}},
+    {"id": "lp", "name": "Loop", "type": "loop", "position": {"x": 0, "y": 200},
+     "config": {"inputField": "data.result.items", "itemVariableName": "it", "outputField": "res", "continueOnError": true}},
+    {"id": "mark", "name": "Mark", "type": "code", "position": {"x": -100, "y": 300},
+     "config": {"code": "return { seen: input.it };"}},
+    {"id": "gate", "name": "Gate", "type": "if-condition", "position": {"x": -100, "y": 400},
+     "config": {"combineWith": "and", "conditions": [{"field": "data.result.seen", "operator": "equals", "value": "b"}]}},
+    {"id": "stopHere", "name": "StopHere", "type": "code", "position": {"x": -250, "y": 500},
+     "config": {"code": "return { hit: true };"}},
+    {"id": "carryOn", "name": "CarryOn", "type": "code", "position": {"x": 50, "y": 500},
+     "config": {"code": "return { next: true };"}},
+    {"id": "after", "name": "After", "type": "code", "position": {"x": 200, "y": 300},
+     "config": {"code": "const d=(input&&input.data)||{}; return { ran: true, processed: d.totalProcessed };"}}
+  ],
+  "connections": [
+    {"sourceNodeId": "n1", "sourceOutput": "main", "targetNodeId": "src", "targetInput": "data"},
+    {"sourceNodeId": "src", "sourceOutput": "main", "targetNodeId": "lp", "targetInput": "data"},
+    {"sourceNodeId": "lp", "sourceOutput": "loop", "targetNodeId": "mark", "targetInput": "data"},
+    {"sourceNodeId": "mark", "sourceOutput": "main", "targetNodeId": "gate", "targetInput": "data"},
+    {"sourceNodeId": "gate", "sourceOutput": "true", "targetNodeId": "stopHere", "targetInput": "data"},
+    {"sourceNodeId": "gate", "sourceOutput": "false", "targetNodeId": "carryOn", "targetInput": "data"},
+    {"sourceNodeId": "carryOn", "sourceOutput": "main", "targetNodeId": "lp", "targetInput": "data"},
+    {"sourceNodeId": "lp", "sourceOutput": "done", "targetNodeId": "after", "targetInput": "data"}
+  ],
+  "expectStatus": "completed",
+  "expect": {
+    "after": {"status": "completed", "output": {"result": {"ran": true, "processed": 2}}}
+  }
+}

+ 19 - 0
tests/nodes/sdcpp-health.json

@@ -0,0 +1,19 @@
+{
+  "name": "verify-sdcpp-health",
+  "nodes": [
+    {"id": "n1", "name": "Trigger", "type": "click-trigger", "position": {"x": 0, "y": 0}, "config": {}},
+    {"id": "up", "name": "Reachable", "type": "sdcpp-health", "position": {"x": 0, "y": 100},
+     "config": {"serverUrl": "http://mulan:8077"}},
+    {"id": "down", "name": "Unreachable", "type": "sdcpp-health", "position": {"x": 0, "y": 200},
+     "config": {"serverUrl": "http://localhost:9", "timeout": 3000}}
+  ],
+  "connections": [
+    {"sourceNodeId": "n1", "sourceOutput": "main", "targetNodeId": "up", "targetInput": "data"},
+    {"sourceNodeId": "up", "sourceOutput": "main", "targetNodeId": "down", "targetInput": "data"}
+  ],
+  "expectStatus": "completed",
+  "expect": {
+    "up": {"status": "completed", "output": {"healthy": true, "reachable": true}},
+    "down": {"status": "completed", "output": {"healthy": false, "reachable": false}}
+  }
+}