|
|
@@ -34,7 +34,7 @@ const configSchema = {
|
|
|
{ title: 'Advanced', fields: ['vaeFormat', 'prediction', 'rngType', 'samplerRngType',
|
|
|
'loraApplyMode', 'vaeConvDirect', 'diffusionConvDirect',
|
|
|
'taePreviewOnly', 'forceSdxlVaeConvScale', 'backend', 'paramsBackend', 'rpcServers',
|
|
|
- 'modelArgs', 'tensorTypeRules', 'options', 'timeout'] }
|
|
|
+ 'modelArgs', 'tensorTypeRules', 'options', 'timeout', 'loadWaitMs'] }
|
|
|
],
|
|
|
// Pressing this asks the server what it has loaded and writes it into the
|
|
|
// settings below - the model and every component it reports - so a workflow
|
|
|
@@ -269,8 +269,13 @@ const configSchema = {
|
|
|
},
|
|
|
timeout: {
|
|
|
type: 'number', title: 'Timeout (ms)',
|
|
|
- description: 'Loading reads gigabytes from disk and can take minutes',
|
|
|
+ description: 'How long to wait on any one request to the server. Loading reads gigabytes from disk, so the request that starts it is given this long',
|
|
|
default: 300000
|
|
|
+ },
|
|
|
+ loadWaitMs: {
|
|
|
+ type: 'number', title: 'Wait For Loading (ms)',
|
|
|
+ description: 'How long to keep watching after the load has started. This is separate from the timeout above because a request giving up says nothing about whether the server is still working - it usually is, and the node follows it through /health rather than reporting a failure that is really just impatience',
|
|
|
+ default: 900000
|
|
|
}
|
|
|
},
|
|
|
required: []
|
|
|
@@ -374,10 +379,29 @@ function readHealth(server, timeout) {
|
|
|
method: 'GET',
|
|
|
url: server + '/health',
|
|
|
timeout: timeout,
|
|
|
+ // Reading health is safe to repeat, and the moment it matters most is
|
|
|
+ // the moment the server is busiest.
|
|
|
+ retries: 2,
|
|
|
+ retryDelayMs: 1000,
|
|
|
what: 'reading server health'
|
|
|
});
|
|
|
}
|
|
|
|
|
|
+// The same, but a server that does not answer is treated as one that is busy
|
|
|
+// rather than one that has failed. A machine part-way through loading eleven
|
|
|
+// gigabytes answers /health slowly or not at all - that is what loading looks
|
|
|
+// like from outside, and throwing there ended the whole run for the one thing
|
|
|
+// the node was waiting for.
|
|
|
+function pollHealth(server, timeout) {
|
|
|
+ try {
|
|
|
+ return readHealth(server, timeout);
|
|
|
+ } catch (e) {
|
|
|
+ smartbotic.log.info('SD.cpp: no answer from /health while loading (' +
|
|
|
+ ((e && e.message) || e) + '), still waiting');
|
|
|
+ return null;
|
|
|
+ }
|
|
|
+}
|
|
|
+
|
|
|
function putIfSet(target, key, value) {
|
|
|
if (value === undefined || value === null || value === '') {
|
|
|
return;
|
|
|
@@ -464,6 +488,8 @@ function optionsThatDiffer(wanted, current) {
|
|
|
async function execute(config, input, context) {
|
|
|
const server = normalizeServer(config.serverUrl);
|
|
|
const timeout = config.timeout > 0 ? config.timeout : 300000;
|
|
|
+ // Watching costs nothing, so it is allowed to outlast any single request.
|
|
|
+ const loadWait = config.loadWaitMs > 0 ? config.loadWaitMs : 900000;
|
|
|
const modelName = String(config.modelName || '').trim();
|
|
|
|
|
|
if (!modelName) {
|
|
|
@@ -555,14 +581,32 @@ async function execute(config, input, context) {
|
|
|
|
|
|
// Loading unloads whatever was in the slot first, and the server holds a
|
|
|
// mutex for the duration, so this blocks until the weights are resident.
|
|
|
- const loaded = call({
|
|
|
- method: 'POST',
|
|
|
- url: server + '/models/load',
|
|
|
- headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token },
|
|
|
- body: JSON.stringify(body),
|
|
|
- timeout: timeout,
|
|
|
- what: 'loading model ' + modelName
|
|
|
- });
|
|
|
+ let loaded;
|
|
|
+ try {
|
|
|
+ loaded = call({
|
|
|
+ method: 'POST',
|
|
|
+ url: server + '/models/load',
|
|
|
+ headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token },
|
|
|
+ body: JSON.stringify(body),
|
|
|
+ timeout: timeout,
|
|
|
+ // Deliberately not retried: the server holds a mutex for the whole
|
|
|
+ // load, so a second request would queue behind the first and load
|
|
|
+ // the same weights twice.
|
|
|
+ what: 'loading model ' + modelName
|
|
|
+ });
|
|
|
+ } catch (loadError) {
|
|
|
+ // This call giving up does not mean the server did. If it is still
|
|
|
+ // loading, that is the answer to what happened - so the wait below
|
|
|
+ // finds out how it goes rather than reporting a failure that is really
|
|
|
+ // just impatience.
|
|
|
+ const probe = pollHealth(server, 15000);
|
|
|
+ if (!probe || probe.model_loading !== true) {
|
|
|
+ throw loadError;
|
|
|
+ }
|
|
|
+ smartbotic.log.info('SD.cpp: the load request stopped waiting, but the server is still ' +
|
|
|
+ 'loading ' + (probe.loading_model_name || modelName) + ' - following it through /health');
|
|
|
+ loaded = {};
|
|
|
+ }
|
|
|
|
|
|
// The load call comes back before the model is in memory. The API
|
|
|
// documentation describes it as blocking, and it is not: /health reports
|
|
|
@@ -570,26 +614,45 @@ async function execute(config, input, context) {
|
|
|
// would tell the workflow the model is ready and let the next node ask it
|
|
|
// to generate, which fails with "no model loaded" - a confusing way to
|
|
|
// learn that this node lied.
|
|
|
- const deadline = startedAt + timeout;
|
|
|
- let after = readHealth(server, 15000);
|
|
|
+ const deadline = Date.now() + loadWait;
|
|
|
+ let after = pollHealth(server, 15000);
|
|
|
let lastStep = -1;
|
|
|
-
|
|
|
- while (after.model_loading === true && Date.now() < deadline) {
|
|
|
- const step = after.loading_step;
|
|
|
- const total = after.loading_total_steps;
|
|
|
- if (typeof step === 'number' && step !== lastStep) {
|
|
|
- lastStep = step;
|
|
|
- smartbotic.log.info('SD.cpp: loading ' + (after.loading_model_name || modelName) +
|
|
|
- ' - ' + step + (total ? '/' + total : ''));
|
|
|
+ let silentPolls = 0;
|
|
|
+
|
|
|
+ while ((after === null || after.model_loading === true) && Date.now() < deadline) {
|
|
|
+ if (after === null) {
|
|
|
+ silentPolls++;
|
|
|
+ } else {
|
|
|
+ silentPolls = 0;
|
|
|
+ const step = after.loading_step;
|
|
|
+ const total = after.loading_total_steps;
|
|
|
+ if (typeof step === 'number' && step !== lastStep) {
|
|
|
+ lastStep = step;
|
|
|
+ smartbotic.log.info('SD.cpp: loading ' + (after.loading_model_name || modelName) +
|
|
|
+ ' - ' + step + (total ? '/' + total : ''));
|
|
|
+ }
|
|
|
}
|
|
|
smartbotic.utils.sleep(2000);
|
|
|
- after = readHealth(server, 15000);
|
|
|
+ after = pollHealth(server, 15000);
|
|
|
+ }
|
|
|
+
|
|
|
+ const waitedSeconds = Math.round((Date.now() - startedAt) / 1000);
|
|
|
+
|
|
|
+ if (after === null) {
|
|
|
+ throw new Error('SD.cpp: ' + modelName + ' was asked for ' + waitedSeconds +
|
|
|
+ 's ago and the server has stopped answering /health (' + silentPolls +
|
|
|
+ ' polls in a row went unanswered). It may still be loading - check the ' +
|
|
|
+ 'server, and raise the timeout on this node if this model is simply slow');
|
|
|
}
|
|
|
|
|
|
if (after.model_loading === true) {
|
|
|
- throw new Error('SD.cpp: ' + modelName + ' was still loading after ' +
|
|
|
- Math.round((Date.now() - startedAt) / 1000) + 's. It may still finish on the server; ' +
|
|
|
- 'raise the timeout on this node if this model is simply slow to load');
|
|
|
+ throw new Error('SD.cpp: ' + modelName + ' was still loading after ' + waitedSeconds +
|
|
|
+ 's' + (typeof after.loading_step === 'number'
|
|
|
+ ? ' (at step ' + after.loading_step +
|
|
|
+ (after.loading_total_steps ? ' of ' + after.loading_total_steps : '') + ')'
|
|
|
+ : '') +
|
|
|
+ '. It may still finish on the server; raise the timeout on this node if this ' +
|
|
|
+ 'model is simply slow to load');
|
|
|
}
|
|
|
|
|
|
if (after.model_loaded !== true) {
|