|
@@ -534,6 +534,25 @@ async function execute(config, input, context) {
|
|
|
smartbotic.log.info('SD.cpp: loading ' + modelName +
|
|
smartbotic.log.info('SD.cpp: loading ' + modelName +
|
|
|
(current ? ' (replacing ' + current + ')' : ''));
|
|
(current ? ' (replacing ' + current + ')' : ''));
|
|
|
|
|
|
|
|
|
|
+ // The slot has to be emptied first. The API documentation says a load
|
|
|
|
|
+ // replaces whatever is there, but the server answers "A model is already
|
|
|
|
|
+ // loaded. Call POST /models/unload first" - so it is unloaded here rather
|
|
|
|
|
+ // than leaving every reload to fail on a server that already has a model.
|
|
|
|
|
+ //
|
|
|
|
|
+ // This is also the only way to change the settings of a model that is
|
|
|
|
|
+ // already loaded, which is the case this node exists to handle.
|
|
|
|
|
+ if (health.model_loaded === true) {
|
|
|
|
|
+ call({
|
|
|
|
|
+ method: 'POST',
|
|
|
|
|
+ url: server + '/models/unload',
|
|
|
|
|
+ headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token },
|
|
|
|
|
+ body: '{}',
|
|
|
|
|
+ timeout: Math.min(timeout, 60000),
|
|
|
|
|
+ what: 'unloading ' + (current || 'the current model') + ' before loading ' + modelName
|
|
|
|
|
+ });
|
|
|
|
|
+ smartbotic.log.info('SD.cpp: unloaded ' + (current || 'the previous model'));
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
// Loading unloads whatever was in the slot first, and the server holds a
|
|
// Loading unloads whatever was in the slot first, and the server holds a
|
|
|
// mutex for the duration, so this blocks until the weights are resident.
|
|
// mutex for the duration, so this blocks until the weights are resident.
|
|
|
const loaded = call({
|
|
const loaded = call({
|
|
@@ -545,7 +564,38 @@ async function execute(config, input, context) {
|
|
|
what: 'loading model ' + modelName
|
|
what: 'loading model ' + modelName
|
|
|
});
|
|
});
|
|
|
|
|
|
|
|
- const after = readHealth(server, Math.min(timeout, 15000));
|
|
|
|
|
|
|
+ // The load call comes back before the model is in memory. The API
|
|
|
|
|
+ // documentation describes it as blocking, and it is not: /health reports
|
|
|
|
|
+ // model_loading with a step count for some time afterwards. Returning here
|
|
|
|
|
+ // would tell the workflow the model is ready and let the next node ask it
|
|
|
|
|
+ // to generate, which fails with "no model loaded" - a confusing way to
|
|
|
|
|
+ // learn that this node lied.
|
|
|
|
|
+ const deadline = startedAt + timeout;
|
|
|
|
|
+ let after = readHealth(server, 15000);
|
|
|
|
|
+ let lastStep = -1;
|
|
|
|
|
+
|
|
|
|
|
+ while (after.model_loading === true && Date.now() < deadline) {
|
|
|
|
|
+ const step = after.loading_step;
|
|
|
|
|
+ const total = after.loading_total_steps;
|
|
|
|
|
+ if (typeof step === 'number' && step !== lastStep) {
|
|
|
|
|
+ lastStep = step;
|
|
|
|
|
+ smartbotic.log.info('SD.cpp: loading ' + (after.loading_model_name || modelName) +
|
|
|
|
|
+ ' - ' + step + (total ? '/' + total : ''));
|
|
|
|
|
+ }
|
|
|
|
|
+ smartbotic.utils.sleep(2000);
|
|
|
|
|
+ after = readHealth(server, 15000);
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ if (after.model_loading === true) {
|
|
|
|
|
+ throw new Error('SD.cpp: ' + modelName + ' was still loading after ' +
|
|
|
|
|
+ Math.round((Date.now() - startedAt) / 1000) + 's. It may still finish on the server; ' +
|
|
|
|
|
+ 'raise the timeout on this node if this model is simply slow to load');
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
|
|
+ if (after.model_loaded !== true) {
|
|
|
|
|
+ throw new Error('SD.cpp: the server accepted the load but has no model loaded afterwards' +
|
|
|
|
|
+ (after.last_error ? ': ' + after.last_error : ''));
|
|
|
|
|
+ }
|
|
|
|
|
|
|
|
return {
|
|
return {
|
|
|
modelName: loaded.model_name || modelName,
|
|
modelName: loaded.model_name || modelName,
|