/** * @node sdcpp-model-load * @name SD.cpp Load Model * @category sdcpp * @version 1.0.0 * @description Make sure a model is loaded, without reloading one that already is * @icon box */ // The credential this node wants, named so it can be found. It is stored as a // plain basic credential - that is what decides how it is encrypted - and this // only says which basic credential is the SD.cpp one. Anything that accepts a // basic credential still accepts this, and this node still accepts a plain // basic credential, because the shape is identical. const credentialTypes = [ { id: 'sdcpp', label: 'SD.cpp Server', baseType: 'basic', description: 'The username and password you sign in to sdcpp-restapi with. The node exchanges them for a token before every call', usernameLabel: 'Username', passwordLabel: 'Password' } ]; const configSchema = { type: 'object', uiGroups: [ { title: 'Connection', fields: ['serverUrl', 'credentialId'] }, { title: 'Model', fields: ['modelName', 'modelType', 'whenDifferent', 'force'] }, { title: 'Components', fields: ['vae', 'clipL', 'clipG', 't5xxl', 'llm', 'taesd', 'controlnet', 'ipAdapter'] }, { title: 'Loading', fields: ['flashAttn', 'diffusionFlashAttn', 'enableMmap', 'eagerLoad', 'streamLayers', 'maxVram', 'nThreads', 'weightType'] }, { title: 'Advanced', fields: ['vaeFormat', 'prediction', 'rngType', 'samplerRngType', 'loraApplyMode', 'vaeConvDirect', 'diffusionConvDirect', 'taePreviewOnly', 'forceSdxlVaeConvScale', 'backend', 'paramsBackend', 'rpcServers', 'modelArgs', 'tensorTypeRules', 'options', 'timeout', 'loadWaitMs'] } ], // Pressing this asks the server what it has loaded and writes it into the // settings below - the model and every component it reports - so a workflow // can be built from a server that is already set up the way it should be, // rather than by typing the same names again. prefill: { label: 'Take from the server', description: 'Fill these in from the model the server currently has loaded', node: 'sdcpp-health', needs: ['serverUrl'], map: { modelName: 'modelName', modelType: 'modelType', 'loadedComponents.vae': 'vae', 'loadedComponents.clip_l': 'clipL', 'loadedComponents.clip_g': 'clipG', 'loadedComponents.t5xxl': 't5xxl', 'loadedComponents.llm': 'llm', 'loadedComponents.taesd': 'taesd', 'loadedComponents.controlnet': 'controlnet', 'loadedComponents.ip_adapter': 'ipAdapter', 'loadOptions.flash_attn': 'flashAttn', 'loadOptions.diffusion_flash_attn': 'diffusionFlashAttn', 'loadOptions.enable_mmap': 'enableMmap', 'loadOptions.eager_load': 'eagerLoad', 'loadOptions.stream_layers': 'streamLayers', 'loadOptions.max_vram': 'maxVram', 'loadOptions.n_threads': 'nThreads', 'loadOptions.vae_format': 'vaeFormat', 'loadOptions.rng_type': 'rngType', 'loadOptions.lora_apply_mode': 'loraApplyMode', 'loadOptions.vae_conv_direct': 'vaeConvDirect', 'loadOptions.diffusion_conv_direct': 'diffusionConvDirect', 'loadOptions.tae_preview_only': 'taePreviewOnly', 'loadOptions.force_sdxl_vae_conv_scale': 'forceSdxlVaeConvScale', 'loadOptions.rng_type': 'rngType', 'loadOptions.lora_apply_mode': 'loraApplyMode', 'loadOptions.backend': 'backend', 'loadOptions.rpc_servers': 'rpcServers', 'loadOptions.model_args': 'modelArgs', 'loadOptions.backend': 'backend', 'loadOptions.params_backend': 'paramsBackend', 'loadOptions.rpc_servers': 'rpcServers', 'loadOptions.model_args': 'modelArgs' } }, properties: { serverUrl: { type: 'string', title: 'Server URL', description: 'Base address of the sdcpp-restapi server', default: 'http://localhost:8077' }, credentialId: { type: 'string', title: 'Credential', description: 'A basic credential holding the sdcpp-restapi username and password', dynamicOptions: { source: 'credentials', filter: { type: ['sdcpp', 'basic'] } } }, modelName: { type: 'string', title: 'Model', description: 'File name of the model, relative to its type directory. Browse lists what the server has of the Model Type chosen below', dynamicOptions: { source: 'node', node: 'sdcpp-model', config: { listOnly: true }, itemsPath: 'models', valueKey: 'name', labelKey: 'name', needs: ['serverUrl', 'credentialId'] } }, modelType: { type: 'string', title: 'Model Type', enum: ['', 'checkpoint', 'diffusion'], default: '', description: 'checkpoint bundles U-Net, CLIP and VAE and suits SD1, SD2 and SDXL. diffusion holds only the U-Net or DiT and needs its components named separately, which is how Flux, SD3, Qwen, Wan and Z-Image load' }, vae: { type: 'string', title: 'VAE', description: 'Component file name', dynamicOptions: { source: 'node', node: 'sdcpp-model', config: { listOnly: true, modelType: 'vae' }, itemsPath: 'models', valueKey: 'name', labelKey: 'name', needs: ['serverUrl', 'credentialId'] } }, clipL: { type: 'string', title: 'CLIP-L', description: 'Component file name', dynamicOptions: { source: 'node', node: 'sdcpp-model', config: { listOnly: true, modelType: 'clip' }, itemsPath: 'models', valueKey: 'name', labelKey: 'name', needs: ['serverUrl', 'credentialId'] } }, clipG: { type: 'string', title: 'CLIP-G', description: 'Component file name', dynamicOptions: { source: 'node', node: 'sdcpp-model', config: { listOnly: true, modelType: 'clip' }, itemsPath: 'models', valueKey: 'name', labelKey: 'name', needs: ['serverUrl', 'credentialId'] } }, t5xxl: { type: 'string', title: 'T5-XXL', description: 'Component file name', dynamicOptions: { source: 'node', node: 'sdcpp-model', config: { listOnly: true, modelType: 't5' }, itemsPath: 'models', valueKey: 'name', labelKey: 'name', needs: ['serverUrl', 'credentialId'] } }, llm: { type: 'string', title: 'LLM', description: 'Component file name, used by Z-Image, Qwen, Anima and Flux2', dynamicOptions: { source: 'node', node: 'sdcpp-model', config: { listOnly: true, modelType: 'llm' }, itemsPath: 'models', valueKey: 'name', labelKey: 'name', needs: ['serverUrl', 'credentialId'] } }, taesd: { type: 'string', title: 'TAESD', description: 'Tiny autoencoder for progress previews', dynamicOptions: { source: 'node', node: 'sdcpp-model', config: { listOnly: true, modelType: 'taesd' }, itemsPath: 'models', valueKey: 'name', labelKey: 'name', needs: ['serverUrl', 'credentialId'] } }, controlnet: { type: 'string', title: 'ControlNet', description: 'Component file name', dynamicOptions: { source: 'node', node: 'sdcpp-model', config: { listOnly: true, modelType: 'controlnet' }, itemsPath: 'models', valueKey: 'name', labelKey: 'name', needs: ['serverUrl', 'credentialId'] } }, ipAdapter: { type: 'string', title: 'IP-Adapter', description: 'Component file name. An IP-Adapter lets a generation take its style and subject from a reference image - load it here, then set the reference on the generation node. Classic and Plus/Resampler variants both work', dynamicOptions: { source: 'node', node: 'sdcpp-model', config: { listOnly: true, modelType: 'ip_adapter' }, itemsPath: 'models', valueKey: 'name', labelKey: 'name', needs: ['serverUrl', 'credentialId'] } }, flashAttn: { type: 'boolean', title: 'Flash Attention', description: 'For CLIP and T5. A large speed and memory win on modern GPUs' }, diffusionFlashAttn: { type: 'boolean', title: 'Flash Attention (diffusion)', description: 'Flash attention for the diffusion model specifically' }, enableMmap: { type: 'boolean', title: 'Memory-map Weights', description: 'Recommended for large files' }, eagerLoad: { type: 'boolean', title: 'Eager Load', description: 'Move every parameter to the compute backend at load time instead of on demand' }, streamLayers: { type: 'boolean', title: 'Stream Layers', description: 'Stream diffusion layers when the model does not fit in VRAM. Pair with a VRAM budget' }, maxVram: { type: 'number', title: 'VRAM Budget (GiB)', description: 'Budget for segmented parameter offload. 0 leaves it to the server' }, nThreads: { type: 'number', title: 'CPU Threads', description: '-1 lets the server decide' }, weightType: { type: 'string', title: 'Weight Type', enum: ['', 'f32', 'f16', 'bf16', 'q8_0', 'q5_0', 'q5_1', 'q4_0', 'q4_1', 'q4_k', 'q5_k', 'q6_k', 'q8_k', 'q3_k', 'q2_k', 'mxfp4', 'nvfp4', 'q1_0'], enumLabels: ['auto - use the weights in the file', 'f32', 'f16', 'bf16', 'q8_0', 'q5_0', 'q5_1', 'q4_0', 'q4_1', 'q4_k', 'q5_k', 'q6_k', 'q8_k', 'q3_k', 'q2_k', 'mxfp4', 'nvfp4', 'q1_0'], default: '', description: 'Force a quantisation. Left on auto, sd.cpp uses whatever the model file holds' }, vaeFormat: { type: 'string', title: 'VAE Format', enum: ['', 'auto', 'flux', 'sd3', 'flux2', 'wan'], enumLabels: ['not set - leave it to the server', 'auto - detect from the file', 'flux', 'sd3', 'flux2', 'wan'], default: '', description: 'Override VAE format detection' }, prediction: { type: 'string', title: 'Prediction Type', enum: ['', 'eps', 'v', 'edm_v', 'sd3_flow', 'flux_flow', 'flux2_flow', 'sefi_flow', 'minit2i_flow'], enumLabels: ['auto - detect from the model', 'eps', 'v', 'edm_v', 'sd3_flow', 'flux_flow', 'flux2_flow', 'sefi_flow', 'minit2i_flow'], default: '', description: 'Override the prediction type. Left on auto it is detected from the model' }, rngType: { type: 'string', title: 'RNG', enum: ['', 'cuda', 'std_default', 'cpu'], enumLabels: ['server default', 'cuda', 'std_default', 'cpu'], default: '', description: 'Affects whether a seed reproduces across backends' }, samplerRngType: { type: 'string', title: 'Sampler RNG', enum: ['', 'cuda', 'std_default', 'cpu'], enumLabels: ['same as RNG above', 'cuda', 'std_default', 'cpu'], default: '', description: 'Override the RNG for sampling only. The server does not report this one, so Fill in leaves it alone' }, loraApplyMode: { type: 'string', title: 'LoRA Apply Mode', enum: ['', 'auto', 'immediately', 'at_runtime'], enumLabels: ['server default', 'auto', 'immediately', 'at_runtime'], default: '', description: 'When a LoRA named in a prompt is applied' }, vaeConvDirect: { type: 'boolean', title: 'Direct VAE Convolution', description: 'Use the ggml_conv2d_direct path for the VAE' }, diffusionConvDirect: { type: 'boolean', title: 'Direct Diffusion Convolution', description: 'Use the ggml_conv2d_direct path for the diffusion model' }, forceSdxlVaeConvScale: { type: 'boolean', title: 'Force SDXL VAE Conv Scale', description: 'SDXL-specific VAE convolution scaling' }, taePreviewOnly: { type: 'boolean', title: 'TAESD For Preview Only', description: 'Load TAESD purely to render progress previews, skipping the full VAE' }, backend: { type: 'string', title: 'Backend', description: 'Per-component placement, such as te=cpu,vae=cpu,controlnet=cpu' }, paramsBackend: { type: 'string', title: 'Parameter Backend', description: 'Global parameter placement, such as *=cpu to hold weights in system RAM' }, rpcServers: { type: 'string', title: 'RPC Servers', description: 'Comma separated RPC backend endpoints' }, modelArgs: { type: 'string', title: 'Model Args', description: 'Architecture-specific key=value knobs, comma separated' }, tensorTypeRules: { type: 'string', title: 'Tensor Type Rules', description: 'Per-tensor weight overrides using regex, such as ^vae\\.=f16' }, options: { type: 'object', title: 'Other Load Options', description: 'Extra load options passed through, such as flash_attn, enable_mmap, weight_type, stream_layers or max_vram' }, whenDifferent: { type: 'string', title: 'When A Different Model Is Loaded', enum: ['load', 'fail'], default: 'load', description: 'load swaps it. fail stops the run instead - for a workflow that depends on a particular model already being in place and should not quietly spend minutes swapping it' }, force: { type: 'boolean', title: 'Force Reload', default: false, description: 'Load again even when the right model is already loaded. Costs the full load time; useful after changing components or options, which this node cannot see from outside', showWhen: { field: 'whenDifferent', value: 'load' } }, timeout: { type: 'number', title: 'Timeout (ms)', description: 'How long to wait on any one request to the server. Loading reads gigabytes from disk, so the request that starts it is given this long', default: 300000 }, loadWaitMs: { type: 'number', title: 'Wait For Loading (ms)', description: 'How long to keep watching after the load has started. This is separate from the timeout above because a request giving up says nothing about whether the server is still working - it usually is, and the node follows it through /health rather than reporting a failure that is really just impatience', default: 900000 } }, required: [] }; const inputSchema = { type: 'object', properties: { data: { type: 'any' } } }; const outputSchema = { type: 'object', properties: { modelName: { type: 'string', description: 'The model that is loaded now' }, modelType: { type: 'string' }, architecture: { type: 'string', description: 'Architecture the server detected, which decides generation defaults' }, loaded: { type: 'boolean', description: 'True when this node performed a load' }, alreadyLoaded: { type: 'boolean', description: 'True when the right model, with the right settings, was already in place' }, reloadedFor: { type: 'array', description: 'Settings that differed on an otherwise-correct model, when that is why it was reloaded' }, previousModel: { type: 'string', description: 'What was loaded before, when this node swapped it' }, loadedComponents: { type: 'object' }, elapsedMs: { type: 'number' } } }; function normalizeServer(url) { const value = String(url || '').trim(); if (!value) { throw new Error('SD.cpp: a server URL is required, such as http://localhost:8077'); } return value.replace(/\/+$/, ''); } function readCredential(credentialId) { const auth = smartbotic.credentials.get(credentialId); if (!auth || auth.success !== true) { throw new Error('SD.cpp: could not read the credential: ' + ((auth && auth.error) || 'unknown error')); } const value = auth.headerValue || ''; if (value.indexOf('Basic ') !== 0) { throw new Error('SD.cpp: the credential must be a basic one, holding the sdcpp-restapi ' + 'username and password'); } const decoded = smartbotic.utils.base64Decode(value.substring(6)); const separator = decoded.indexOf(':'); if (separator < 1) { throw new Error('SD.cpp: the credential is malformed, expected a username and a password'); } return { username: decoded.substring(0, separator), password: decoded.substring(separator + 1) }; } function call(options) { const response = smartbotic.http.request(options); let body = response.data; if (typeof body === 'string' && body.length > 0) { try { body = JSON.parse(body); } catch (e) { const snippet = body.substring(0, 200).replace(/\s+/g, ' '); throw new Error('SD.cpp: ' + options.what + ' returned HTTP ' + response.status + ' with a body that is not JSON: ' + snippet); } } if (response.status < 200 || response.status >= 300) { const detail = (body && (body.message || body.error)) || ('HTTP ' + response.status); throw new Error('SD.cpp: ' + options.what + ' failed: ' + detail); } return body || {}; } function login(server, credential, timeout) { const session = call({ method: 'POST', url: server + '/auth/login', headers: { 'Content-Type': 'application/json' }, body: JSON.stringify({ username: credential.username, password: credential.password }), timeout: timeout, what: 'signing in' }); if (!session.token) { throw new Error('SD.cpp: the server accepted the login but returned no token'); } return session.token; } // /health is unauthenticated, and it is the only way to find out what is // already loaded without asking for a token first. function readHealth(server, timeout) { return call({ method: 'GET', url: server + '/health', timeout: timeout, // Reading health is safe to repeat, and the moment it matters most is // the moment the server is busiest. retries: 2, retryDelayMs: 1000, what: 'reading server health' }); } // The same, but a server that does not answer is treated as one that is busy // rather than one that has failed. A machine part-way through loading eleven // gigabytes answers /health slowly or not at all - that is what loading looks // like from outside, and throwing there ended the whole run for the one thing // the node was waiting for. function pollHealth(server, timeout) { try { return readHealth(server, timeout); } catch (e) { smartbotic.log.info('SD.cpp: no answer from /health while loading (' + ((e && e.message) || e) + '), still waiting'); return null; } } function putIfSet(target, key, value) { if (value === undefined || value === null || value === '') { return; } target[key] = value; } const LOAD_OPTIONS = [ { setting: 'flashAttn', server: 'flash_attn' }, { setting: 'diffusionFlashAttn', server: 'diffusion_flash_attn' }, { setting: 'enableMmap', server: 'enable_mmap' }, { setting: 'eagerLoad', server: 'eager_load' }, { setting: 'streamLayers', server: 'stream_layers' }, { setting: 'maxVram', server: 'max_vram' }, { setting: 'nThreads', server: 'n_threads' }, { setting: 'weightType', server: 'weight_type' }, { setting: 'vaeFormat', server: 'vae_format' }, { setting: 'prediction', server: 'prediction' }, { setting: 'rngType', server: 'rng_type' }, { setting: 'samplerRngType', server: 'sampler_rng_type' }, { setting: 'loraApplyMode', server: 'lora_apply_mode' }, { setting: 'vaeConvDirect', server: 'vae_conv_direct' }, { setting: 'diffusionConvDirect', server: 'diffusion_conv_direct' }, { setting: 'taePreviewOnly', server: 'tae_preview_only' }, { setting: 'forceSdxlVaeConvScale', server: 'force_sdxl_vae_conv_scale' }, { setting: 'backend', server: 'backend' }, { setting: 'paramsBackend', server: 'params_backend' }, { setting: 'rpcServers', server: 'rpc_servers' }, { setting: 'modelArgs', server: 'model_args' }, { setting: 'tensorTypeRules', server: 'tensor_type_rules' }, ]; // The options this node asks for, as the server names them. Only settings that // were actually filled in are included: an untouched setting means "whatever // the server does", not "the default", so it is neither sent nor compared. function wantedOptions(config) { var wanted = {}; for (var i = 0; i < LOAD_OPTIONS.length; i++) { var entry = LOAD_OPTIONS[i]; var value = config[entry.setting]; if (value === undefined || value === null || value === '') continue; if (typeof value === 'number' && !isFinite(value)) continue; wanted[entry.server] = value; } if (config.options && typeof config.options === 'object') { var keys = Object.keys(config.options); for (var k = 0; k < keys.length; k++) { var extra = config.options[keys[k]]; if (extra !== undefined && extra !== null && extra !== '') { wanted[keys[k]] = extra; } } } return wanted; } // Which of them the server is not currently loaded with. // // This is what makes the node able to correct a server someone else changed: // the same model loaded with streaming off is not the same thing as the model // this workflow needs, and reloading it is the whole point of saying so here. function optionsThatDiffer(wanted, current) { var differing = []; var keys = Object.keys(wanted); for (var i = 0; i < keys.length; i++) { var key = keys[i]; var have = current ? current[key] : undefined; var want = wanted[key]; // Numbers arrive as 0 or 0.0 depending on the field, and a boolean may // come back as a string from a form, so compare on value rather than // on type. var same = (typeof want === 'number' || typeof have === 'number') ? Number(have) === Number(want) : String(have) === String(want); if (!same) { differing.push(key + ': server has ' + JSON.stringify(have) + ', this node wants ' + JSON.stringify(want)); } } return differing; } async function execute(config, input, context) { const server = normalizeServer(config.serverUrl); const timeout = config.timeout > 0 ? config.timeout : 300000; // Watching costs nothing, so it is allowed to outlast any single request. const loadWait = config.loadWaitMs > 0 ? config.loadWaitMs : 900000; const modelName = String(config.modelName || '').trim(); if (!modelName) { throw new Error('SD.cpp: a model name is required. Connect an SD.cpp Model node, ' + 'or type the file name'); } // Ask what is loaded before loading anything. A load takes minutes and // unloads whatever was there, so doing it when the right model is already // resident is pure cost - and on a shared server it disrupts other work. const startedAt = Date.now(); const health = readHealth(server, Math.min(timeout, 15000)); const current = health.model_name || ''; const sameModel = current === modelName; // The same model loaded with different settings is not the model this // workflow asked for. The server swaps models between queue items, so what // is loaded now may have been put there by something else entirely. const wanted = wantedOptions(config); const differing = optionsThatDiffer(wanted, health.load_options || {}); if (sameModel && differing.length === 0 && config.force !== true) { smartbotic.log.info('SD.cpp: ' + modelName + ' is already loaded with the wanted settings'); return { modelName: current, modelType: health.model_type || '', architecture: health.model_architecture || '', loaded: false, alreadyLoaded: true, reloadedFor: [], previousModel: '', loadedComponents: health.loaded_components || {}, elapsedMs: Date.now() - startedAt }; } if ((config.whenDifferent || 'load') === 'fail' && (!sameModel || differing.length > 0)) { if (!sameModel) { throw new Error('SD.cpp: this workflow expects "' + modelName + '" to be loaded, but ' + (current ? 'the server has "' + current + '"' : 'no model is loaded') + '. Set When A Different Model Is Loaded to "load" to swap it automatically'); } throw new Error('SD.cpp: "' + modelName + '" is loaded, but not with the settings this ' + 'workflow needs - ' + differing.join('; ') + '. Set When A Different Model Is Loaded to "load" to reload it'); } if (sameModel && differing.length > 0) { smartbotic.log.info('SD.cpp: reloading ' + modelName + ' because ' + differing.join('; ')); } const credential = readCredential(config.credentialId); const token = login(server, credential, Math.min(timeout, 30000)); const body = { model_name: modelName }; putIfSet(body, 'model_type', config.modelType); putIfSet(body, 'vae', config.vae); putIfSet(body, 'clip_l', config.clipL); putIfSet(body, 'clip_g', config.clipG); putIfSet(body, 't5xxl', config.t5xxl); putIfSet(body, 'llm', config.llm); putIfSet(body, 'taesd', config.taesd); putIfSet(body, 'controlnet', config.controlnet); putIfSet(body, 'ip_adapter', config.ipAdapter); if (Object.keys(wanted).length > 0) { body.options = wanted; } smartbotic.log.info('SD.cpp: loading ' + modelName + (current ? ' (replacing ' + current + ')' : '')); // The slot has to be emptied first. The API documentation says a load // replaces whatever is there, but the server answers 409 "A model is // already loaded. Call POST /models/unload first" - so it is unloaded here // rather than leaving every reload to fail on a server that already has a // model. (A refused load is harmless: the resident model stays put.) // // This is also the only way to change the settings of a model that is // already loaded, which is the case this node exists to handle. // // It does mean everything between here and a finished load runs with the // server holding nothing. The server never unloads on its own, so an empty // slot afterwards is always something that happened in this window - which // is why the failure paths below say so rather than leaving the next run to // discover it. let emptiedTheSlot = false; if (health.model_loaded === true) { call({ method: 'POST', url: server + '/models/unload', headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token }, body: '{}', timeout: Math.min(timeout, 60000), what: 'unloading ' + (current || 'the current model') + ' before loading ' + modelName }); smartbotic.log.info('SD.cpp: unloaded ' + (current || 'the previous model')); emptiedTheSlot = true; } // Loading unloads whatever was in the slot first, and the server holds a // mutex for the duration, so this blocks until the weights are resident. let loaded; try { loaded = call({ method: 'POST', url: server + '/models/load', headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token }, body: JSON.stringify(body), timeout: timeout, // Deliberately not retried: the server holds a mutex for the whole // load, so a second request would queue behind the first and load // the same weights twice. what: 'loading model ' + modelName }); } catch (loadError) { // This call giving up does not mean the server did. If it is still // loading, that is the answer to what happened - so the wait below // finds out how it goes rather than reporting a failure that is really // just impatience. const probe = pollHealth(server, 15000); if (!probe || probe.model_loading !== true) { // The slot was emptied to make room and the load did not take, so // the server now holds nothing. One more attempt is worth it: there // is nothing left to lose, the usual cause is a moment of // slowness, and the alternative is leaving the server worse than it // was found. if (emptiedTheSlot && (!probe || probe.model_loaded !== true)) { smartbotic.log.warn('SD.cpp: the load failed and the server now has no model. ' + 'Trying once more before giving up'); try { loaded = call({ method: 'POST', url: server + '/models/load', headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token }, body: JSON.stringify(body), timeout: timeout, what: 'loading model ' + modelName + ' (second attempt)' }); // Falls through to the wait below, the same as a first // attempt that worked - the model still has to finish // loading either way. } catch (secondError) { throw new Error('SD.cpp: could not load ' + modelName + ', and the server ' + 'is now holding no model at all - it was unloaded to make room. ' + 'Nothing will generate until a load succeeds. First attempt: ' + ((loadError && loadError.message) || loadError) + '. Second: ' + ((secondError && secondError.message) || secondError)); } } throw loadError; } smartbotic.log.info('SD.cpp: the load request stopped waiting, but the server is still ' + 'loading ' + (probe.loading_model_name || modelName) + ' - following it through /health'); loaded = {}; } // The load call comes back before the model is in memory. The API // documentation describes it as blocking, and it is not: /health reports // model_loading with a step count for some time afterwards. Returning here // would tell the workflow the model is ready and let the next node ask it // to generate, which fails with "no model loaded" - a confusing way to // learn that this node lied. const deadline = Date.now() + loadWait; let after = pollHealth(server, 15000); let lastStep = -1; let silentPolls = 0; while ((after === null || after.model_loading === true) && Date.now() < deadline) { if (after === null) { silentPolls++; } else { silentPolls = 0; const step = after.loading_step; const total = after.loading_total_steps; if (typeof step === 'number' && step !== lastStep) { lastStep = step; smartbotic.log.info('SD.cpp: loading ' + (after.loading_model_name || modelName) + ' - ' + step + (total ? '/' + total : '')); } } smartbotic.utils.sleep(2000); after = pollHealth(server, 15000); } const waitedSeconds = Math.round((Date.now() - startedAt) / 1000); if (after === null) { throw new Error('SD.cpp: ' + modelName + ' was asked for ' + waitedSeconds + 's ago and the server has stopped answering /health (' + silentPolls + ' polls in a row went unanswered). It may still be loading - check the ' + 'server, and raise the timeout on this node if this model is simply slow'); } if (after.model_loading === true) { throw new Error('SD.cpp: ' + modelName + ' was still loading after ' + waitedSeconds + 's' + (typeof after.loading_step === 'number' ? ' (at step ' + after.loading_step + (after.loading_total_steps ? ' of ' + after.loading_total_steps : '') + ')' : '') + '. It may still finish on the server; raise the timeout on this node if this ' + 'model is simply slow to load'); } if (after.model_loaded !== true) { throw new Error('SD.cpp: the server accepted the load but has no model loaded afterwards' + (after.last_error ? ': ' + after.last_error : '')); } return { modelName: loaded.model_name || modelName, modelType: loaded.model_type || config.modelType || '', architecture: after.model_architecture || '', loaded: true, alreadyLoaded: false, // Empty when the model itself changed; otherwise the settings that // forced a reload of a model that was already there. reloadedFor: sameModel ? differing : [], previousModel: sameModel ? '' : current, loadedComponents: loaded.loaded_components || after.loaded_components || {}, elapsedMs: Date.now() - startedAt }; } module.exports = { configSchema, inputSchema, outputSchema, execute };