| 123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716 |
- /**
- * @node sdcpp-model-load
- * @name SD.cpp Load Model
- * @category sdcpp
- * @version 1.0.0
- * @description Make sure a model is loaded, without reloading one that already is
- * @icon box
- */
- // The credential this node wants, named so it can be found. It is stored as a
- // plain basic credential - that is what decides how it is encrypted - and this
- // only says which basic credential is the SD.cpp one. Anything that accepts a
- // basic credential still accepts this, and this node still accepts a plain
- // basic credential, because the shape is identical.
- const credentialTypes = [
- {
- id: 'sdcpp',
- label: 'SD.cpp Server',
- baseType: 'basic',
- description: 'The username and password you sign in to sdcpp-restapi with. The node exchanges them for a token before every call',
- usernameLabel: 'Username',
- passwordLabel: 'Password'
- }
- ];
- const configSchema = {
- type: 'object',
- uiGroups: [
- { title: 'Connection', fields: ['serverUrl', 'credentialId'] },
- { title: 'Model', fields: ['modelName', 'modelType', 'whenDifferent', 'force'] },
- { title: 'Components', fields: ['vae', 'clipL', 'clipG', 't5xxl', 'llm', 'taesd', 'controlnet'] },
- { title: 'Loading', fields: ['flashAttn', 'diffusionFlashAttn', 'enableMmap', 'eagerLoad',
- 'streamLayers', 'maxVram', 'nThreads', 'weightType'] },
- { title: 'Advanced', fields: ['vaeFormat', 'prediction', 'rngType', 'samplerRngType',
- 'loraApplyMode', 'vaeConvDirect', 'diffusionConvDirect',
- 'taePreviewOnly', 'forceSdxlVaeConvScale', 'backend', 'paramsBackend', 'rpcServers',
- 'modelArgs', 'tensorTypeRules', 'options', 'timeout', 'loadWaitMs'] }
- ],
- // Pressing this asks the server what it has loaded and writes it into the
- // settings below - the model and every component it reports - so a workflow
- // can be built from a server that is already set up the way it should be,
- // rather than by typing the same names again.
- prefill: {
- label: 'Take from the server',
- description: 'Fill these in from the model the server currently has loaded',
- node: 'sdcpp-health',
- needs: ['serverUrl'],
- map: {
- modelName: 'modelName',
- modelType: 'modelType',
- 'loadedComponents.vae': 'vae',
- 'loadedComponents.clip_l': 'clipL',
- 'loadedComponents.clip_g': 'clipG',
- 'loadedComponents.t5xxl': 't5xxl',
- 'loadedComponents.llm': 'llm',
- 'loadedComponents.taesd': 'taesd',
- 'loadedComponents.controlnet': 'controlnet',
- 'loadOptions.flash_attn': 'flashAttn',
- 'loadOptions.diffusion_flash_attn': 'diffusionFlashAttn',
- 'loadOptions.enable_mmap': 'enableMmap',
- 'loadOptions.eager_load': 'eagerLoad',
- 'loadOptions.stream_layers': 'streamLayers',
- 'loadOptions.max_vram': 'maxVram',
- 'loadOptions.n_threads': 'nThreads',
- 'loadOptions.vae_format': 'vaeFormat',
- 'loadOptions.rng_type': 'rngType',
- 'loadOptions.lora_apply_mode': 'loraApplyMode',
- 'loadOptions.vae_conv_direct': 'vaeConvDirect',
- 'loadOptions.diffusion_conv_direct': 'diffusionConvDirect',
- 'loadOptions.tae_preview_only': 'taePreviewOnly',
- 'loadOptions.force_sdxl_vae_conv_scale': 'forceSdxlVaeConvScale',
- 'loadOptions.rng_type': 'rngType',
- 'loadOptions.lora_apply_mode': 'loraApplyMode',
- 'loadOptions.backend': 'backend',
- 'loadOptions.rpc_servers': 'rpcServers',
- 'loadOptions.model_args': 'modelArgs',
- 'loadOptions.backend': 'backend',
- 'loadOptions.params_backend': 'paramsBackend',
- 'loadOptions.rpc_servers': 'rpcServers',
- 'loadOptions.model_args': 'modelArgs'
- }
- },
- properties: {
- serverUrl: {
- type: 'string', title: 'Server URL',
- description: 'Base address of the sdcpp-restapi server',
- default: 'http://localhost:8077'
- },
- credentialId: {
- type: 'string', title: 'Credential',
- description: 'A basic credential holding the sdcpp-restapi username and password',
- dynamicOptions: { source: 'credentials', filter: { type: ['sdcpp', 'basic'] } }
- },
- modelName: {
- type: 'string', title: 'Model',
- description: 'File name of the model, relative to its type directory. Browse lists what the server has of the Model Type chosen below',
- dynamicOptions: {
- source: 'node',
- node: 'sdcpp-model',
- config: { listOnly: true },
- itemsPath: 'models',
- valueKey: 'name',
- labelKey: 'name',
- needs: ['serverUrl', 'credentialId']
- }
- },
- modelType: {
- type: 'string', title: 'Model Type',
- enum: ['', 'checkpoint', 'diffusion'],
- default: '',
- description: 'checkpoint bundles U-Net, CLIP and VAE and suits SD1, SD2 and SDXL. diffusion holds only the U-Net or DiT and needs its components named separately, which is how Flux, SD3, Qwen, Wan and Z-Image load'
- },
- vae: {
- type: 'string', title: 'VAE',
- description: 'Component file name',
- dynamicOptions: {
- source: 'node',
- node: 'sdcpp-model',
- config: { listOnly: true, modelType: 'vae' },
- itemsPath: 'models',
- valueKey: 'name',
- labelKey: 'name',
- needs: ['serverUrl', 'credentialId']
- }
- },
- clipL: {
- type: 'string', title: 'CLIP-L',
- description: 'Component file name',
- dynamicOptions: {
- source: 'node',
- node: 'sdcpp-model',
- config: { listOnly: true, modelType: 'clip' },
- itemsPath: 'models',
- valueKey: 'name',
- labelKey: 'name',
- needs: ['serverUrl', 'credentialId']
- }
- },
- clipG: {
- type: 'string', title: 'CLIP-G',
- description: 'Component file name',
- dynamicOptions: {
- source: 'node',
- node: 'sdcpp-model',
- config: { listOnly: true, modelType: 'clip' },
- itemsPath: 'models',
- valueKey: 'name',
- labelKey: 'name',
- needs: ['serverUrl', 'credentialId']
- }
- },
- t5xxl: {
- type: 'string', title: 'T5-XXL',
- description: 'Component file name',
- dynamicOptions: {
- source: 'node',
- node: 'sdcpp-model',
- config: { listOnly: true, modelType: 't5' },
- itemsPath: 'models',
- valueKey: 'name',
- labelKey: 'name',
- needs: ['serverUrl', 'credentialId']
- }
- },
- llm: {
- type: 'string', title: 'LLM',
- description: 'Component file name, used by Z-Image, Qwen, Anima and Flux2',
- dynamicOptions: {
- source: 'node',
- node: 'sdcpp-model',
- config: { listOnly: true, modelType: 'llm' },
- itemsPath: 'models',
- valueKey: 'name',
- labelKey: 'name',
- needs: ['serverUrl', 'credentialId']
- }
- },
- taesd: {
- type: 'string', title: 'TAESD',
- description: 'Tiny autoencoder for progress previews',
- dynamicOptions: {
- source: 'node',
- node: 'sdcpp-model',
- config: { listOnly: true, modelType: 'taesd' },
- itemsPath: 'models',
- valueKey: 'name',
- labelKey: 'name',
- needs: ['serverUrl', 'credentialId']
- }
- },
- controlnet: {
- type: 'string', title: 'ControlNet',
- description: 'Component file name',
- dynamicOptions: {
- source: 'node',
- node: 'sdcpp-model',
- config: { listOnly: true, modelType: 'controlnet' },
- itemsPath: 'models',
- valueKey: 'name',
- labelKey: 'name',
- needs: ['serverUrl', 'credentialId']
- }
- },
- flashAttn: { type: 'boolean', title: 'Flash Attention', description: 'For CLIP and T5. A large speed and memory win on modern GPUs' },
- diffusionFlashAttn: { type: 'boolean', title: 'Flash Attention (diffusion)', description: 'Flash attention for the diffusion model specifically' },
- enableMmap: { type: 'boolean', title: 'Memory-map Weights', description: 'Recommended for large files' },
- eagerLoad: { type: 'boolean', title: 'Eager Load', description: 'Move every parameter to the compute backend at load time instead of on demand' },
- streamLayers: { type: 'boolean', title: 'Stream Layers', description: 'Stream diffusion layers when the model does not fit in VRAM. Pair with a VRAM budget' },
- maxVram: { type: 'number', title: 'VRAM Budget (GiB)', description: 'Budget for segmented parameter offload. 0 leaves it to the server' },
- nThreads: { type: 'number', title: 'CPU Threads', description: '-1 lets the server decide' },
- weightType: {
- type: 'string', title: 'Weight Type',
- enum: ['', 'f32', 'f16', 'bf16', 'q8_0', 'q5_0', 'q5_1', 'q4_0', 'q4_1', 'q4_k', 'q5_k', 'q6_k', 'q8_k', 'q3_k', 'q2_k', 'mxfp4', 'nvfp4', 'q1_0'],
- enumLabels: ['auto - use the weights in the file', 'f32', 'f16', 'bf16', 'q8_0', 'q5_0', 'q5_1', 'q4_0', 'q4_1', 'q4_k', 'q5_k', 'q6_k', 'q8_k', 'q3_k', 'q2_k', 'mxfp4', 'nvfp4', 'q1_0'],
- default: '', description: 'Force a quantisation. Left on auto, sd.cpp uses whatever the model file holds'
- },
- vaeFormat: {
- type: 'string', title: 'VAE Format',
- enum: ['', 'auto', 'flux', 'sd3', 'flux2', 'wan'],
- enumLabels: ['not set - leave it to the server', 'auto - detect from the file', 'flux', 'sd3', 'flux2', 'wan'],
- default: '', description: 'Override VAE format detection'
- },
- prediction: {
- type: 'string', title: 'Prediction Type',
- enum: ['', 'eps', 'v', 'edm_v', 'sd3_flow', 'flux_flow', 'flux2_flow', 'sefi_flow', 'minit2i_flow'],
- enumLabels: ['auto - detect from the model', 'eps', 'v', 'edm_v', 'sd3_flow', 'flux_flow', 'flux2_flow', 'sefi_flow', 'minit2i_flow'],
- default: '', description: 'Override the prediction type. Left on auto it is detected from the model'
- },
- rngType: {
- type: 'string', title: 'RNG', enum: ['', 'cuda', 'std_default', 'cpu'],
- enumLabels: ['server default', 'cuda', 'std_default', 'cpu'],
- default: '', description: 'Affects whether a seed reproduces across backends'
- },
- samplerRngType: {
- type: 'string', title: 'Sampler RNG', enum: ['', 'cuda', 'std_default', 'cpu'],
- enumLabels: ['same as RNG above', 'cuda', 'std_default', 'cpu'],
- default: '',
- description: 'Override the RNG for sampling only. The server does not report this one, so Fill in leaves it alone'
- },
- loraApplyMode: {
- type: 'string', title: 'LoRA Apply Mode', enum: ['', 'auto', 'immediately', 'at_runtime'],
- enumLabels: ['server default', 'auto', 'immediately', 'at_runtime'],
- default: '', description: 'When a LoRA named in a prompt is applied'
- },
- vaeConvDirect: { type: 'boolean', title: 'Direct VAE Convolution', description: 'Use the ggml_conv2d_direct path for the VAE' },
- diffusionConvDirect: { type: 'boolean', title: 'Direct Diffusion Convolution', description: 'Use the ggml_conv2d_direct path for the diffusion model' },
- forceSdxlVaeConvScale: { type: 'boolean', title: 'Force SDXL VAE Conv Scale', description: 'SDXL-specific VAE convolution scaling' },
- taePreviewOnly: { type: 'boolean', title: 'TAESD For Preview Only', description: 'Load TAESD purely to render progress previews, skipping the full VAE' },
- backend: { type: 'string', title: 'Backend', description: 'Per-component placement, such as te=cpu,vae=cpu,controlnet=cpu' },
- paramsBackend: { type: 'string', title: 'Parameter Backend', description: 'Global parameter placement, such as *=cpu to hold weights in system RAM' },
- rpcServers: { type: 'string', title: 'RPC Servers', description: 'Comma separated RPC backend endpoints' },
- modelArgs: { type: 'string', title: 'Model Args', description: 'Architecture-specific key=value knobs, comma separated' },
- tensorTypeRules: { type: 'string', title: 'Tensor Type Rules', description: 'Per-tensor weight overrides using regex, such as ^vae\\.=f16' },
- options: {
- type: 'object', title: 'Other Load Options',
- description: 'Extra load options passed through, such as flash_attn, enable_mmap, weight_type, stream_layers or max_vram'
- },
- whenDifferent: {
- type: 'string', title: 'When A Different Model Is Loaded',
- enum: ['load', 'fail'],
- default: 'load',
- description: 'load swaps it. fail stops the run instead - for a workflow that depends on a particular model already being in place and should not quietly spend minutes swapping it'
- },
- force: {
- type: 'boolean', title: 'Force Reload',
- default: false,
- description: 'Load again even when the right model is already loaded. Costs the full load time; useful after changing components or options, which this node cannot see from outside',
- showWhen: { field: 'whenDifferent', value: 'load' }
- },
- timeout: {
- type: 'number', title: 'Timeout (ms)',
- description: 'How long to wait on any one request to the server. Loading reads gigabytes from disk, so the request that starts it is given this long',
- default: 300000
- },
- loadWaitMs: {
- type: 'number', title: 'Wait For Loading (ms)',
- description: 'How long to keep watching after the load has started. This is separate from the timeout above because a request giving up says nothing about whether the server is still working - it usually is, and the node follows it through /health rather than reporting a failure that is really just impatience',
- default: 900000
- }
- },
- required: []
- };
- const inputSchema = { type: 'object', properties: { data: { type: 'any' } } };
- const outputSchema = {
- type: 'object',
- properties: {
- modelName: { type: 'string', description: 'The model that is loaded now' },
- modelType: { type: 'string' },
- architecture: { type: 'string', description: 'Architecture the server detected, which decides generation defaults' },
- loaded: { type: 'boolean', description: 'True when this node performed a load' },
- alreadyLoaded: { type: 'boolean', description: 'True when the right model, with the right settings, was already in place' },
- reloadedFor: { type: 'array', description: 'Settings that differed on an otherwise-correct model, when that is why it was reloaded' },
- previousModel: { type: 'string', description: 'What was loaded before, when this node swapped it' },
- loadedComponents: { type: 'object' },
- elapsedMs: { type: 'number' }
- }
- };
- function normalizeServer(url) {
- const value = String(url || '').trim();
- if (!value) {
- throw new Error('SD.cpp: a server URL is required, such as http://localhost:8077');
- }
- return value.replace(/\/+$/, '');
- }
- function readCredential(credentialId) {
- const auth = smartbotic.credentials.get(credentialId);
- if (!auth || auth.success !== true) {
- throw new Error('SD.cpp: could not read the credential: ' +
- ((auth && auth.error) || 'unknown error'));
- }
- const value = auth.headerValue || '';
- if (value.indexOf('Basic ') !== 0) {
- throw new Error('SD.cpp: the credential must be a basic one, holding the sdcpp-restapi ' +
- 'username and password');
- }
- const decoded = smartbotic.utils.base64Decode(value.substring(6));
- const separator = decoded.indexOf(':');
- if (separator < 1) {
- throw new Error('SD.cpp: the credential is malformed, expected a username and a password');
- }
- return {
- username: decoded.substring(0, separator),
- password: decoded.substring(separator + 1)
- };
- }
- function call(options) {
- const response = smartbotic.http.request(options);
- let body = response.data;
- if (typeof body === 'string' && body.length > 0) {
- try {
- body = JSON.parse(body);
- } catch (e) {
- const snippet = body.substring(0, 200).replace(/\s+/g, ' ');
- throw new Error('SD.cpp: ' + options.what + ' returned HTTP ' + response.status +
- ' with a body that is not JSON: ' + snippet);
- }
- }
- if (response.status < 200 || response.status >= 300) {
- const detail = (body && (body.message || body.error)) || ('HTTP ' + response.status);
- throw new Error('SD.cpp: ' + options.what + ' failed: ' + detail);
- }
- return body || {};
- }
- function login(server, credential, timeout) {
- const session = call({
- method: 'POST',
- url: server + '/auth/login',
- headers: { 'Content-Type': 'application/json' },
- body: JSON.stringify({
- username: credential.username,
- password: credential.password
- }),
- timeout: timeout,
- what: 'signing in'
- });
- if (!session.token) {
- throw new Error('SD.cpp: the server accepted the login but returned no token');
- }
- return session.token;
- }
- // /health is unauthenticated, and it is the only way to find out what is
- // already loaded without asking for a token first.
- function readHealth(server, timeout) {
- return call({
- method: 'GET',
- url: server + '/health',
- timeout: timeout,
- // Reading health is safe to repeat, and the moment it matters most is
- // the moment the server is busiest.
- retries: 2,
- retryDelayMs: 1000,
- what: 'reading server health'
- });
- }
- // The same, but a server that does not answer is treated as one that is busy
- // rather than one that has failed. A machine part-way through loading eleven
- // gigabytes answers /health slowly or not at all - that is what loading looks
- // like from outside, and throwing there ended the whole run for the one thing
- // the node was waiting for.
- function pollHealth(server, timeout) {
- try {
- return readHealth(server, timeout);
- } catch (e) {
- smartbotic.log.info('SD.cpp: no answer from /health while loading (' +
- ((e && e.message) || e) + '), still waiting');
- return null;
- }
- }
- function putIfSet(target, key, value) {
- if (value === undefined || value === null || value === '') {
- return;
- }
- target[key] = value;
- }
- const LOAD_OPTIONS = [
- { setting: 'flashAttn', server: 'flash_attn' },
- { setting: 'diffusionFlashAttn', server: 'diffusion_flash_attn' },
- { setting: 'enableMmap', server: 'enable_mmap' },
- { setting: 'eagerLoad', server: 'eager_load' },
- { setting: 'streamLayers', server: 'stream_layers' },
- { setting: 'maxVram', server: 'max_vram' },
- { setting: 'nThreads', server: 'n_threads' },
- { setting: 'weightType', server: 'weight_type' },
- { setting: 'vaeFormat', server: 'vae_format' },
- { setting: 'prediction', server: 'prediction' },
- { setting: 'rngType', server: 'rng_type' },
- { setting: 'samplerRngType', server: 'sampler_rng_type' },
- { setting: 'loraApplyMode', server: 'lora_apply_mode' },
- { setting: 'vaeConvDirect', server: 'vae_conv_direct' },
- { setting: 'diffusionConvDirect', server: 'diffusion_conv_direct' },
- { setting: 'taePreviewOnly', server: 'tae_preview_only' },
- { setting: 'forceSdxlVaeConvScale', server: 'force_sdxl_vae_conv_scale' },
- { setting: 'backend', server: 'backend' },
- { setting: 'paramsBackend', server: 'params_backend' },
- { setting: 'rpcServers', server: 'rpc_servers' },
- { setting: 'modelArgs', server: 'model_args' },
- { setting: 'tensorTypeRules', server: 'tensor_type_rules' },
- ];
- // The options this node asks for, as the server names them. Only settings that
- // were actually filled in are included: an untouched setting means "whatever
- // the server does", not "the default", so it is neither sent nor compared.
- function wantedOptions(config) {
- var wanted = {};
- for (var i = 0; i < LOAD_OPTIONS.length; i++) {
- var entry = LOAD_OPTIONS[i];
- var value = config[entry.setting];
- if (value === undefined || value === null || value === '') continue;
- if (typeof value === 'number' && !isFinite(value)) continue;
- wanted[entry.server] = value;
- }
- if (config.options && typeof config.options === 'object') {
- var keys = Object.keys(config.options);
- for (var k = 0; k < keys.length; k++) {
- var extra = config.options[keys[k]];
- if (extra !== undefined && extra !== null && extra !== '') {
- wanted[keys[k]] = extra;
- }
- }
- }
- return wanted;
- }
- // Which of them the server is not currently loaded with.
- //
- // This is what makes the node able to correct a server someone else changed:
- // the same model loaded with streaming off is not the same thing as the model
- // this workflow needs, and reloading it is the whole point of saying so here.
- function optionsThatDiffer(wanted, current) {
- var differing = [];
- var keys = Object.keys(wanted);
- for (var i = 0; i < keys.length; i++) {
- var key = keys[i];
- var have = current ? current[key] : undefined;
- var want = wanted[key];
- // Numbers arrive as 0 or 0.0 depending on the field, and a boolean may
- // come back as a string from a form, so compare on value rather than
- // on type.
- var same = (typeof want === 'number' || typeof have === 'number')
- ? Number(have) === Number(want)
- : String(have) === String(want);
- if (!same) {
- differing.push(key + ': server has ' + JSON.stringify(have) +
- ', this node wants ' + JSON.stringify(want));
- }
- }
- return differing;
- }
- async function execute(config, input, context) {
- const server = normalizeServer(config.serverUrl);
- const timeout = config.timeout > 0 ? config.timeout : 300000;
- // Watching costs nothing, so it is allowed to outlast any single request.
- const loadWait = config.loadWaitMs > 0 ? config.loadWaitMs : 900000;
- const modelName = String(config.modelName || '').trim();
- if (!modelName) {
- throw new Error('SD.cpp: a model name is required. Connect an SD.cpp Model node, ' +
- 'or type the file name');
- }
- // Ask what is loaded before loading anything. A load takes minutes and
- // unloads whatever was there, so doing it when the right model is already
- // resident is pure cost - and on a shared server it disrupts other work.
- const startedAt = Date.now();
- const health = readHealth(server, Math.min(timeout, 15000));
- const current = health.model_name || '';
- const sameModel = current === modelName;
- // The same model loaded with different settings is not the model this
- // workflow asked for. The server swaps models between queue items, so what
- // is loaded now may have been put there by something else entirely.
- const wanted = wantedOptions(config);
- const differing = optionsThatDiffer(wanted, health.load_options || {});
- if (sameModel && differing.length === 0 && config.force !== true) {
- smartbotic.log.info('SD.cpp: ' + modelName + ' is already loaded with the wanted settings');
- return {
- modelName: current,
- modelType: health.model_type || '',
- architecture: health.model_architecture || '',
- loaded: false,
- alreadyLoaded: true,
- reloadedFor: [],
- previousModel: '',
- loadedComponents: health.loaded_components || {},
- elapsedMs: Date.now() - startedAt
- };
- }
- if ((config.whenDifferent || 'load') === 'fail' && (!sameModel || differing.length > 0)) {
- if (!sameModel) {
- throw new Error('SD.cpp: this workflow expects "' + modelName + '" to be loaded, but ' +
- (current ? 'the server has "' + current + '"' : 'no model is loaded') +
- '. Set When A Different Model Is Loaded to "load" to swap it automatically');
- }
- throw new Error('SD.cpp: "' + modelName + '" is loaded, but not with the settings this ' +
- 'workflow needs - ' + differing.join('; ') +
- '. Set When A Different Model Is Loaded to "load" to reload it');
- }
- if (sameModel && differing.length > 0) {
- smartbotic.log.info('SD.cpp: reloading ' + modelName + ' because ' + differing.join('; '));
- }
- const credential = readCredential(config.credentialId);
- const token = login(server, credential, Math.min(timeout, 30000));
- const body = { model_name: modelName };
- putIfSet(body, 'model_type', config.modelType);
- putIfSet(body, 'vae', config.vae);
- putIfSet(body, 'clip_l', config.clipL);
- putIfSet(body, 'clip_g', config.clipG);
- putIfSet(body, 't5xxl', config.t5xxl);
- putIfSet(body, 'llm', config.llm);
- putIfSet(body, 'taesd', config.taesd);
- putIfSet(body, 'controlnet', config.controlnet);
- if (Object.keys(wanted).length > 0) {
- body.options = wanted;
- }
- smartbotic.log.info('SD.cpp: loading ' + modelName +
- (current ? ' (replacing ' + current + ')' : ''));
- // The slot has to be emptied first. The API documentation says a load
- // replaces whatever is there, but the server answers 409 "A model is
- // already loaded. Call POST /models/unload first" - so it is unloaded here
- // rather than leaving every reload to fail on a server that already has a
- // model. (A refused load is harmless: the resident model stays put.)
- //
- // This is also the only way to change the settings of a model that is
- // already loaded, which is the case this node exists to handle.
- //
- // It does mean everything between here and a finished load runs with the
- // server holding nothing. The server never unloads on its own, so an empty
- // slot afterwards is always something that happened in this window - which
- // is why the failure paths below say so rather than leaving the next run to
- // discover it.
- let emptiedTheSlot = false;
- if (health.model_loaded === true) {
- call({
- method: 'POST',
- url: server + '/models/unload',
- headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token },
- body: '{}',
- timeout: Math.min(timeout, 60000),
- what: 'unloading ' + (current || 'the current model') + ' before loading ' + modelName
- });
- smartbotic.log.info('SD.cpp: unloaded ' + (current || 'the previous model'));
- emptiedTheSlot = true;
- }
- // Loading unloads whatever was in the slot first, and the server holds a
- // mutex for the duration, so this blocks until the weights are resident.
- let loaded;
- try {
- loaded = call({
- method: 'POST',
- url: server + '/models/load',
- headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token },
- body: JSON.stringify(body),
- timeout: timeout,
- // Deliberately not retried: the server holds a mutex for the whole
- // load, so a second request would queue behind the first and load
- // the same weights twice.
- what: 'loading model ' + modelName
- });
- } catch (loadError) {
- // This call giving up does not mean the server did. If it is still
- // loading, that is the answer to what happened - so the wait below
- // finds out how it goes rather than reporting a failure that is really
- // just impatience.
- const probe = pollHealth(server, 15000);
- if (!probe || probe.model_loading !== true) {
- // The slot was emptied to make room and the load did not take, so
- // the server now holds nothing. One more attempt is worth it: there
- // is nothing left to lose, the usual cause is a moment of
- // slowness, and the alternative is leaving the server worse than it
- // was found.
- if (emptiedTheSlot && (!probe || probe.model_loaded !== true)) {
- smartbotic.log.warn('SD.cpp: the load failed and the server now has no model. ' +
- 'Trying once more before giving up');
- try {
- loaded = call({
- method: 'POST',
- url: server + '/models/load',
- headers: { 'Content-Type': 'application/json',
- 'Authorization': 'Bearer ' + token },
- body: JSON.stringify(body),
- timeout: timeout,
- what: 'loading model ' + modelName + ' (second attempt)'
- });
- // Falls through to the wait below, the same as a first
- // attempt that worked - the model still has to finish
- // loading either way.
- } catch (secondError) {
- throw new Error('SD.cpp: could not load ' + modelName + ', and the server ' +
- 'is now holding no model at all - it was unloaded to make room. ' +
- 'Nothing will generate until a load succeeds. First attempt: ' +
- ((loadError && loadError.message) || loadError) + '. Second: ' +
- ((secondError && secondError.message) || secondError));
- }
- }
- throw loadError;
- }
- smartbotic.log.info('SD.cpp: the load request stopped waiting, but the server is still ' +
- 'loading ' + (probe.loading_model_name || modelName) + ' - following it through /health');
- loaded = {};
- }
- // The load call comes back before the model is in memory. The API
- // documentation describes it as blocking, and it is not: /health reports
- // model_loading with a step count for some time afterwards. Returning here
- // would tell the workflow the model is ready and let the next node ask it
- // to generate, which fails with "no model loaded" - a confusing way to
- // learn that this node lied.
- const deadline = Date.now() + loadWait;
- let after = pollHealth(server, 15000);
- let lastStep = -1;
- let silentPolls = 0;
- while ((after === null || after.model_loading === true) && Date.now() < deadline) {
- if (after === null) {
- silentPolls++;
- } else {
- silentPolls = 0;
- const step = after.loading_step;
- const total = after.loading_total_steps;
- if (typeof step === 'number' && step !== lastStep) {
- lastStep = step;
- smartbotic.log.info('SD.cpp: loading ' + (after.loading_model_name || modelName) +
- ' - ' + step + (total ? '/' + total : ''));
- }
- }
- smartbotic.utils.sleep(2000);
- after = pollHealth(server, 15000);
- }
- const waitedSeconds = Math.round((Date.now() - startedAt) / 1000);
- if (after === null) {
- throw new Error('SD.cpp: ' + modelName + ' was asked for ' + waitedSeconds +
- 's ago and the server has stopped answering /health (' + silentPolls +
- ' polls in a row went unanswered). It may still be loading - check the ' +
- 'server, and raise the timeout on this node if this model is simply slow');
- }
- if (after.model_loading === true) {
- throw new Error('SD.cpp: ' + modelName + ' was still loading after ' + waitedSeconds +
- 's' + (typeof after.loading_step === 'number'
- ? ' (at step ' + after.loading_step +
- (after.loading_total_steps ? ' of ' + after.loading_total_steps : '') + ')'
- : '') +
- '. It may still finish on the server; raise the timeout on this node if this ' +
- 'model is simply slow to load');
- }
- if (after.model_loaded !== true) {
- throw new Error('SD.cpp: the server accepted the load but has no model loaded afterwards' +
- (after.last_error ? ': ' + after.last_error : ''));
- }
- return {
- modelName: loaded.model_name || modelName,
- modelType: loaded.model_type || config.modelType || '',
- architecture: after.model_architecture || '',
- loaded: true,
- alreadyLoaded: false,
- // Empty when the model itself changed; otherwise the settings that
- // forced a reload of a model that was already there.
- reloadedFor: sameModel ? differing : [],
- previousModel: sameModel ? '' : current,
- loadedComponents: loaded.loaded_components || after.loaded_components || {},
- elapsedMs: Date.now() - startedAt
- };
- }
- module.exports = { configSchema, inputSchema, outputSchema, execute };
|