sdcpp-model-load.js 28 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615
  1. /**
  2. * @node sdcpp-model-load
  3. * @name SD.cpp Load Model
  4. * @category sdcpp
  5. * @version 1.0.0
  6. * @description Make sure a model is loaded, without reloading one that already is
  7. * @icon box
  8. */
  9. // The credential this node wants, named so it can be found. It is stored as a
  10. // plain basic credential - that is what decides how it is encrypted - and this
  11. // only says which basic credential is the SD.cpp one. Anything that accepts a
  12. // basic credential still accepts this, and this node still accepts a plain
  13. // basic credential, because the shape is identical.
  14. const credentialTypes = [
  15. {
  16. id: 'sdcpp',
  17. label: 'SD.cpp Server',
  18. baseType: 'basic',
  19. description: 'The username and password you sign in to sdcpp-restapi with. The node exchanges them for a token before every call',
  20. usernameLabel: 'Username',
  21. passwordLabel: 'Password'
  22. }
  23. ];
  24. const configSchema = {
  25. type: 'object',
  26. uiGroups: [
  27. { title: 'Connection', fields: ['serverUrl', 'credentialId'] },
  28. { title: 'Model', fields: ['modelName', 'modelType', 'whenDifferent', 'force'] },
  29. { title: 'Components', fields: ['vae', 'clipL', 'clipG', 't5xxl', 'llm', 'taesd', 'controlnet'] },
  30. { title: 'Loading', fields: ['flashAttn', 'diffusionFlashAttn', 'enableMmap', 'eagerLoad',
  31. 'streamLayers', 'maxVram', 'nThreads', 'weightType'] },
  32. { title: 'Advanced', fields: ['vaeFormat', 'prediction', 'rngType', 'samplerRngType',
  33. 'loraApplyMode', 'vaeConvDirect', 'diffusionConvDirect',
  34. 'taePreviewOnly', 'forceSdxlVaeConvScale', 'backend', 'paramsBackend', 'rpcServers',
  35. 'modelArgs', 'tensorTypeRules', 'options', 'timeout'] }
  36. ],
  37. // Pressing this asks the server what it has loaded and writes it into the
  38. // settings below - the model and every component it reports - so a workflow
  39. // can be built from a server that is already set up the way it should be,
  40. // rather than by typing the same names again.
  41. prefill: {
  42. label: 'Take from the server',
  43. description: 'Fill these in from the model the server currently has loaded',
  44. node: 'sdcpp-health',
  45. needs: ['serverUrl'],
  46. map: {
  47. modelName: 'modelName',
  48. modelType: 'modelType',
  49. 'loadedComponents.vae': 'vae',
  50. 'loadedComponents.clip_l': 'clipL',
  51. 'loadedComponents.clip_g': 'clipG',
  52. 'loadedComponents.t5xxl': 't5xxl',
  53. 'loadedComponents.llm': 'llm',
  54. 'loadedComponents.taesd': 'taesd',
  55. 'loadedComponents.controlnet': 'controlnet',
  56. 'loadOptions.flash_attn': 'flashAttn',
  57. 'loadOptions.diffusion_flash_attn': 'diffusionFlashAttn',
  58. 'loadOptions.enable_mmap': 'enableMmap',
  59. 'loadOptions.eager_load': 'eagerLoad',
  60. 'loadOptions.stream_layers': 'streamLayers',
  61. 'loadOptions.max_vram': 'maxVram',
  62. 'loadOptions.n_threads': 'nThreads',
  63. 'loadOptions.vae_format': 'vaeFormat',
  64. 'loadOptions.rng_type': 'rngType',
  65. 'loadOptions.lora_apply_mode': 'loraApplyMode',
  66. 'loadOptions.vae_conv_direct': 'vaeConvDirect',
  67. 'loadOptions.diffusion_conv_direct': 'diffusionConvDirect',
  68. 'loadOptions.tae_preview_only': 'taePreviewOnly',
  69. 'loadOptions.force_sdxl_vae_conv_scale': 'forceSdxlVaeConvScale',
  70. 'loadOptions.rng_type': 'rngType',
  71. 'loadOptions.lora_apply_mode': 'loraApplyMode',
  72. 'loadOptions.backend': 'backend',
  73. 'loadOptions.rpc_servers': 'rpcServers',
  74. 'loadOptions.model_args': 'modelArgs',
  75. 'loadOptions.backend': 'backend',
  76. 'loadOptions.params_backend': 'paramsBackend',
  77. 'loadOptions.rpc_servers': 'rpcServers',
  78. 'loadOptions.model_args': 'modelArgs'
  79. }
  80. },
  81. properties: {
  82. serverUrl: {
  83. type: 'string', title: 'Server URL',
  84. description: 'Base address of the sdcpp-restapi server',
  85. default: 'http://localhost:8077'
  86. },
  87. credentialId: {
  88. type: 'string', title: 'Credential',
  89. description: 'A basic credential holding the sdcpp-restapi username and password',
  90. dynamicOptions: { source: 'credentials', filter: { type: ['sdcpp', 'basic'] } }
  91. },
  92. modelName: {
  93. type: 'string', title: 'Model',
  94. description: 'File name of the model, relative to its type directory. Browse lists what the server has of the Model Type chosen below',
  95. dynamicOptions: {
  96. source: 'node',
  97. node: 'sdcpp-model',
  98. config: { listOnly: true },
  99. itemsPath: 'models',
  100. valueKey: 'name',
  101. labelKey: 'name',
  102. needs: ['serverUrl', 'credentialId']
  103. }
  104. },
  105. modelType: {
  106. type: 'string', title: 'Model Type',
  107. enum: ['', 'checkpoint', 'diffusion'],
  108. default: '',
  109. description: 'checkpoint bundles U-Net, CLIP and VAE and suits SD1, SD2 and SDXL. diffusion holds only the U-Net or DiT and needs its components named separately, which is how Flux, SD3, Qwen, Wan and Z-Image load'
  110. },
  111. vae: {
  112. type: 'string', title: 'VAE',
  113. description: 'Component file name',
  114. dynamicOptions: {
  115. source: 'node',
  116. node: 'sdcpp-model',
  117. config: { listOnly: true, modelType: 'vae' },
  118. itemsPath: 'models',
  119. valueKey: 'name',
  120. labelKey: 'name',
  121. needs: ['serverUrl', 'credentialId']
  122. }
  123. },
  124. clipL: {
  125. type: 'string', title: 'CLIP-L',
  126. description: 'Component file name',
  127. dynamicOptions: {
  128. source: 'node',
  129. node: 'sdcpp-model',
  130. config: { listOnly: true, modelType: 'clip' },
  131. itemsPath: 'models',
  132. valueKey: 'name',
  133. labelKey: 'name',
  134. needs: ['serverUrl', 'credentialId']
  135. }
  136. },
  137. clipG: {
  138. type: 'string', title: 'CLIP-G',
  139. description: 'Component file name',
  140. dynamicOptions: {
  141. source: 'node',
  142. node: 'sdcpp-model',
  143. config: { listOnly: true, modelType: 'clip' },
  144. itemsPath: 'models',
  145. valueKey: 'name',
  146. labelKey: 'name',
  147. needs: ['serverUrl', 'credentialId']
  148. }
  149. },
  150. t5xxl: {
  151. type: 'string', title: 'T5-XXL',
  152. description: 'Component file name',
  153. dynamicOptions: {
  154. source: 'node',
  155. node: 'sdcpp-model',
  156. config: { listOnly: true, modelType: 't5' },
  157. itemsPath: 'models',
  158. valueKey: 'name',
  159. labelKey: 'name',
  160. needs: ['serverUrl', 'credentialId']
  161. }
  162. },
  163. llm: {
  164. type: 'string', title: 'LLM',
  165. description: 'Component file name, used by Z-Image, Qwen, Anima and Flux2',
  166. dynamicOptions: {
  167. source: 'node',
  168. node: 'sdcpp-model',
  169. config: { listOnly: true, modelType: 'llm' },
  170. itemsPath: 'models',
  171. valueKey: 'name',
  172. labelKey: 'name',
  173. needs: ['serverUrl', 'credentialId']
  174. }
  175. },
  176. taesd: {
  177. type: 'string', title: 'TAESD',
  178. description: 'Tiny autoencoder for progress previews',
  179. dynamicOptions: {
  180. source: 'node',
  181. node: 'sdcpp-model',
  182. config: { listOnly: true, modelType: 'taesd' },
  183. itemsPath: 'models',
  184. valueKey: 'name',
  185. labelKey: 'name',
  186. needs: ['serverUrl', 'credentialId']
  187. }
  188. },
  189. controlnet: {
  190. type: 'string', title: 'ControlNet',
  191. description: 'Component file name',
  192. dynamicOptions: {
  193. source: 'node',
  194. node: 'sdcpp-model',
  195. config: { listOnly: true, modelType: 'controlnet' },
  196. itemsPath: 'models',
  197. valueKey: 'name',
  198. labelKey: 'name',
  199. needs: ['serverUrl', 'credentialId']
  200. }
  201. },
  202. flashAttn: { type: 'boolean', title: 'Flash Attention', description: 'For CLIP and T5. A large speed and memory win on modern GPUs' },
  203. diffusionFlashAttn: { type: 'boolean', title: 'Flash Attention (diffusion)', description: 'Flash attention for the diffusion model specifically' },
  204. enableMmap: { type: 'boolean', title: 'Memory-map Weights', description: 'Recommended for large files' },
  205. eagerLoad: { type: 'boolean', title: 'Eager Load', description: 'Move every parameter to the compute backend at load time instead of on demand' },
  206. streamLayers: { type: 'boolean', title: 'Stream Layers', description: 'Stream diffusion layers when the model does not fit in VRAM. Pair with a VRAM budget' },
  207. maxVram: { type: 'number', title: 'VRAM Budget (GiB)', description: 'Budget for segmented parameter offload. 0 leaves it to the server' },
  208. nThreads: { type: 'number', title: 'CPU Threads', description: '-1 lets the server decide' },
  209. weightType: {
  210. type: 'string', title: 'Weight Type',
  211. enum: ['', 'f32', 'f16', 'bf16', 'q8_0', 'q5_0', 'q5_1', 'q4_0', 'q4_1', 'q4_k', 'q5_k', 'q6_k', 'q8_k', 'q3_k', 'q2_k', 'mxfp4', 'nvfp4', 'q1_0'],
  212. enumLabels: ['auto - use the weights in the file', 'f32', 'f16', 'bf16', 'q8_0', 'q5_0', 'q5_1', 'q4_0', 'q4_1', 'q4_k', 'q5_k', 'q6_k', 'q8_k', 'q3_k', 'q2_k', 'mxfp4', 'nvfp4', 'q1_0'],
  213. default: '', description: 'Force a quantisation. Left on auto, sd.cpp uses whatever the model file holds'
  214. },
  215. vaeFormat: {
  216. type: 'string', title: 'VAE Format',
  217. enum: ['', 'auto', 'flux', 'sd3', 'flux2', 'wan'],
  218. enumLabels: ['not set - leave it to the server', 'auto - detect from the file', 'flux', 'sd3', 'flux2', 'wan'],
  219. default: '', description: 'Override VAE format detection'
  220. },
  221. prediction: {
  222. type: 'string', title: 'Prediction Type',
  223. enum: ['', 'eps', 'v', 'edm_v', 'sd3_flow', 'flux_flow', 'flux2_flow', 'sefi_flow', 'minit2i_flow'],
  224. enumLabels: ['auto - detect from the model', 'eps', 'v', 'edm_v', 'sd3_flow', 'flux_flow', 'flux2_flow', 'sefi_flow', 'minit2i_flow'],
  225. default: '', description: 'Override the prediction type. Left on auto it is detected from the model'
  226. },
  227. rngType: {
  228. type: 'string', title: 'RNG', enum: ['', 'cuda', 'std_default', 'cpu'],
  229. enumLabels: ['server default', 'cuda', 'std_default', 'cpu'],
  230. default: '', description: 'Affects whether a seed reproduces across backends'
  231. },
  232. samplerRngType: {
  233. type: 'string', title: 'Sampler RNG', enum: ['', 'cuda', 'std_default', 'cpu'],
  234. enumLabels: ['same as RNG above', 'cuda', 'std_default', 'cpu'],
  235. default: '',
  236. description: 'Override the RNG for sampling only. The server does not report this one, so Fill in leaves it alone'
  237. },
  238. loraApplyMode: {
  239. type: 'string', title: 'LoRA Apply Mode', enum: ['', 'auto', 'immediately', 'at_runtime'],
  240. enumLabels: ['server default', 'auto', 'immediately', 'at_runtime'],
  241. default: '', description: 'When a LoRA named in a prompt is applied'
  242. },
  243. vaeConvDirect: { type: 'boolean', title: 'Direct VAE Convolution', description: 'Use the ggml_conv2d_direct path for the VAE' },
  244. diffusionConvDirect: { type: 'boolean', title: 'Direct Diffusion Convolution', description: 'Use the ggml_conv2d_direct path for the diffusion model' },
  245. forceSdxlVaeConvScale: { type: 'boolean', title: 'Force SDXL VAE Conv Scale', description: 'SDXL-specific VAE convolution scaling' },
  246. taePreviewOnly: { type: 'boolean', title: 'TAESD For Preview Only', description: 'Load TAESD purely to render progress previews, skipping the full VAE' },
  247. backend: { type: 'string', title: 'Backend', description: 'Per-component placement, such as te=cpu,vae=cpu,controlnet=cpu' },
  248. paramsBackend: { type: 'string', title: 'Parameter Backend', description: 'Global parameter placement, such as *=cpu to hold weights in system RAM' },
  249. rpcServers: { type: 'string', title: 'RPC Servers', description: 'Comma separated RPC backend endpoints' },
  250. modelArgs: { type: 'string', title: 'Model Args', description: 'Architecture-specific key=value knobs, comma separated' },
  251. tensorTypeRules: { type: 'string', title: 'Tensor Type Rules', description: 'Per-tensor weight overrides using regex, such as ^vae\\.=f16' },
  252. options: {
  253. type: 'object', title: 'Other Load Options',
  254. description: 'Extra load options passed through, such as flash_attn, enable_mmap, weight_type, stream_layers or max_vram'
  255. },
  256. whenDifferent: {
  257. type: 'string', title: 'When A Different Model Is Loaded',
  258. enum: ['load', 'fail'],
  259. default: 'load',
  260. description: 'load swaps it. fail stops the run instead - for a workflow that depends on a particular model already being in place and should not quietly spend minutes swapping it'
  261. },
  262. force: {
  263. type: 'boolean', title: 'Force Reload',
  264. default: false,
  265. description: 'Load again even when the right model is already loaded. Costs the full load time; useful after changing components or options, which this node cannot see from outside',
  266. showWhen: { field: 'whenDifferent', value: 'load' }
  267. },
  268. timeout: {
  269. type: 'number', title: 'Timeout (ms)',
  270. description: 'Loading reads gigabytes from disk and can take minutes',
  271. default: 300000
  272. }
  273. },
  274. required: []
  275. };
  276. const inputSchema = { type: 'object', properties: { data: { type: 'any' } } };
  277. const outputSchema = {
  278. type: 'object',
  279. properties: {
  280. modelName: { type: 'string', description: 'The model that is loaded now' },
  281. modelType: { type: 'string' },
  282. architecture: { type: 'string', description: 'Architecture the server detected, which decides generation defaults' },
  283. loaded: { type: 'boolean', description: 'True when this node performed a load' },
  284. alreadyLoaded: { type: 'boolean', description: 'True when the right model, with the right settings, was already in place' },
  285. reloadedFor: { type: 'array', description: 'Settings that differed on an otherwise-correct model, when that is why it was reloaded' },
  286. previousModel: { type: 'string', description: 'What was loaded before, when this node swapped it' },
  287. loadedComponents: { type: 'object' },
  288. elapsedMs: { type: 'number' }
  289. }
  290. };
  291. function normalizeServer(url) {
  292. const value = String(url || '').trim();
  293. if (!value) {
  294. throw new Error('SD.cpp: a server URL is required, such as http://localhost:8077');
  295. }
  296. return value.replace(/\/+$/, '');
  297. }
  298. function readCredential(credentialId) {
  299. const auth = smartbotic.credentials.get(credentialId);
  300. if (!auth || auth.success !== true) {
  301. throw new Error('SD.cpp: could not read the credential: ' +
  302. ((auth && auth.error) || 'unknown error'));
  303. }
  304. const value = auth.headerValue || '';
  305. if (value.indexOf('Basic ') !== 0) {
  306. throw new Error('SD.cpp: the credential must be a basic one, holding the sdcpp-restapi ' +
  307. 'username and password');
  308. }
  309. const decoded = smartbotic.utils.base64Decode(value.substring(6));
  310. const separator = decoded.indexOf(':');
  311. if (separator < 1) {
  312. throw new Error('SD.cpp: the credential is malformed, expected a username and a password');
  313. }
  314. return {
  315. username: decoded.substring(0, separator),
  316. password: decoded.substring(separator + 1)
  317. };
  318. }
  319. function call(options) {
  320. const response = smartbotic.http.request(options);
  321. let body = response.data;
  322. if (typeof body === 'string' && body.length > 0) {
  323. try {
  324. body = JSON.parse(body);
  325. } catch (e) {
  326. const snippet = body.substring(0, 200).replace(/\s+/g, ' ');
  327. throw new Error('SD.cpp: ' + options.what + ' returned HTTP ' + response.status +
  328. ' with a body that is not JSON: ' + snippet);
  329. }
  330. }
  331. if (response.status < 200 || response.status >= 300) {
  332. const detail = (body && (body.message || body.error)) || ('HTTP ' + response.status);
  333. throw new Error('SD.cpp: ' + options.what + ' failed: ' + detail);
  334. }
  335. return body || {};
  336. }
  337. function login(server, credential, timeout) {
  338. const session = call({
  339. method: 'POST',
  340. url: server + '/auth/login',
  341. headers: { 'Content-Type': 'application/json' },
  342. body: JSON.stringify({
  343. username: credential.username,
  344. password: credential.password
  345. }),
  346. timeout: timeout,
  347. what: 'signing in'
  348. });
  349. if (!session.token) {
  350. throw new Error('SD.cpp: the server accepted the login but returned no token');
  351. }
  352. return session.token;
  353. }
  354. // /health is unauthenticated, and it is the only way to find out what is
  355. // already loaded without asking for a token first.
  356. function readHealth(server, timeout) {
  357. return call({
  358. method: 'GET',
  359. url: server + '/health',
  360. timeout: timeout,
  361. what: 'reading server health'
  362. });
  363. }
  364. function putIfSet(target, key, value) {
  365. if (value === undefined || value === null || value === '') {
  366. return;
  367. }
  368. target[key] = value;
  369. }
  370. const LOAD_OPTIONS = [
  371. { setting: 'flashAttn', server: 'flash_attn' },
  372. { setting: 'diffusionFlashAttn', server: 'diffusion_flash_attn' },
  373. { setting: 'enableMmap', server: 'enable_mmap' },
  374. { setting: 'eagerLoad', server: 'eager_load' },
  375. { setting: 'streamLayers', server: 'stream_layers' },
  376. { setting: 'maxVram', server: 'max_vram' },
  377. { setting: 'nThreads', server: 'n_threads' },
  378. { setting: 'weightType', server: 'weight_type' },
  379. { setting: 'vaeFormat', server: 'vae_format' },
  380. { setting: 'prediction', server: 'prediction' },
  381. { setting: 'rngType', server: 'rng_type' },
  382. { setting: 'samplerRngType', server: 'sampler_rng_type' },
  383. { setting: 'loraApplyMode', server: 'lora_apply_mode' },
  384. { setting: 'vaeConvDirect', server: 'vae_conv_direct' },
  385. { setting: 'diffusionConvDirect', server: 'diffusion_conv_direct' },
  386. { setting: 'taePreviewOnly', server: 'tae_preview_only' },
  387. { setting: 'forceSdxlVaeConvScale', server: 'force_sdxl_vae_conv_scale' },
  388. { setting: 'backend', server: 'backend' },
  389. { setting: 'paramsBackend', server: 'params_backend' },
  390. { setting: 'rpcServers', server: 'rpc_servers' },
  391. { setting: 'modelArgs', server: 'model_args' },
  392. { setting: 'tensorTypeRules', server: 'tensor_type_rules' },
  393. ];
  394. // The options this node asks for, as the server names them. Only settings that
  395. // were actually filled in are included: an untouched setting means "whatever
  396. // the server does", not "the default", so it is neither sent nor compared.
  397. function wantedOptions(config) {
  398. var wanted = {};
  399. for (var i = 0; i < LOAD_OPTIONS.length; i++) {
  400. var entry = LOAD_OPTIONS[i];
  401. var value = config[entry.setting];
  402. if (value === undefined || value === null || value === '') continue;
  403. if (typeof value === 'number' && !isFinite(value)) continue;
  404. wanted[entry.server] = value;
  405. }
  406. if (config.options && typeof config.options === 'object') {
  407. var keys = Object.keys(config.options);
  408. for (var k = 0; k < keys.length; k++) {
  409. var extra = config.options[keys[k]];
  410. if (extra !== undefined && extra !== null && extra !== '') {
  411. wanted[keys[k]] = extra;
  412. }
  413. }
  414. }
  415. return wanted;
  416. }
  417. // Which of them the server is not currently loaded with.
  418. //
  419. // This is what makes the node able to correct a server someone else changed:
  420. // the same model loaded with streaming off is not the same thing as the model
  421. // this workflow needs, and reloading it is the whole point of saying so here.
  422. function optionsThatDiffer(wanted, current) {
  423. var differing = [];
  424. var keys = Object.keys(wanted);
  425. for (var i = 0; i < keys.length; i++) {
  426. var key = keys[i];
  427. var have = current ? current[key] : undefined;
  428. var want = wanted[key];
  429. // Numbers arrive as 0 or 0.0 depending on the field, and a boolean may
  430. // come back as a string from a form, so compare on value rather than
  431. // on type.
  432. var same = (typeof want === 'number' || typeof have === 'number')
  433. ? Number(have) === Number(want)
  434. : String(have) === String(want);
  435. if (!same) {
  436. differing.push(key + ': server has ' + JSON.stringify(have) +
  437. ', this node wants ' + JSON.stringify(want));
  438. }
  439. }
  440. return differing;
  441. }
  442. async function execute(config, input, context) {
  443. const server = normalizeServer(config.serverUrl);
  444. const timeout = config.timeout > 0 ? config.timeout : 300000;
  445. const modelName = String(config.modelName || '').trim();
  446. if (!modelName) {
  447. throw new Error('SD.cpp: a model name is required. Connect an SD.cpp Model node, ' +
  448. 'or type the file name');
  449. }
  450. // Ask what is loaded before loading anything. A load takes minutes and
  451. // unloads whatever was there, so doing it when the right model is already
  452. // resident is pure cost - and on a shared server it disrupts other work.
  453. const startedAt = Date.now();
  454. const health = readHealth(server, Math.min(timeout, 15000));
  455. const current = health.model_name || '';
  456. const sameModel = current === modelName;
  457. // The same model loaded with different settings is not the model this
  458. // workflow asked for. The server swaps models between queue items, so what
  459. // is loaded now may have been put there by something else entirely.
  460. const wanted = wantedOptions(config);
  461. const differing = optionsThatDiffer(wanted, health.load_options || {});
  462. if (sameModel && differing.length === 0 && config.force !== true) {
  463. smartbotic.log.info('SD.cpp: ' + modelName + ' is already loaded with the wanted settings');
  464. return {
  465. modelName: current,
  466. modelType: health.model_type || '',
  467. architecture: health.model_architecture || '',
  468. loaded: false,
  469. alreadyLoaded: true,
  470. reloadedFor: [],
  471. previousModel: '',
  472. loadedComponents: health.loaded_components || {},
  473. elapsedMs: Date.now() - startedAt
  474. };
  475. }
  476. if ((config.whenDifferent || 'load') === 'fail' && (!sameModel || differing.length > 0)) {
  477. if (!sameModel) {
  478. throw new Error('SD.cpp: this workflow expects "' + modelName + '" to be loaded, but ' +
  479. (current ? 'the server has "' + current + '"' : 'no model is loaded') +
  480. '. Set When A Different Model Is Loaded to "load" to swap it automatically');
  481. }
  482. throw new Error('SD.cpp: "' + modelName + '" is loaded, but not with the settings this ' +
  483. 'workflow needs - ' + differing.join('; ') +
  484. '. Set When A Different Model Is Loaded to "load" to reload it');
  485. }
  486. if (sameModel && differing.length > 0) {
  487. smartbotic.log.info('SD.cpp: reloading ' + modelName + ' because ' + differing.join('; '));
  488. }
  489. const credential = readCredential(config.credentialId);
  490. const token = login(server, credential, Math.min(timeout, 30000));
  491. const body = { model_name: modelName };
  492. putIfSet(body, 'model_type', config.modelType);
  493. putIfSet(body, 'vae', config.vae);
  494. putIfSet(body, 'clip_l', config.clipL);
  495. putIfSet(body, 'clip_g', config.clipG);
  496. putIfSet(body, 't5xxl', config.t5xxl);
  497. putIfSet(body, 'llm', config.llm);
  498. putIfSet(body, 'taesd', config.taesd);
  499. putIfSet(body, 'controlnet', config.controlnet);
  500. if (Object.keys(wanted).length > 0) {
  501. body.options = wanted;
  502. }
  503. smartbotic.log.info('SD.cpp: loading ' + modelName +
  504. (current ? ' (replacing ' + current + ')' : ''));
  505. // The slot has to be emptied first. The API documentation says a load
  506. // replaces whatever is there, but the server answers "A model is already
  507. // loaded. Call POST /models/unload first" - so it is unloaded here rather
  508. // than leaving every reload to fail on a server that already has a model.
  509. //
  510. // This is also the only way to change the settings of a model that is
  511. // already loaded, which is the case this node exists to handle.
  512. if (health.model_loaded === true) {
  513. call({
  514. method: 'POST',
  515. url: server + '/models/unload',
  516. headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token },
  517. body: '{}',
  518. timeout: Math.min(timeout, 60000),
  519. what: 'unloading ' + (current || 'the current model') + ' before loading ' + modelName
  520. });
  521. smartbotic.log.info('SD.cpp: unloaded ' + (current || 'the previous model'));
  522. }
  523. // Loading unloads whatever was in the slot first, and the server holds a
  524. // mutex for the duration, so this blocks until the weights are resident.
  525. const loaded = call({
  526. method: 'POST',
  527. url: server + '/models/load',
  528. headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token },
  529. body: JSON.stringify(body),
  530. timeout: timeout,
  531. what: 'loading model ' + modelName
  532. });
  533. // The load call comes back before the model is in memory. The API
  534. // documentation describes it as blocking, and it is not: /health reports
  535. // model_loading with a step count for some time afterwards. Returning here
  536. // would tell the workflow the model is ready and let the next node ask it
  537. // to generate, which fails with "no model loaded" - a confusing way to
  538. // learn that this node lied.
  539. const deadline = startedAt + timeout;
  540. let after = readHealth(server, 15000);
  541. let lastStep = -1;
  542. while (after.model_loading === true && Date.now() < deadline) {
  543. const step = after.loading_step;
  544. const total = after.loading_total_steps;
  545. if (typeof step === 'number' && step !== lastStep) {
  546. lastStep = step;
  547. smartbotic.log.info('SD.cpp: loading ' + (after.loading_model_name || modelName) +
  548. ' - ' + step + (total ? '/' + total : ''));
  549. }
  550. smartbotic.utils.sleep(2000);
  551. after = readHealth(server, 15000);
  552. }
  553. if (after.model_loading === true) {
  554. throw new Error('SD.cpp: ' + modelName + ' was still loading after ' +
  555. Math.round((Date.now() - startedAt) / 1000) + 's. It may still finish on the server; ' +
  556. 'raise the timeout on this node if this model is simply slow to load');
  557. }
  558. if (after.model_loaded !== true) {
  559. throw new Error('SD.cpp: the server accepted the load but has no model loaded afterwards' +
  560. (after.last_error ? ': ' + after.last_error : ''));
  561. }
  562. return {
  563. modelName: loaded.model_name || modelName,
  564. modelType: loaded.model_type || config.modelType || '',
  565. architecture: after.model_architecture || '',
  566. loaded: true,
  567. alreadyLoaded: false,
  568. // Empty when the model itself changed; otherwise the settings that
  569. // forced a reload of a model that was already there.
  570. reloadedFor: sameModel ? differing : [],
  571. previousModel: sameModel ? '' : current,
  572. loadedComponents: loaded.loaded_components || after.loaded_components || {},
  573. elapsedMs: Date.now() - startedAt
  574. };
  575. }
  576. module.exports = { configSchema, inputSchema, outputSchema, execute };