sdcpp-model-load.js 33 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716
  1. /**
  2. * @node sdcpp-model-load
  3. * @name SD.cpp Load Model
  4. * @category sdcpp
  5. * @version 1.0.0
  6. * @description Make sure a model is loaded, without reloading one that already is
  7. * @icon box
  8. */
  9. // The credential this node wants, named so it can be found. It is stored as a
  10. // plain basic credential - that is what decides how it is encrypted - and this
  11. // only says which basic credential is the SD.cpp one. Anything that accepts a
  12. // basic credential still accepts this, and this node still accepts a plain
  13. // basic credential, because the shape is identical.
  14. const credentialTypes = [
  15. {
  16. id: 'sdcpp',
  17. label: 'SD.cpp Server',
  18. baseType: 'basic',
  19. description: 'The username and password you sign in to sdcpp-restapi with. The node exchanges them for a token before every call',
  20. usernameLabel: 'Username',
  21. passwordLabel: 'Password'
  22. }
  23. ];
  24. const configSchema = {
  25. type: 'object',
  26. uiGroups: [
  27. { title: 'Connection', fields: ['serverUrl', 'credentialId'] },
  28. { title: 'Model', fields: ['modelName', 'modelType', 'whenDifferent', 'force'] },
  29. { title: 'Components', fields: ['vae', 'clipL', 'clipG', 't5xxl', 'llm', 'taesd', 'controlnet'] },
  30. { title: 'Loading', fields: ['flashAttn', 'diffusionFlashAttn', 'enableMmap', 'eagerLoad',
  31. 'streamLayers', 'maxVram', 'nThreads', 'weightType'] },
  32. { title: 'Advanced', fields: ['vaeFormat', 'prediction', 'rngType', 'samplerRngType',
  33. 'loraApplyMode', 'vaeConvDirect', 'diffusionConvDirect',
  34. 'taePreviewOnly', 'forceSdxlVaeConvScale', 'backend', 'paramsBackend', 'rpcServers',
  35. 'modelArgs', 'tensorTypeRules', 'options', 'timeout', 'loadWaitMs'] }
  36. ],
  37. // Pressing this asks the server what it has loaded and writes it into the
  38. // settings below - the model and every component it reports - so a workflow
  39. // can be built from a server that is already set up the way it should be,
  40. // rather than by typing the same names again.
  41. prefill: {
  42. label: 'Take from the server',
  43. description: 'Fill these in from the model the server currently has loaded',
  44. node: 'sdcpp-health',
  45. needs: ['serverUrl'],
  46. map: {
  47. modelName: 'modelName',
  48. modelType: 'modelType',
  49. 'loadedComponents.vae': 'vae',
  50. 'loadedComponents.clip_l': 'clipL',
  51. 'loadedComponents.clip_g': 'clipG',
  52. 'loadedComponents.t5xxl': 't5xxl',
  53. 'loadedComponents.llm': 'llm',
  54. 'loadedComponents.taesd': 'taesd',
  55. 'loadedComponents.controlnet': 'controlnet',
  56. 'loadOptions.flash_attn': 'flashAttn',
  57. 'loadOptions.diffusion_flash_attn': 'diffusionFlashAttn',
  58. 'loadOptions.enable_mmap': 'enableMmap',
  59. 'loadOptions.eager_load': 'eagerLoad',
  60. 'loadOptions.stream_layers': 'streamLayers',
  61. 'loadOptions.max_vram': 'maxVram',
  62. 'loadOptions.n_threads': 'nThreads',
  63. 'loadOptions.vae_format': 'vaeFormat',
  64. 'loadOptions.rng_type': 'rngType',
  65. 'loadOptions.lora_apply_mode': 'loraApplyMode',
  66. 'loadOptions.vae_conv_direct': 'vaeConvDirect',
  67. 'loadOptions.diffusion_conv_direct': 'diffusionConvDirect',
  68. 'loadOptions.tae_preview_only': 'taePreviewOnly',
  69. 'loadOptions.force_sdxl_vae_conv_scale': 'forceSdxlVaeConvScale',
  70. 'loadOptions.rng_type': 'rngType',
  71. 'loadOptions.lora_apply_mode': 'loraApplyMode',
  72. 'loadOptions.backend': 'backend',
  73. 'loadOptions.rpc_servers': 'rpcServers',
  74. 'loadOptions.model_args': 'modelArgs',
  75. 'loadOptions.backend': 'backend',
  76. 'loadOptions.params_backend': 'paramsBackend',
  77. 'loadOptions.rpc_servers': 'rpcServers',
  78. 'loadOptions.model_args': 'modelArgs'
  79. }
  80. },
  81. properties: {
  82. serverUrl: {
  83. type: 'string', title: 'Server URL',
  84. description: 'Base address of the sdcpp-restapi server',
  85. default: 'http://localhost:8077'
  86. },
  87. credentialId: {
  88. type: 'string', title: 'Credential',
  89. description: 'A basic credential holding the sdcpp-restapi username and password',
  90. dynamicOptions: { source: 'credentials', filter: { type: ['sdcpp', 'basic'] } }
  91. },
  92. modelName: {
  93. type: 'string', title: 'Model',
  94. description: 'File name of the model, relative to its type directory. Browse lists what the server has of the Model Type chosen below',
  95. dynamicOptions: {
  96. source: 'node',
  97. node: 'sdcpp-model',
  98. config: { listOnly: true },
  99. itemsPath: 'models',
  100. valueKey: 'name',
  101. labelKey: 'name',
  102. needs: ['serverUrl', 'credentialId']
  103. }
  104. },
  105. modelType: {
  106. type: 'string', title: 'Model Type',
  107. enum: ['', 'checkpoint', 'diffusion'],
  108. default: '',
  109. description: 'checkpoint bundles U-Net, CLIP and VAE and suits SD1, SD2 and SDXL. diffusion holds only the U-Net or DiT and needs its components named separately, which is how Flux, SD3, Qwen, Wan and Z-Image load'
  110. },
  111. vae: {
  112. type: 'string', title: 'VAE',
  113. description: 'Component file name',
  114. dynamicOptions: {
  115. source: 'node',
  116. node: 'sdcpp-model',
  117. config: { listOnly: true, modelType: 'vae' },
  118. itemsPath: 'models',
  119. valueKey: 'name',
  120. labelKey: 'name',
  121. needs: ['serverUrl', 'credentialId']
  122. }
  123. },
  124. clipL: {
  125. type: 'string', title: 'CLIP-L',
  126. description: 'Component file name',
  127. dynamicOptions: {
  128. source: 'node',
  129. node: 'sdcpp-model',
  130. config: { listOnly: true, modelType: 'clip' },
  131. itemsPath: 'models',
  132. valueKey: 'name',
  133. labelKey: 'name',
  134. needs: ['serverUrl', 'credentialId']
  135. }
  136. },
  137. clipG: {
  138. type: 'string', title: 'CLIP-G',
  139. description: 'Component file name',
  140. dynamicOptions: {
  141. source: 'node',
  142. node: 'sdcpp-model',
  143. config: { listOnly: true, modelType: 'clip' },
  144. itemsPath: 'models',
  145. valueKey: 'name',
  146. labelKey: 'name',
  147. needs: ['serverUrl', 'credentialId']
  148. }
  149. },
  150. t5xxl: {
  151. type: 'string', title: 'T5-XXL',
  152. description: 'Component file name',
  153. dynamicOptions: {
  154. source: 'node',
  155. node: 'sdcpp-model',
  156. config: { listOnly: true, modelType: 't5' },
  157. itemsPath: 'models',
  158. valueKey: 'name',
  159. labelKey: 'name',
  160. needs: ['serverUrl', 'credentialId']
  161. }
  162. },
  163. llm: {
  164. type: 'string', title: 'LLM',
  165. description: 'Component file name, used by Z-Image, Qwen, Anima and Flux2',
  166. dynamicOptions: {
  167. source: 'node',
  168. node: 'sdcpp-model',
  169. config: { listOnly: true, modelType: 'llm' },
  170. itemsPath: 'models',
  171. valueKey: 'name',
  172. labelKey: 'name',
  173. needs: ['serverUrl', 'credentialId']
  174. }
  175. },
  176. taesd: {
  177. type: 'string', title: 'TAESD',
  178. description: 'Tiny autoencoder for progress previews',
  179. dynamicOptions: {
  180. source: 'node',
  181. node: 'sdcpp-model',
  182. config: { listOnly: true, modelType: 'taesd' },
  183. itemsPath: 'models',
  184. valueKey: 'name',
  185. labelKey: 'name',
  186. needs: ['serverUrl', 'credentialId']
  187. }
  188. },
  189. controlnet: {
  190. type: 'string', title: 'ControlNet',
  191. description: 'Component file name',
  192. dynamicOptions: {
  193. source: 'node',
  194. node: 'sdcpp-model',
  195. config: { listOnly: true, modelType: 'controlnet' },
  196. itemsPath: 'models',
  197. valueKey: 'name',
  198. labelKey: 'name',
  199. needs: ['serverUrl', 'credentialId']
  200. }
  201. },
  202. flashAttn: { type: 'boolean', title: 'Flash Attention', description: 'For CLIP and T5. A large speed and memory win on modern GPUs' },
  203. diffusionFlashAttn: { type: 'boolean', title: 'Flash Attention (diffusion)', description: 'Flash attention for the diffusion model specifically' },
  204. enableMmap: { type: 'boolean', title: 'Memory-map Weights', description: 'Recommended for large files' },
  205. eagerLoad: { type: 'boolean', title: 'Eager Load', description: 'Move every parameter to the compute backend at load time instead of on demand' },
  206. streamLayers: { type: 'boolean', title: 'Stream Layers', description: 'Stream diffusion layers when the model does not fit in VRAM. Pair with a VRAM budget' },
  207. maxVram: { type: 'number', title: 'VRAM Budget (GiB)', description: 'Budget for segmented parameter offload. 0 leaves it to the server' },
  208. nThreads: { type: 'number', title: 'CPU Threads', description: '-1 lets the server decide' },
  209. weightType: {
  210. type: 'string', title: 'Weight Type',
  211. enum: ['', 'f32', 'f16', 'bf16', 'q8_0', 'q5_0', 'q5_1', 'q4_0', 'q4_1', 'q4_k', 'q5_k', 'q6_k', 'q8_k', 'q3_k', 'q2_k', 'mxfp4', 'nvfp4', 'q1_0'],
  212. enumLabels: ['auto - use the weights in the file', 'f32', 'f16', 'bf16', 'q8_0', 'q5_0', 'q5_1', 'q4_0', 'q4_1', 'q4_k', 'q5_k', 'q6_k', 'q8_k', 'q3_k', 'q2_k', 'mxfp4', 'nvfp4', 'q1_0'],
  213. default: '', description: 'Force a quantisation. Left on auto, sd.cpp uses whatever the model file holds'
  214. },
  215. vaeFormat: {
  216. type: 'string', title: 'VAE Format',
  217. enum: ['', 'auto', 'flux', 'sd3', 'flux2', 'wan'],
  218. enumLabels: ['not set - leave it to the server', 'auto - detect from the file', 'flux', 'sd3', 'flux2', 'wan'],
  219. default: '', description: 'Override VAE format detection'
  220. },
  221. prediction: {
  222. type: 'string', title: 'Prediction Type',
  223. enum: ['', 'eps', 'v', 'edm_v', 'sd3_flow', 'flux_flow', 'flux2_flow', 'sefi_flow', 'minit2i_flow'],
  224. enumLabels: ['auto - detect from the model', 'eps', 'v', 'edm_v', 'sd3_flow', 'flux_flow', 'flux2_flow', 'sefi_flow', 'minit2i_flow'],
  225. default: '', description: 'Override the prediction type. Left on auto it is detected from the model'
  226. },
  227. rngType: {
  228. type: 'string', title: 'RNG', enum: ['', 'cuda', 'std_default', 'cpu'],
  229. enumLabels: ['server default', 'cuda', 'std_default', 'cpu'],
  230. default: '', description: 'Affects whether a seed reproduces across backends'
  231. },
  232. samplerRngType: {
  233. type: 'string', title: 'Sampler RNG', enum: ['', 'cuda', 'std_default', 'cpu'],
  234. enumLabels: ['same as RNG above', 'cuda', 'std_default', 'cpu'],
  235. default: '',
  236. description: 'Override the RNG for sampling only. The server does not report this one, so Fill in leaves it alone'
  237. },
  238. loraApplyMode: {
  239. type: 'string', title: 'LoRA Apply Mode', enum: ['', 'auto', 'immediately', 'at_runtime'],
  240. enumLabels: ['server default', 'auto', 'immediately', 'at_runtime'],
  241. default: '', description: 'When a LoRA named in a prompt is applied'
  242. },
  243. vaeConvDirect: { type: 'boolean', title: 'Direct VAE Convolution', description: 'Use the ggml_conv2d_direct path for the VAE' },
  244. diffusionConvDirect: { type: 'boolean', title: 'Direct Diffusion Convolution', description: 'Use the ggml_conv2d_direct path for the diffusion model' },
  245. forceSdxlVaeConvScale: { type: 'boolean', title: 'Force SDXL VAE Conv Scale', description: 'SDXL-specific VAE convolution scaling' },
  246. taePreviewOnly: { type: 'boolean', title: 'TAESD For Preview Only', description: 'Load TAESD purely to render progress previews, skipping the full VAE' },
  247. backend: { type: 'string', title: 'Backend', description: 'Per-component placement, such as te=cpu,vae=cpu,controlnet=cpu' },
  248. paramsBackend: { type: 'string', title: 'Parameter Backend', description: 'Global parameter placement, such as *=cpu to hold weights in system RAM' },
  249. rpcServers: { type: 'string', title: 'RPC Servers', description: 'Comma separated RPC backend endpoints' },
  250. modelArgs: { type: 'string', title: 'Model Args', description: 'Architecture-specific key=value knobs, comma separated' },
  251. tensorTypeRules: { type: 'string', title: 'Tensor Type Rules', description: 'Per-tensor weight overrides using regex, such as ^vae\\.=f16' },
  252. options: {
  253. type: 'object', title: 'Other Load Options',
  254. description: 'Extra load options passed through, such as flash_attn, enable_mmap, weight_type, stream_layers or max_vram'
  255. },
  256. whenDifferent: {
  257. type: 'string', title: 'When A Different Model Is Loaded',
  258. enum: ['load', 'fail'],
  259. default: 'load',
  260. description: 'load swaps it. fail stops the run instead - for a workflow that depends on a particular model already being in place and should not quietly spend minutes swapping it'
  261. },
  262. force: {
  263. type: 'boolean', title: 'Force Reload',
  264. default: false,
  265. description: 'Load again even when the right model is already loaded. Costs the full load time; useful after changing components or options, which this node cannot see from outside',
  266. showWhen: { field: 'whenDifferent', value: 'load' }
  267. },
  268. timeout: {
  269. type: 'number', title: 'Timeout (ms)',
  270. description: 'How long to wait on any one request to the server. Loading reads gigabytes from disk, so the request that starts it is given this long',
  271. default: 300000
  272. },
  273. loadWaitMs: {
  274. type: 'number', title: 'Wait For Loading (ms)',
  275. description: 'How long to keep watching after the load has started. This is separate from the timeout above because a request giving up says nothing about whether the server is still working - it usually is, and the node follows it through /health rather than reporting a failure that is really just impatience',
  276. default: 900000
  277. }
  278. },
  279. required: []
  280. };
  281. const inputSchema = { type: 'object', properties: { data: { type: 'any' } } };
  282. const outputSchema = {
  283. type: 'object',
  284. properties: {
  285. modelName: { type: 'string', description: 'The model that is loaded now' },
  286. modelType: { type: 'string' },
  287. architecture: { type: 'string', description: 'Architecture the server detected, which decides generation defaults' },
  288. loaded: { type: 'boolean', description: 'True when this node performed a load' },
  289. alreadyLoaded: { type: 'boolean', description: 'True when the right model, with the right settings, was already in place' },
  290. reloadedFor: { type: 'array', description: 'Settings that differed on an otherwise-correct model, when that is why it was reloaded' },
  291. previousModel: { type: 'string', description: 'What was loaded before, when this node swapped it' },
  292. loadedComponents: { type: 'object' },
  293. elapsedMs: { type: 'number' }
  294. }
  295. };
  296. function normalizeServer(url) {
  297. const value = String(url || '').trim();
  298. if (!value) {
  299. throw new Error('SD.cpp: a server URL is required, such as http://localhost:8077');
  300. }
  301. return value.replace(/\/+$/, '');
  302. }
  303. function readCredential(credentialId) {
  304. const auth = smartbotic.credentials.get(credentialId);
  305. if (!auth || auth.success !== true) {
  306. throw new Error('SD.cpp: could not read the credential: ' +
  307. ((auth && auth.error) || 'unknown error'));
  308. }
  309. const value = auth.headerValue || '';
  310. if (value.indexOf('Basic ') !== 0) {
  311. throw new Error('SD.cpp: the credential must be a basic one, holding the sdcpp-restapi ' +
  312. 'username and password');
  313. }
  314. const decoded = smartbotic.utils.base64Decode(value.substring(6));
  315. const separator = decoded.indexOf(':');
  316. if (separator < 1) {
  317. throw new Error('SD.cpp: the credential is malformed, expected a username and a password');
  318. }
  319. return {
  320. username: decoded.substring(0, separator),
  321. password: decoded.substring(separator + 1)
  322. };
  323. }
  324. function call(options) {
  325. const response = smartbotic.http.request(options);
  326. let body = response.data;
  327. if (typeof body === 'string' && body.length > 0) {
  328. try {
  329. body = JSON.parse(body);
  330. } catch (e) {
  331. const snippet = body.substring(0, 200).replace(/\s+/g, ' ');
  332. throw new Error('SD.cpp: ' + options.what + ' returned HTTP ' + response.status +
  333. ' with a body that is not JSON: ' + snippet);
  334. }
  335. }
  336. if (response.status < 200 || response.status >= 300) {
  337. const detail = (body && (body.message || body.error)) || ('HTTP ' + response.status);
  338. throw new Error('SD.cpp: ' + options.what + ' failed: ' + detail);
  339. }
  340. return body || {};
  341. }
  342. function login(server, credential, timeout) {
  343. const session = call({
  344. method: 'POST',
  345. url: server + '/auth/login',
  346. headers: { 'Content-Type': 'application/json' },
  347. body: JSON.stringify({
  348. username: credential.username,
  349. password: credential.password
  350. }),
  351. timeout: timeout,
  352. what: 'signing in'
  353. });
  354. if (!session.token) {
  355. throw new Error('SD.cpp: the server accepted the login but returned no token');
  356. }
  357. return session.token;
  358. }
  359. // /health is unauthenticated, and it is the only way to find out what is
  360. // already loaded without asking for a token first.
  361. function readHealth(server, timeout) {
  362. return call({
  363. method: 'GET',
  364. url: server + '/health',
  365. timeout: timeout,
  366. // Reading health is safe to repeat, and the moment it matters most is
  367. // the moment the server is busiest.
  368. retries: 2,
  369. retryDelayMs: 1000,
  370. what: 'reading server health'
  371. });
  372. }
  373. // The same, but a server that does not answer is treated as one that is busy
  374. // rather than one that has failed. A machine part-way through loading eleven
  375. // gigabytes answers /health slowly or not at all - that is what loading looks
  376. // like from outside, and throwing there ended the whole run for the one thing
  377. // the node was waiting for.
  378. function pollHealth(server, timeout) {
  379. try {
  380. return readHealth(server, timeout);
  381. } catch (e) {
  382. smartbotic.log.info('SD.cpp: no answer from /health while loading (' +
  383. ((e && e.message) || e) + '), still waiting');
  384. return null;
  385. }
  386. }
  387. function putIfSet(target, key, value) {
  388. if (value === undefined || value === null || value === '') {
  389. return;
  390. }
  391. target[key] = value;
  392. }
  393. const LOAD_OPTIONS = [
  394. { setting: 'flashAttn', server: 'flash_attn' },
  395. { setting: 'diffusionFlashAttn', server: 'diffusion_flash_attn' },
  396. { setting: 'enableMmap', server: 'enable_mmap' },
  397. { setting: 'eagerLoad', server: 'eager_load' },
  398. { setting: 'streamLayers', server: 'stream_layers' },
  399. { setting: 'maxVram', server: 'max_vram' },
  400. { setting: 'nThreads', server: 'n_threads' },
  401. { setting: 'weightType', server: 'weight_type' },
  402. { setting: 'vaeFormat', server: 'vae_format' },
  403. { setting: 'prediction', server: 'prediction' },
  404. { setting: 'rngType', server: 'rng_type' },
  405. { setting: 'samplerRngType', server: 'sampler_rng_type' },
  406. { setting: 'loraApplyMode', server: 'lora_apply_mode' },
  407. { setting: 'vaeConvDirect', server: 'vae_conv_direct' },
  408. { setting: 'diffusionConvDirect', server: 'diffusion_conv_direct' },
  409. { setting: 'taePreviewOnly', server: 'tae_preview_only' },
  410. { setting: 'forceSdxlVaeConvScale', server: 'force_sdxl_vae_conv_scale' },
  411. { setting: 'backend', server: 'backend' },
  412. { setting: 'paramsBackend', server: 'params_backend' },
  413. { setting: 'rpcServers', server: 'rpc_servers' },
  414. { setting: 'modelArgs', server: 'model_args' },
  415. { setting: 'tensorTypeRules', server: 'tensor_type_rules' },
  416. ];
  417. // The options this node asks for, as the server names them. Only settings that
  418. // were actually filled in are included: an untouched setting means "whatever
  419. // the server does", not "the default", so it is neither sent nor compared.
  420. function wantedOptions(config) {
  421. var wanted = {};
  422. for (var i = 0; i < LOAD_OPTIONS.length; i++) {
  423. var entry = LOAD_OPTIONS[i];
  424. var value = config[entry.setting];
  425. if (value === undefined || value === null || value === '') continue;
  426. if (typeof value === 'number' && !isFinite(value)) continue;
  427. wanted[entry.server] = value;
  428. }
  429. if (config.options && typeof config.options === 'object') {
  430. var keys = Object.keys(config.options);
  431. for (var k = 0; k < keys.length; k++) {
  432. var extra = config.options[keys[k]];
  433. if (extra !== undefined && extra !== null && extra !== '') {
  434. wanted[keys[k]] = extra;
  435. }
  436. }
  437. }
  438. return wanted;
  439. }
  440. // Which of them the server is not currently loaded with.
  441. //
  442. // This is what makes the node able to correct a server someone else changed:
  443. // the same model loaded with streaming off is not the same thing as the model
  444. // this workflow needs, and reloading it is the whole point of saying so here.
  445. function optionsThatDiffer(wanted, current) {
  446. var differing = [];
  447. var keys = Object.keys(wanted);
  448. for (var i = 0; i < keys.length; i++) {
  449. var key = keys[i];
  450. var have = current ? current[key] : undefined;
  451. var want = wanted[key];
  452. // Numbers arrive as 0 or 0.0 depending on the field, and a boolean may
  453. // come back as a string from a form, so compare on value rather than
  454. // on type.
  455. var same = (typeof want === 'number' || typeof have === 'number')
  456. ? Number(have) === Number(want)
  457. : String(have) === String(want);
  458. if (!same) {
  459. differing.push(key + ': server has ' + JSON.stringify(have) +
  460. ', this node wants ' + JSON.stringify(want));
  461. }
  462. }
  463. return differing;
  464. }
  465. async function execute(config, input, context) {
  466. const server = normalizeServer(config.serverUrl);
  467. const timeout = config.timeout > 0 ? config.timeout : 300000;
  468. // Watching costs nothing, so it is allowed to outlast any single request.
  469. const loadWait = config.loadWaitMs > 0 ? config.loadWaitMs : 900000;
  470. const modelName = String(config.modelName || '').trim();
  471. if (!modelName) {
  472. throw new Error('SD.cpp: a model name is required. Connect an SD.cpp Model node, ' +
  473. 'or type the file name');
  474. }
  475. // Ask what is loaded before loading anything. A load takes minutes and
  476. // unloads whatever was there, so doing it when the right model is already
  477. // resident is pure cost - and on a shared server it disrupts other work.
  478. const startedAt = Date.now();
  479. const health = readHealth(server, Math.min(timeout, 15000));
  480. const current = health.model_name || '';
  481. const sameModel = current === modelName;
  482. // The same model loaded with different settings is not the model this
  483. // workflow asked for. The server swaps models between queue items, so what
  484. // is loaded now may have been put there by something else entirely.
  485. const wanted = wantedOptions(config);
  486. const differing = optionsThatDiffer(wanted, health.load_options || {});
  487. if (sameModel && differing.length === 0 && config.force !== true) {
  488. smartbotic.log.info('SD.cpp: ' + modelName + ' is already loaded with the wanted settings');
  489. return {
  490. modelName: current,
  491. modelType: health.model_type || '',
  492. architecture: health.model_architecture || '',
  493. loaded: false,
  494. alreadyLoaded: true,
  495. reloadedFor: [],
  496. previousModel: '',
  497. loadedComponents: health.loaded_components || {},
  498. elapsedMs: Date.now() - startedAt
  499. };
  500. }
  501. if ((config.whenDifferent || 'load') === 'fail' && (!sameModel || differing.length > 0)) {
  502. if (!sameModel) {
  503. throw new Error('SD.cpp: this workflow expects "' + modelName + '" to be loaded, but ' +
  504. (current ? 'the server has "' + current + '"' : 'no model is loaded') +
  505. '. Set When A Different Model Is Loaded to "load" to swap it automatically');
  506. }
  507. throw new Error('SD.cpp: "' + modelName + '" is loaded, but not with the settings this ' +
  508. 'workflow needs - ' + differing.join('; ') +
  509. '. Set When A Different Model Is Loaded to "load" to reload it');
  510. }
  511. if (sameModel && differing.length > 0) {
  512. smartbotic.log.info('SD.cpp: reloading ' + modelName + ' because ' + differing.join('; '));
  513. }
  514. const credential = readCredential(config.credentialId);
  515. const token = login(server, credential, Math.min(timeout, 30000));
  516. const body = { model_name: modelName };
  517. putIfSet(body, 'model_type', config.modelType);
  518. putIfSet(body, 'vae', config.vae);
  519. putIfSet(body, 'clip_l', config.clipL);
  520. putIfSet(body, 'clip_g', config.clipG);
  521. putIfSet(body, 't5xxl', config.t5xxl);
  522. putIfSet(body, 'llm', config.llm);
  523. putIfSet(body, 'taesd', config.taesd);
  524. putIfSet(body, 'controlnet', config.controlnet);
  525. if (Object.keys(wanted).length > 0) {
  526. body.options = wanted;
  527. }
  528. smartbotic.log.info('SD.cpp: loading ' + modelName +
  529. (current ? ' (replacing ' + current + ')' : ''));
  530. // The slot has to be emptied first. The API documentation says a load
  531. // replaces whatever is there, but the server answers 409 "A model is
  532. // already loaded. Call POST /models/unload first" - so it is unloaded here
  533. // rather than leaving every reload to fail on a server that already has a
  534. // model. (A refused load is harmless: the resident model stays put.)
  535. //
  536. // This is also the only way to change the settings of a model that is
  537. // already loaded, which is the case this node exists to handle.
  538. //
  539. // It does mean everything between here and a finished load runs with the
  540. // server holding nothing. The server never unloads on its own, so an empty
  541. // slot afterwards is always something that happened in this window - which
  542. // is why the failure paths below say so rather than leaving the next run to
  543. // discover it.
  544. let emptiedTheSlot = false;
  545. if (health.model_loaded === true) {
  546. call({
  547. method: 'POST',
  548. url: server + '/models/unload',
  549. headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token },
  550. body: '{}',
  551. timeout: Math.min(timeout, 60000),
  552. what: 'unloading ' + (current || 'the current model') + ' before loading ' + modelName
  553. });
  554. smartbotic.log.info('SD.cpp: unloaded ' + (current || 'the previous model'));
  555. emptiedTheSlot = true;
  556. }
  557. // Loading unloads whatever was in the slot first, and the server holds a
  558. // mutex for the duration, so this blocks until the weights are resident.
  559. let loaded;
  560. try {
  561. loaded = call({
  562. method: 'POST',
  563. url: server + '/models/load',
  564. headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token },
  565. body: JSON.stringify(body),
  566. timeout: timeout,
  567. // Deliberately not retried: the server holds a mutex for the whole
  568. // load, so a second request would queue behind the first and load
  569. // the same weights twice.
  570. what: 'loading model ' + modelName
  571. });
  572. } catch (loadError) {
  573. // This call giving up does not mean the server did. If it is still
  574. // loading, that is the answer to what happened - so the wait below
  575. // finds out how it goes rather than reporting a failure that is really
  576. // just impatience.
  577. const probe = pollHealth(server, 15000);
  578. if (!probe || probe.model_loading !== true) {
  579. // The slot was emptied to make room and the load did not take, so
  580. // the server now holds nothing. One more attempt is worth it: there
  581. // is nothing left to lose, the usual cause is a moment of
  582. // slowness, and the alternative is leaving the server worse than it
  583. // was found.
  584. if (emptiedTheSlot && (!probe || probe.model_loaded !== true)) {
  585. smartbotic.log.warn('SD.cpp: the load failed and the server now has no model. ' +
  586. 'Trying once more before giving up');
  587. try {
  588. loaded = call({
  589. method: 'POST',
  590. url: server + '/models/load',
  591. headers: { 'Content-Type': 'application/json',
  592. 'Authorization': 'Bearer ' + token },
  593. body: JSON.stringify(body),
  594. timeout: timeout,
  595. what: 'loading model ' + modelName + ' (second attempt)'
  596. });
  597. // Falls through to the wait below, the same as a first
  598. // attempt that worked - the model still has to finish
  599. // loading either way.
  600. } catch (secondError) {
  601. throw new Error('SD.cpp: could not load ' + modelName + ', and the server ' +
  602. 'is now holding no model at all - it was unloaded to make room. ' +
  603. 'Nothing will generate until a load succeeds. First attempt: ' +
  604. ((loadError && loadError.message) || loadError) + '. Second: ' +
  605. ((secondError && secondError.message) || secondError));
  606. }
  607. }
  608. throw loadError;
  609. }
  610. smartbotic.log.info('SD.cpp: the load request stopped waiting, but the server is still ' +
  611. 'loading ' + (probe.loading_model_name || modelName) + ' - following it through /health');
  612. loaded = {};
  613. }
  614. // The load call comes back before the model is in memory. The API
  615. // documentation describes it as blocking, and it is not: /health reports
  616. // model_loading with a step count for some time afterwards. Returning here
  617. // would tell the workflow the model is ready and let the next node ask it
  618. // to generate, which fails with "no model loaded" - a confusing way to
  619. // learn that this node lied.
  620. const deadline = Date.now() + loadWait;
  621. let after = pollHealth(server, 15000);
  622. let lastStep = -1;
  623. let silentPolls = 0;
  624. while ((after === null || after.model_loading === true) && Date.now() < deadline) {
  625. if (after === null) {
  626. silentPolls++;
  627. } else {
  628. silentPolls = 0;
  629. const step = after.loading_step;
  630. const total = after.loading_total_steps;
  631. if (typeof step === 'number' && step !== lastStep) {
  632. lastStep = step;
  633. smartbotic.log.info('SD.cpp: loading ' + (after.loading_model_name || modelName) +
  634. ' - ' + step + (total ? '/' + total : ''));
  635. }
  636. }
  637. smartbotic.utils.sleep(2000);
  638. after = pollHealth(server, 15000);
  639. }
  640. const waitedSeconds = Math.round((Date.now() - startedAt) / 1000);
  641. if (after === null) {
  642. throw new Error('SD.cpp: ' + modelName + ' was asked for ' + waitedSeconds +
  643. 's ago and the server has stopped answering /health (' + silentPolls +
  644. ' polls in a row went unanswered). It may still be loading - check the ' +
  645. 'server, and raise the timeout on this node if this model is simply slow');
  646. }
  647. if (after.model_loading === true) {
  648. throw new Error('SD.cpp: ' + modelName + ' was still loading after ' + waitedSeconds +
  649. 's' + (typeof after.loading_step === 'number'
  650. ? ' (at step ' + after.loading_step +
  651. (after.loading_total_steps ? ' of ' + after.loading_total_steps : '') + ')'
  652. : '') +
  653. '. It may still finish on the server; raise the timeout on this node if this ' +
  654. 'model is simply slow to load');
  655. }
  656. if (after.model_loaded !== true) {
  657. throw new Error('SD.cpp: the server accepted the load but has no model loaded afterwards' +
  658. (after.last_error ? ': ' + after.last_error : ''));
  659. }
  660. return {
  661. modelName: loaded.model_name || modelName,
  662. modelType: loaded.model_type || config.modelType || '',
  663. architecture: after.model_architecture || '',
  664. loaded: true,
  665. alreadyLoaded: false,
  666. // Empty when the model itself changed; otherwise the settings that
  667. // forced a reload of a model that was already there.
  668. reloadedFor: sameModel ? differing : [],
  669. previousModel: sameModel ? '' : current,
  670. loadedComponents: loaded.loaded_components || after.loaded_components || {},
  671. elapsedMs: Date.now() - startedAt
  672. };
  673. }
  674. module.exports = { configSchema, inputSchema, outputSchema, execute };