sdcpp-model-load.js 36 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731732733734735736737738739740741742743744745746747748749750751752753754755756757758759760761
  1. /**
  2. * @node sdcpp-model-load
  3. * @name SD.cpp Load Model
  4. * @category sdcpp
  5. * @version 1.0.0
  6. * @description Make sure a model is loaded, without reloading one that already is
  7. * @icon box
  8. */
  9. // The credential this node wants, named so it can be found. It is stored as a
  10. // plain basic credential - that is what decides how it is encrypted - and this
  11. // only says which basic credential is the SD.cpp one. Anything that accepts a
  12. // basic credential still accepts this, and this node still accepts a plain
  13. // basic credential, because the shape is identical.
  14. const credentialTypes = [
  15. {
  16. id: 'sdcpp',
  17. label: 'SD.cpp Server',
  18. baseType: 'basic',
  19. description: 'The username and password you sign in to sdcpp-restapi with. The node exchanges them for a token before every call',
  20. usernameLabel: 'Username',
  21. passwordLabel: 'Password'
  22. }
  23. ];
  24. const configSchema = {
  25. type: 'object',
  26. uiGroups: [
  27. { title: 'Connection', fields: ['serverUrl', 'credentialId'] },
  28. { title: 'Model', fields: ['modelName', 'modelType', 'whenDifferent', 'force'] },
  29. { title: 'Components', fields: ['vae', 'clipL', 'clipG', 't5xxl', 'llm', 'taesd', 'controlnet', 'ipAdapter'] },
  30. { title: 'Loading', fields: ['flashAttn', 'diffusionFlashAttn', 'enableMmap', 'eagerLoad',
  31. 'disablePrefetch', 'disableSegmentedCompute', 'maxVram', 'nThreads', 'weightType'] },
  32. { title: 'Advanced', fields: ['vaeFormat', 'prediction', 'rngType', 'samplerRngType',
  33. 'loraApplyMode', 'vaeConvDirect', 'diffusionConvDirect',
  34. 'taePreviewOnly', 'forceSdxlVaeConvScale', 'backend', 'paramsBackend', 'rpcServers',
  35. 'modelArgs', 'tensorTypeRules', 'options', 'timeout', 'loadWaitMs'] }
  36. ],
  37. // Pressing this asks the server what it has loaded and writes it into the
  38. // settings below - the model and every component it reports - so a workflow
  39. // can be built from a server that is already set up the way it should be,
  40. // rather than by typing the same names again.
  41. prefill: {
  42. label: 'Take from the server',
  43. description: 'Fill these in from the model the server currently has loaded',
  44. node: 'sdcpp-health',
  45. needs: ['serverUrl'],
  46. map: {
  47. modelName: 'modelName',
  48. modelType: 'modelType',
  49. 'loadedComponents.vae': 'vae',
  50. 'loadedComponents.clip_l': 'clipL',
  51. 'loadedComponents.clip_g': 'clipG',
  52. 'loadedComponents.t5xxl': 't5xxl',
  53. 'loadedComponents.llm': 'llm',
  54. 'loadedComponents.taesd': 'taesd',
  55. 'loadedComponents.controlnet': 'controlnet',
  56. 'loadedComponents.ip_adapter': 'ipAdapter',
  57. 'loadOptions.flash_attn': 'flashAttn',
  58. 'loadOptions.diffusion_flash_attn': 'diffusionFlashAttn',
  59. 'loadOptions.enable_mmap': 'enableMmap',
  60. 'loadOptions.eager_load': 'eagerLoad',
  61. 'loadOptions.disable_prefetch': 'disablePrefetch',
  62. 'loadOptions.disable_segmented_compute': 'disableSegmentedCompute',
  63. 'loadOptions.max_vram': 'maxVram',
  64. 'loadOptions.n_threads': 'nThreads',
  65. 'loadOptions.vae_format': 'vaeFormat',
  66. 'loadOptions.rng_type': 'rngType',
  67. 'loadOptions.lora_apply_mode': 'loraApplyMode',
  68. 'loadOptions.vae_conv_direct': 'vaeConvDirect',
  69. 'loadOptions.diffusion_conv_direct': 'diffusionConvDirect',
  70. 'loadOptions.tae_preview_only': 'taePreviewOnly',
  71. 'loadOptions.force_sdxl_vae_conv_scale': 'forceSdxlVaeConvScale',
  72. 'loadOptions.rng_type': 'rngType',
  73. 'loadOptions.lora_apply_mode': 'loraApplyMode',
  74. 'loadOptions.backend': 'backend',
  75. 'loadOptions.rpc_servers': 'rpcServers',
  76. 'loadOptions.model_args': 'modelArgs',
  77. 'loadOptions.backend': 'backend',
  78. 'loadOptions.params_backend': 'paramsBackend',
  79. 'loadOptions.rpc_servers': 'rpcServers',
  80. 'loadOptions.model_args': 'modelArgs'
  81. }
  82. },
  83. properties: {
  84. serverUrl: {
  85. type: 'string', title: 'Server URL',
  86. description: 'Base address of the sdcpp-restapi server',
  87. default: 'http://localhost:8077'
  88. },
  89. credentialId: {
  90. type: 'string', title: 'Credential',
  91. description: 'A basic credential holding the sdcpp-restapi username and password',
  92. dynamicOptions: { source: 'credentials', filter: { type: ['sdcpp', 'basic'] } }
  93. },
  94. modelName: {
  95. type: 'string', title: 'Model',
  96. description: 'File name of the model, relative to its type directory. Browse lists what the server has of the Model Type chosen below',
  97. dynamicOptions: {
  98. source: 'node',
  99. node: 'sdcpp-model',
  100. config: { listOnly: true },
  101. itemsPath: 'models',
  102. valueKey: 'name',
  103. labelKey: 'name',
  104. needs: ['serverUrl', 'credentialId']
  105. }
  106. },
  107. modelType: {
  108. type: 'string', title: 'Model Type',
  109. enum: ['', 'checkpoint', 'diffusion'],
  110. default: '',
  111. description: 'checkpoint bundles U-Net, CLIP and VAE and suits SD1, SD2 and SDXL. diffusion holds only the U-Net or DiT and needs its components named separately, which is how Flux, SD3, Qwen, Wan and Z-Image load'
  112. },
  113. vae: {
  114. type: 'string', title: 'VAE',
  115. description: 'Component file name',
  116. dynamicOptions: {
  117. source: 'node',
  118. node: 'sdcpp-model',
  119. config: { listOnly: true, modelType: 'vae' },
  120. itemsPath: 'models',
  121. valueKey: 'name',
  122. labelKey: 'name',
  123. needs: ['serverUrl', 'credentialId']
  124. }
  125. },
  126. clipL: {
  127. type: 'string', title: 'CLIP-L',
  128. description: 'Component file name',
  129. dynamicOptions: {
  130. source: 'node',
  131. node: 'sdcpp-model',
  132. config: { listOnly: true, modelType: 'clip' },
  133. itemsPath: 'models',
  134. valueKey: 'name',
  135. labelKey: 'name',
  136. needs: ['serverUrl', 'credentialId']
  137. }
  138. },
  139. clipG: {
  140. type: 'string', title: 'CLIP-G',
  141. description: 'Component file name',
  142. dynamicOptions: {
  143. source: 'node',
  144. node: 'sdcpp-model',
  145. config: { listOnly: true, modelType: 'clip' },
  146. itemsPath: 'models',
  147. valueKey: 'name',
  148. labelKey: 'name',
  149. needs: ['serverUrl', 'credentialId']
  150. }
  151. },
  152. t5xxl: {
  153. type: 'string', title: 'T5-XXL',
  154. description: 'Component file name',
  155. dynamicOptions: {
  156. source: 'node',
  157. node: 'sdcpp-model',
  158. config: { listOnly: true, modelType: 't5' },
  159. itemsPath: 'models',
  160. valueKey: 'name',
  161. labelKey: 'name',
  162. needs: ['serverUrl', 'credentialId']
  163. }
  164. },
  165. llm: {
  166. type: 'string', title: 'LLM',
  167. description: 'Component file name, used by Z-Image, Qwen, Anima and Flux2',
  168. dynamicOptions: {
  169. source: 'node',
  170. node: 'sdcpp-model',
  171. config: { listOnly: true, modelType: 'llm' },
  172. itemsPath: 'models',
  173. valueKey: 'name',
  174. labelKey: 'name',
  175. needs: ['serverUrl', 'credentialId']
  176. }
  177. },
  178. taesd: {
  179. type: 'string', title: 'TAESD',
  180. description: 'Tiny autoencoder for progress previews',
  181. dynamicOptions: {
  182. source: 'node',
  183. node: 'sdcpp-model',
  184. config: { listOnly: true, modelType: 'taesd' },
  185. itemsPath: 'models',
  186. valueKey: 'name',
  187. labelKey: 'name',
  188. needs: ['serverUrl', 'credentialId']
  189. }
  190. },
  191. controlnet: {
  192. type: 'string', title: 'ControlNet',
  193. description: 'Component file name',
  194. dynamicOptions: {
  195. source: 'node',
  196. node: 'sdcpp-model',
  197. config: { listOnly: true, modelType: 'controlnet' },
  198. itemsPath: 'models',
  199. valueKey: 'name',
  200. labelKey: 'name',
  201. needs: ['serverUrl', 'credentialId']
  202. }
  203. },
  204. ipAdapter: {
  205. type: 'string', title: 'IP-Adapter',
  206. description: 'Component file name. An IP-Adapter lets a generation take its style and subject from a reference image - load it here, then set the reference on the generation node. Classic and Plus/Resampler variants both work',
  207. dynamicOptions: {
  208. source: 'node',
  209. node: 'sdcpp-model',
  210. config: { listOnly: true, modelType: 'ip_adapter' },
  211. itemsPath: 'models',
  212. valueKey: 'name',
  213. labelKey: 'name',
  214. needs: ['serverUrl', 'credentialId']
  215. }
  216. },
  217. flashAttn: { type: 'boolean', title: 'Flash Attention', description: 'For CLIP and T5. A large speed and memory win on modern GPUs' },
  218. diffusionFlashAttn: { type: 'boolean', title: 'Flash Attention (diffusion)', description: 'Flash attention for the diffusion model specifically' },
  219. enableMmap: { type: 'boolean', title: 'Memory-map Weights', description: 'Recommended for large files' },
  220. eagerLoad: { type: 'boolean', title: 'Eager Load', description: 'Move every parameter to the compute backend at load time instead of on demand' },
  221. disablePrefetch: { type: 'boolean', title: 'Disable Prefetch', description: 'Turn off prefetching of the next layer\'s weights. On by default upstream - only switch this off to diagnose a problem, it costs speed' },
  222. disableSegmentedCompute: { type: 'boolean', title: 'Disable Segmented Compute', description: 'Turn off running the diffusion graph in segments. On by default upstream, and what lets a model larger than VRAM run at all - switching it off will OOM on a big model' },
  223. maxVram: {
  224. type: 'number',
  225. title: 'VRAM Budget (GiB)',
  226. default: 0,
  227. description: '0 (the default) lets sd.cpp re-check free VRAM continuously and use what is actually there - the safest setting, and the right one unless you have a specific reason. A positive N caps managed weights and runner buffers at N GiB regardless of what is free, which is what you want when sharing the card with something else and you need a hard ceiling. The old -1 is no longer accepted; the server coerces any negative value to 0, which matches what -1 was asking for'
  228. },
  229. nThreads: { type: 'number', title: 'CPU Threads', description: '-1 lets the server decide' },
  230. weightType: {
  231. type: 'string', title: 'Weight Type',
  232. enum: ['', 'f32', 'f16', 'bf16', 'q8_0', 'q5_0', 'q5_1', 'q4_0', 'q4_1', 'q4_k', 'q5_k', 'q6_k', 'q8_k', 'q3_k', 'q2_k', 'mxfp4', 'nvfp4', 'q1_0'],
  233. enumLabels: ['auto - use the weights in the file', 'f32', 'f16', 'bf16', 'q8_0', 'q5_0', 'q5_1', 'q4_0', 'q4_1', 'q4_k', 'q5_k', 'q6_k', 'q8_k', 'q3_k', 'q2_k', 'mxfp4', 'nvfp4', 'q1_0'],
  234. default: '', description: 'Force a quantisation. Left on auto, sd.cpp uses whatever the model file holds'
  235. },
  236. vaeFormat: {
  237. type: 'string', title: 'VAE Format',
  238. enum: ['', 'auto', 'flux', 'sd3', 'flux2', 'wan'],
  239. enumLabels: ['not set - leave it to the server', 'auto - detect from the file', 'flux', 'sd3', 'flux2', 'wan'],
  240. default: '', description: 'Override VAE format detection'
  241. },
  242. prediction: {
  243. type: 'string', title: 'Prediction Type',
  244. enum: ['', 'eps', 'v', 'edm_v', 'sd3_flow', 'flux_flow', 'flux2_flow', 'sefi_flow', 'minit2i_flow'],
  245. enumLabels: ['auto - detect from the model', 'eps', 'v', 'edm_v', 'sd3_flow', 'flux_flow', 'flux2_flow', 'sefi_flow', 'minit2i_flow'],
  246. default: '', description: 'Override the prediction type. Left on auto it is detected from the model'
  247. },
  248. rngType: {
  249. type: 'string', title: 'RNG', enum: ['', 'cuda', 'std_default', 'cpu'],
  250. enumLabels: ['server default', 'cuda', 'std_default', 'cpu'],
  251. default: '', description: 'Affects whether a seed reproduces across backends'
  252. },
  253. samplerRngType: {
  254. type: 'string', title: 'Sampler RNG', enum: ['', 'cuda', 'std_default', 'cpu'],
  255. enumLabels: ['same as RNG above', 'cuda', 'std_default', 'cpu'],
  256. default: '',
  257. description: 'Override the RNG for sampling only. The server does not report this one, so Fill in leaves it alone'
  258. },
  259. loraApplyMode: {
  260. type: 'string', title: 'LoRA Apply Mode', enum: ['', 'auto', 'immediately', 'at_runtime'],
  261. enumLabels: ['server default', 'auto', 'immediately', 'at_runtime'],
  262. default: '', description: 'When a LoRA named in a prompt is applied'
  263. },
  264. vaeConvDirect: { type: 'boolean', title: 'Direct VAE Convolution', description: 'Use the ggml_conv2d_direct path for the VAE' },
  265. diffusionConvDirect: { type: 'boolean', title: 'Direct Diffusion Convolution', description: 'Use the ggml_conv2d_direct path for the diffusion model' },
  266. forceSdxlVaeConvScale: { type: 'boolean', title: 'Force SDXL VAE Conv Scale', description: 'SDXL-specific VAE convolution scaling' },
  267. taePreviewOnly: { type: 'boolean', title: 'TAESD For Preview Only', description: 'Load TAESD purely to render progress previews, skipping the full VAE' },
  268. backend: { type: 'string', title: 'Backend', description: 'Per-component placement, such as te=cpu,vae=cpu,controlnet=cpu' },
  269. paramsBackend: { type: 'string', title: 'Parameter Backend', description: 'Global parameter placement, such as *=cpu to hold weights in system RAM' },
  270. rpcServers: { type: 'string', title: 'RPC Servers', description: 'Comma separated RPC backend endpoints' },
  271. modelArgs: { type: 'string', title: 'Model Args', description: 'Architecture-specific key=value knobs, comma separated' },
  272. tensorTypeRules: { type: 'string', title: 'Tensor Type Rules', description: 'Per-tensor weight overrides using regex, such as ^vae\\.=f16' },
  273. options: {
  274. type: 'object', title: 'Other Load Options',
  275. description: 'Extra load options passed through, such as flash_attn, enable_mmap, weight_type, disable_prefetch or max_vram'
  276. },
  277. whenDifferent: {
  278. type: 'string', title: 'When A Different Model Is Loaded',
  279. enum: ['load', 'fail'],
  280. default: 'load',
  281. description: 'load swaps it. fail stops the run instead - for a workflow that depends on a particular model already being in place and should not quietly spend minutes swapping it'
  282. },
  283. force: {
  284. type: 'boolean', title: 'Force Reload',
  285. default: false,
  286. description: 'Load again even when the right model is already loaded. Costs the full load time; useful after changing components or options, which this node cannot see from outside',
  287. showWhen: { field: 'whenDifferent', value: 'load' }
  288. },
  289. timeout: {
  290. type: 'number', title: 'Timeout (ms)',
  291. description: 'How long to wait on any one request to the server. Loading reads gigabytes from disk, so the request that starts it is given this long',
  292. default: 300000
  293. },
  294. loadWaitMs: {
  295. type: 'number', title: 'Wait For Loading (ms)',
  296. description: 'How long to keep watching after the load has started. This is separate from the timeout above because a request giving up says nothing about whether the server is still working - it usually is, and the node follows it through /health rather than reporting a failure that is really just impatience',
  297. default: 900000
  298. }
  299. },
  300. required: []
  301. };
  302. const inputSchema = { type: 'object', properties: { data: { type: 'any' } } };
  303. const outputSchema = {
  304. type: 'object',
  305. properties: {
  306. modelName: { type: 'string', description: 'The model that is loaded now' },
  307. modelType: { type: 'string' },
  308. architecture: { type: 'string', description: 'Architecture the server detected, which decides generation defaults' },
  309. loaded: { type: 'boolean', description: 'True when this node performed a load' },
  310. alreadyLoaded: { type: 'boolean', description: 'True when the right model, with the right settings, was already in place' },
  311. reloadedFor: { type: 'array', description: 'Settings that differed on an otherwise-correct model, when that is why it was reloaded' },
  312. previousModel: { type: 'string', description: 'What was loaded before, when this node swapped it' },
  313. loadedComponents: { type: 'object' },
  314. elapsedMs: { type: 'number' }
  315. }
  316. };
  317. function normalizeServer(url) {
  318. const value = String(url || '').trim();
  319. if (!value) {
  320. throw new Error('SD.cpp: a server URL is required, such as http://localhost:8077');
  321. }
  322. return value.replace(/\/+$/, '');
  323. }
  324. function readCredential(credentialId) {
  325. const auth = smartbotic.credentials.get(credentialId);
  326. if (!auth || auth.success !== true) {
  327. throw new Error('SD.cpp: could not read the credential: ' +
  328. ((auth && auth.error) || 'unknown error'));
  329. }
  330. const value = auth.headerValue || '';
  331. if (value.indexOf('Basic ') !== 0) {
  332. throw new Error('SD.cpp: the credential must be a basic one, holding the sdcpp-restapi ' +
  333. 'username and password');
  334. }
  335. const decoded = smartbotic.utils.base64Decode(value.substring(6));
  336. const separator = decoded.indexOf(':');
  337. if (separator < 1) {
  338. throw new Error('SD.cpp: the credential is malformed, expected a username and a password');
  339. }
  340. return {
  341. username: decoded.substring(0, separator),
  342. password: decoded.substring(separator + 1)
  343. };
  344. }
  345. function call(options) {
  346. const response = smartbotic.http.request(options);
  347. let body = response.data;
  348. if (typeof body === 'string' && body.length > 0) {
  349. try {
  350. body = JSON.parse(body);
  351. } catch (e) {
  352. const snippet = body.substring(0, 200).replace(/\s+/g, ' ');
  353. throw new Error('SD.cpp: ' + options.what + ' returned HTTP ' + response.status +
  354. ' with a body that is not JSON: ' + snippet);
  355. }
  356. }
  357. if (response.status < 200 || response.status >= 300) {
  358. const detail = (body && (body.message || body.error)) || ('HTTP ' + response.status);
  359. throw new Error('SD.cpp: ' + options.what + ' failed: ' + detail);
  360. }
  361. return body || {};
  362. }
  363. function login(server, credential, timeout) {
  364. const session = call({
  365. method: 'POST',
  366. url: server + '/auth/login',
  367. headers: { 'Content-Type': 'application/json' },
  368. body: JSON.stringify({
  369. username: credential.username,
  370. password: credential.password
  371. }),
  372. timeout: timeout,
  373. what: 'signing in'
  374. });
  375. if (!session.token) {
  376. throw new Error('SD.cpp: the server accepted the login but returned no token');
  377. }
  378. return session.token;
  379. }
  380. // /health is unauthenticated, and it is the only way to find out what is
  381. // already loaded without asking for a token first.
  382. function readHealth(server, timeout) {
  383. return call({
  384. method: 'GET',
  385. url: server + '/health',
  386. timeout: timeout,
  387. // Reading health is safe to repeat, and the moment it matters most is
  388. // the moment the server is busiest.
  389. retries: 2,
  390. retryDelayMs: 1000,
  391. what: 'reading server health'
  392. });
  393. }
  394. // The same, but a server that does not answer is treated as one that is busy
  395. // rather than one that has failed. A machine part-way through loading eleven
  396. // gigabytes answers /health slowly or not at all - that is what loading looks
  397. // like from outside, and throwing there ended the whole run for the one thing
  398. // the node was waiting for.
  399. function pollHealth(server, timeout) {
  400. try {
  401. return readHealth(server, timeout);
  402. } catch (e) {
  403. smartbotic.log.info('SD.cpp: no answer from /health while loading (' +
  404. ((e && e.message) || e) + '), still waiting');
  405. return null;
  406. }
  407. }
  408. function putIfSet(target, key, value) {
  409. if (value === undefined || value === null || value === '') {
  410. return;
  411. }
  412. target[key] = value;
  413. }
  414. const LOAD_OPTIONS = [
  415. { setting: 'flashAttn', server: 'flash_attn' },
  416. { setting: 'diffusionFlashAttn', server: 'diffusion_flash_attn' },
  417. { setting: 'enableMmap', server: 'enable_mmap' },
  418. { setting: 'eagerLoad', server: 'eager_load' },
  419. // Layer streaming and segmented compute are how sd.cpp works now - always
  420. // on, with nothing to enable. The old stream_layers toggle is gone
  421. // entirely, and a load still carrying it is rejected outright: "Unknown
  422. // field(s) in /models/load options". These two only turn the behaviour OFF.
  423. { setting: 'disablePrefetch', server: 'disable_prefetch' },
  424. { setting: 'disableSegmentedCompute', server: 'disable_segmented_compute' },
  425. { setting: 'maxVram', server: 'max_vram' },
  426. { setting: 'nThreads', server: 'n_threads' },
  427. { setting: 'weightType', server: 'weight_type' },
  428. { setting: 'vaeFormat', server: 'vae_format' },
  429. { setting: 'prediction', server: 'prediction' },
  430. { setting: 'rngType', server: 'rng_type' },
  431. { setting: 'samplerRngType', server: 'sampler_rng_type' },
  432. { setting: 'loraApplyMode', server: 'lora_apply_mode' },
  433. { setting: 'vaeConvDirect', server: 'vae_conv_direct' },
  434. { setting: 'diffusionConvDirect', server: 'diffusion_conv_direct' },
  435. { setting: 'taePreviewOnly', server: 'tae_preview_only' },
  436. // Absent from /openapi.json but genuinely accepted - it is in the server's
  437. // own allow-list in model_manager.cpp and read into ctx_params. The schema
  438. // is incomplete here, so a field being missing from it is not evidence
  439. // that the server rejects it.
  440. { setting: 'forceSdxlVaeConvScale', server: 'force_sdxl_vae_conv_scale' },
  441. { setting: 'backend', server: 'backend' },
  442. { setting: 'paramsBackend', server: 'params_backend' },
  443. { setting: 'rpcServers', server: 'rpc_servers' },
  444. { setting: 'modelArgs', server: 'model_args' },
  445. { setting: 'tensorTypeRules', server: 'tensor_type_rules' },
  446. ];
  447. // The options this node asks for, as the server names them. Only settings that
  448. // were actually filled in are included: an untouched setting means "whatever
  449. // the server does", not "the default", so it is neither sent nor compared.
  450. function wantedOptions(config) {
  451. var wanted = {};
  452. for (var i = 0; i < LOAD_OPTIONS.length; i++) {
  453. var entry = LOAD_OPTIONS[i];
  454. var value = config[entry.setting];
  455. if (value === undefined || value === null || value === '') continue;
  456. if (typeof value === 'number' && !isFinite(value)) continue;
  457. // max_vram used to take -1 for "auto". The server's own parser
  458. // (ModelLoadParams::from_json) already coerces any negative to 0, which
  459. // is the closest match to that intent - 0 means sd.cpp re-checks free
  460. // VRAM continuously. Normalised here too so the value the node reports
  461. // sending is the value that takes effect, rather than the caller seeing
  462. // -1 in the request and 0 in the behaviour.
  463. if (entry.server === 'max_vram' && value < 0) {
  464. smartbotic.log.info('SD.cpp: max_vram ' + value +
  465. ' is the old "auto" value; sending 0, which is what the server ' +
  466. 'would coerce it to and means "use whatever VRAM is free".');
  467. value = 0;
  468. }
  469. wanted[entry.server] = value;
  470. }
  471. if (config.options && typeof config.options === 'object') {
  472. var keys = Object.keys(config.options);
  473. for (var k = 0; k < keys.length; k++) {
  474. var extra = config.options[keys[k]];
  475. if (extra !== undefined && extra !== null && extra !== '') {
  476. wanted[keys[k]] = extra;
  477. }
  478. }
  479. }
  480. return wanted;
  481. }
  482. // Which of them the server is not currently loaded with.
  483. //
  484. // This is what makes the node able to correct a server someone else changed:
  485. // the same model loaded with streaming off is not the same thing as the model
  486. // this workflow needs, and reloading it is the whole point of saying so here.
  487. function optionsThatDiffer(wanted, current) {
  488. var differing = [];
  489. var keys = Object.keys(wanted);
  490. for (var i = 0; i < keys.length; i++) {
  491. var key = keys[i];
  492. var have = current ? current[key] : undefined;
  493. var want = wanted[key];
  494. // Numbers arrive as 0 or 0.0 depending on the field, and a boolean may
  495. // come back as a string from a form, so compare on value rather than
  496. // on type.
  497. var same = (typeof want === 'number' || typeof have === 'number')
  498. ? Number(have) === Number(want)
  499. : String(have) === String(want);
  500. if (!same) {
  501. differing.push(key + ': server has ' + JSON.stringify(have) +
  502. ', this node wants ' + JSON.stringify(want));
  503. }
  504. }
  505. return differing;
  506. }
  507. async function execute(config, input, context) {
  508. const server = normalizeServer(config.serverUrl);
  509. const timeout = config.timeout > 0 ? config.timeout : 300000;
  510. // Watching costs nothing, so it is allowed to outlast any single request.
  511. const loadWait = config.loadWaitMs > 0 ? config.loadWaitMs : 900000;
  512. const modelName = String(config.modelName || '').trim();
  513. if (!modelName) {
  514. throw new Error('SD.cpp: a model name is required. Connect an SD.cpp Model node, ' +
  515. 'or type the file name');
  516. }
  517. // Ask what is loaded before loading anything. A load takes minutes and
  518. // unloads whatever was there, so doing it when the right model is already
  519. // resident is pure cost - and on a shared server it disrupts other work.
  520. const startedAt = Date.now();
  521. const health = readHealth(server, Math.min(timeout, 15000));
  522. const current = health.model_name || '';
  523. const sameModel = current === modelName;
  524. // The same model loaded with different settings is not the model this
  525. // workflow asked for. The server swaps models between queue items, so what
  526. // is loaded now may have been put there by something else entirely.
  527. const wanted = wantedOptions(config);
  528. const differing = optionsThatDiffer(wanted, health.load_options || {});
  529. if (sameModel && differing.length === 0 && config.force !== true) {
  530. smartbotic.log.info('SD.cpp: ' + modelName + ' is already loaded with the wanted settings');
  531. return {
  532. modelName: current,
  533. modelType: health.model_type || '',
  534. architecture: health.model_architecture || '',
  535. loaded: false,
  536. alreadyLoaded: true,
  537. reloadedFor: [],
  538. previousModel: '',
  539. loadedComponents: health.loaded_components || {},
  540. elapsedMs: Date.now() - startedAt
  541. };
  542. }
  543. if ((config.whenDifferent || 'load') === 'fail' && (!sameModel || differing.length > 0)) {
  544. if (!sameModel) {
  545. throw new Error('SD.cpp: this workflow expects "' + modelName + '" to be loaded, but ' +
  546. (current ? 'the server has "' + current + '"' : 'no model is loaded') +
  547. '. Set When A Different Model Is Loaded to "load" to swap it automatically');
  548. }
  549. throw new Error('SD.cpp: "' + modelName + '" is loaded, but not with the settings this ' +
  550. 'workflow needs - ' + differing.join('; ') +
  551. '. Set When A Different Model Is Loaded to "load" to reload it');
  552. }
  553. if (sameModel && differing.length > 0) {
  554. smartbotic.log.info('SD.cpp: reloading ' + modelName + ' because ' + differing.join('; '));
  555. }
  556. const credential = readCredential(config.credentialId);
  557. const token = login(server, credential, Math.min(timeout, 30000));
  558. const body = { model_name: modelName };
  559. putIfSet(body, 'model_type', config.modelType);
  560. putIfSet(body, 'vae', config.vae);
  561. putIfSet(body, 'clip_l', config.clipL);
  562. putIfSet(body, 'clip_g', config.clipG);
  563. putIfSet(body, 't5xxl', config.t5xxl);
  564. putIfSet(body, 'llm', config.llm);
  565. putIfSet(body, 'taesd', config.taesd);
  566. putIfSet(body, 'controlnet', config.controlnet);
  567. putIfSet(body, 'ip_adapter', config.ipAdapter);
  568. if (Object.keys(wanted).length > 0) {
  569. body.options = wanted;
  570. }
  571. smartbotic.log.info('SD.cpp: loading ' + modelName +
  572. (current ? ' (replacing ' + current + ')' : ''));
  573. // The slot has to be emptied first. The API documentation says a load
  574. // replaces whatever is there, but the server answers 409 "A model is
  575. // already loaded. Call POST /models/unload first" - so it is unloaded here
  576. // rather than leaving every reload to fail on a server that already has a
  577. // model. (A refused load is harmless: the resident model stays put.)
  578. //
  579. // This is also the only way to change the settings of a model that is
  580. // already loaded, which is the case this node exists to handle.
  581. //
  582. // It does mean everything between here and a finished load runs with the
  583. // server holding nothing. The server never unloads on its own, so an empty
  584. // slot afterwards is always something that happened in this window - which
  585. // is why the failure paths below say so rather than leaving the next run to
  586. // discover it.
  587. let emptiedTheSlot = false;
  588. if (health.model_loaded === true) {
  589. call({
  590. method: 'POST',
  591. url: server + '/models/unload',
  592. headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token },
  593. body: '{}',
  594. timeout: Math.min(timeout, 60000),
  595. what: 'unloading ' + (current || 'the current model') + ' before loading ' + modelName
  596. });
  597. smartbotic.log.info('SD.cpp: unloaded ' + (current || 'the previous model'));
  598. emptiedTheSlot = true;
  599. }
  600. // Loading unloads whatever was in the slot first, and the server holds a
  601. // mutex for the duration, so this blocks until the weights are resident.
  602. let loaded;
  603. try {
  604. loaded = call({
  605. method: 'POST',
  606. url: server + '/models/load',
  607. headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token },
  608. body: JSON.stringify(body),
  609. timeout: timeout,
  610. // Deliberately not retried: the server holds a mutex for the whole
  611. // load, so a second request would queue behind the first and load
  612. // the same weights twice.
  613. what: 'loading model ' + modelName
  614. });
  615. } catch (loadError) {
  616. // This call giving up does not mean the server did. If it is still
  617. // loading, that is the answer to what happened - so the wait below
  618. // finds out how it goes rather than reporting a failure that is really
  619. // just impatience.
  620. const probe = pollHealth(server, 15000);
  621. if (!probe || probe.model_loading !== true) {
  622. // The slot was emptied to make room and the load did not take, so
  623. // the server now holds nothing. One more attempt is worth it: there
  624. // is nothing left to lose, the usual cause is a moment of
  625. // slowness, and the alternative is leaving the server worse than it
  626. // was found.
  627. if (emptiedTheSlot && (!probe || probe.model_loaded !== true)) {
  628. smartbotic.log.warn('SD.cpp: the load failed and the server now has no model. ' +
  629. 'Trying once more before giving up');
  630. try {
  631. loaded = call({
  632. method: 'POST',
  633. url: server + '/models/load',
  634. headers: { 'Content-Type': 'application/json',
  635. 'Authorization': 'Bearer ' + token },
  636. body: JSON.stringify(body),
  637. timeout: timeout,
  638. what: 'loading model ' + modelName + ' (second attempt)'
  639. });
  640. // Falls through to the wait below, the same as a first
  641. // attempt that worked - the model still has to finish
  642. // loading either way.
  643. } catch (secondError) {
  644. throw new Error('SD.cpp: could not load ' + modelName + ', and the server ' +
  645. 'is now holding no model at all - it was unloaded to make room. ' +
  646. 'Nothing will generate until a load succeeds. First attempt: ' +
  647. ((loadError && loadError.message) || loadError) + '. Second: ' +
  648. ((secondError && secondError.message) || secondError));
  649. }
  650. }
  651. throw loadError;
  652. }
  653. smartbotic.log.info('SD.cpp: the load request stopped waiting, but the server is still ' +
  654. 'loading ' + (probe.loading_model_name || modelName) + ' - following it through /health');
  655. loaded = {};
  656. }
  657. // The load call comes back before the model is in memory. The API
  658. // documentation describes it as blocking, and it is not: /health reports
  659. // model_loading with a step count for some time afterwards. Returning here
  660. // would tell the workflow the model is ready and let the next node ask it
  661. // to generate, which fails with "no model loaded" - a confusing way to
  662. // learn that this node lied.
  663. const deadline = Date.now() + loadWait;
  664. let after = pollHealth(server, 15000);
  665. let lastStep = -1;
  666. let silentPolls = 0;
  667. while ((after === null || after.model_loading === true) && Date.now() < deadline) {
  668. if (after === null) {
  669. silentPolls++;
  670. } else {
  671. silentPolls = 0;
  672. const step = after.loading_step;
  673. const total = after.loading_total_steps;
  674. if (typeof step === 'number' && step !== lastStep) {
  675. lastStep = step;
  676. smartbotic.log.info('SD.cpp: loading ' + (after.loading_model_name || modelName) +
  677. ' - ' + step + (total ? '/' + total : ''));
  678. }
  679. }
  680. smartbotic.utils.sleep(2000);
  681. after = pollHealth(server, 15000);
  682. }
  683. const waitedSeconds = Math.round((Date.now() - startedAt) / 1000);
  684. if (after === null) {
  685. throw new Error('SD.cpp: ' + modelName + ' was asked for ' + waitedSeconds +
  686. 's ago and the server has stopped answering /health (' + silentPolls +
  687. ' polls in a row went unanswered). It may still be loading - check the ' +
  688. 'server, and raise the timeout on this node if this model is simply slow');
  689. }
  690. if (after.model_loading === true) {
  691. throw new Error('SD.cpp: ' + modelName + ' was still loading after ' + waitedSeconds +
  692. 's' + (typeof after.loading_step === 'number'
  693. ? ' (at step ' + after.loading_step +
  694. (after.loading_total_steps ? ' of ' + after.loading_total_steps : '') + ')'
  695. : '') +
  696. '. It may still finish on the server; raise the timeout on this node if this ' +
  697. 'model is simply slow to load');
  698. }
  699. if (after.model_loaded !== true) {
  700. throw new Error('SD.cpp: the server accepted the load but has no model loaded afterwards' +
  701. (after.last_error ? ': ' + after.last_error : ''));
  702. }
  703. return {
  704. modelName: loaded.model_name || modelName,
  705. modelType: loaded.model_type || config.modelType || '',
  706. architecture: after.model_architecture || '',
  707. loaded: true,
  708. alreadyLoaded: false,
  709. // Empty when the model itself changed; otherwise the settings that
  710. // forced a reload of a model that was already there.
  711. reloadedFor: sameModel ? differing : [],
  712. previousModel: sameModel ? '' : current,
  713. loadedComponents: loaded.loaded_components || after.loaded_components || {},
  714. elapsedMs: Date.now() - startedAt
  715. };
  716. }
  717. module.exports = { configSchema, inputSchema, outputSchema, execute };