sdcpp-model-load.js 34 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550551552553554555556557558559560561562563564565566567568569570571572573574575576577578579580581582583584585586587588589590591592593594595596597598599600601602603604605606607608609610611612613614615616617618619620621622623624625626627628629630631632633634635636637638639640641642643644645646647648649650651652653654655656657658659660661662663664665666667668669670671672673674675676677678679680681682683684685686687688689690691692693694695696697698699700701702703704705706707708709710711712713714715716717718719720721722723724725726727728729730731
  1. /**
  2. * @node sdcpp-model-load
  3. * @name SD.cpp Load Model
  4. * @category sdcpp
  5. * @version 1.0.0
  6. * @description Make sure a model is loaded, without reloading one that already is
  7. * @icon box
  8. */
  9. // The credential this node wants, named so it can be found. It is stored as a
  10. // plain basic credential - that is what decides how it is encrypted - and this
  11. // only says which basic credential is the SD.cpp one. Anything that accepts a
  12. // basic credential still accepts this, and this node still accepts a plain
  13. // basic credential, because the shape is identical.
  14. const credentialTypes = [
  15. {
  16. id: 'sdcpp',
  17. label: 'SD.cpp Server',
  18. baseType: 'basic',
  19. description: 'The username and password you sign in to sdcpp-restapi with. The node exchanges them for a token before every call',
  20. usernameLabel: 'Username',
  21. passwordLabel: 'Password'
  22. }
  23. ];
  24. const configSchema = {
  25. type: 'object',
  26. uiGroups: [
  27. { title: 'Connection', fields: ['serverUrl', 'credentialId'] },
  28. { title: 'Model', fields: ['modelName', 'modelType', 'whenDifferent', 'force'] },
  29. { title: 'Components', fields: ['vae', 'clipL', 'clipG', 't5xxl', 'llm', 'taesd', 'controlnet', 'ipAdapter'] },
  30. { title: 'Loading', fields: ['flashAttn', 'diffusionFlashAttn', 'enableMmap', 'eagerLoad',
  31. 'streamLayers', 'maxVram', 'nThreads', 'weightType'] },
  32. { title: 'Advanced', fields: ['vaeFormat', 'prediction', 'rngType', 'samplerRngType',
  33. 'loraApplyMode', 'vaeConvDirect', 'diffusionConvDirect',
  34. 'taePreviewOnly', 'forceSdxlVaeConvScale', 'backend', 'paramsBackend', 'rpcServers',
  35. 'modelArgs', 'tensorTypeRules', 'options', 'timeout', 'loadWaitMs'] }
  36. ],
  37. // Pressing this asks the server what it has loaded and writes it into the
  38. // settings below - the model and every component it reports - so a workflow
  39. // can be built from a server that is already set up the way it should be,
  40. // rather than by typing the same names again.
  41. prefill: {
  42. label: 'Take from the server',
  43. description: 'Fill these in from the model the server currently has loaded',
  44. node: 'sdcpp-health',
  45. needs: ['serverUrl'],
  46. map: {
  47. modelName: 'modelName',
  48. modelType: 'modelType',
  49. 'loadedComponents.vae': 'vae',
  50. 'loadedComponents.clip_l': 'clipL',
  51. 'loadedComponents.clip_g': 'clipG',
  52. 'loadedComponents.t5xxl': 't5xxl',
  53. 'loadedComponents.llm': 'llm',
  54. 'loadedComponents.taesd': 'taesd',
  55. 'loadedComponents.controlnet': 'controlnet',
  56. 'loadedComponents.ip_adapter': 'ipAdapter',
  57. 'loadOptions.flash_attn': 'flashAttn',
  58. 'loadOptions.diffusion_flash_attn': 'diffusionFlashAttn',
  59. 'loadOptions.enable_mmap': 'enableMmap',
  60. 'loadOptions.eager_load': 'eagerLoad',
  61. 'loadOptions.stream_layers': 'streamLayers',
  62. 'loadOptions.max_vram': 'maxVram',
  63. 'loadOptions.n_threads': 'nThreads',
  64. 'loadOptions.vae_format': 'vaeFormat',
  65. 'loadOptions.rng_type': 'rngType',
  66. 'loadOptions.lora_apply_mode': 'loraApplyMode',
  67. 'loadOptions.vae_conv_direct': 'vaeConvDirect',
  68. 'loadOptions.diffusion_conv_direct': 'diffusionConvDirect',
  69. 'loadOptions.tae_preview_only': 'taePreviewOnly',
  70. 'loadOptions.force_sdxl_vae_conv_scale': 'forceSdxlVaeConvScale',
  71. 'loadOptions.rng_type': 'rngType',
  72. 'loadOptions.lora_apply_mode': 'loraApplyMode',
  73. 'loadOptions.backend': 'backend',
  74. 'loadOptions.rpc_servers': 'rpcServers',
  75. 'loadOptions.model_args': 'modelArgs',
  76. 'loadOptions.backend': 'backend',
  77. 'loadOptions.params_backend': 'paramsBackend',
  78. 'loadOptions.rpc_servers': 'rpcServers',
  79. 'loadOptions.model_args': 'modelArgs'
  80. }
  81. },
  82. properties: {
  83. serverUrl: {
  84. type: 'string', title: 'Server URL',
  85. description: 'Base address of the sdcpp-restapi server',
  86. default: 'http://localhost:8077'
  87. },
  88. credentialId: {
  89. type: 'string', title: 'Credential',
  90. description: 'A basic credential holding the sdcpp-restapi username and password',
  91. dynamicOptions: { source: 'credentials', filter: { type: ['sdcpp', 'basic'] } }
  92. },
  93. modelName: {
  94. type: 'string', title: 'Model',
  95. description: 'File name of the model, relative to its type directory. Browse lists what the server has of the Model Type chosen below',
  96. dynamicOptions: {
  97. source: 'node',
  98. node: 'sdcpp-model',
  99. config: { listOnly: true },
  100. itemsPath: 'models',
  101. valueKey: 'name',
  102. labelKey: 'name',
  103. needs: ['serverUrl', 'credentialId']
  104. }
  105. },
  106. modelType: {
  107. type: 'string', title: 'Model Type',
  108. enum: ['', 'checkpoint', 'diffusion'],
  109. default: '',
  110. description: 'checkpoint bundles U-Net, CLIP and VAE and suits SD1, SD2 and SDXL. diffusion holds only the U-Net or DiT and needs its components named separately, which is how Flux, SD3, Qwen, Wan and Z-Image load'
  111. },
  112. vae: {
  113. type: 'string', title: 'VAE',
  114. description: 'Component file name',
  115. dynamicOptions: {
  116. source: 'node',
  117. node: 'sdcpp-model',
  118. config: { listOnly: true, modelType: 'vae' },
  119. itemsPath: 'models',
  120. valueKey: 'name',
  121. labelKey: 'name',
  122. needs: ['serverUrl', 'credentialId']
  123. }
  124. },
  125. clipL: {
  126. type: 'string', title: 'CLIP-L',
  127. description: 'Component file name',
  128. dynamicOptions: {
  129. source: 'node',
  130. node: 'sdcpp-model',
  131. config: { listOnly: true, modelType: 'clip' },
  132. itemsPath: 'models',
  133. valueKey: 'name',
  134. labelKey: 'name',
  135. needs: ['serverUrl', 'credentialId']
  136. }
  137. },
  138. clipG: {
  139. type: 'string', title: 'CLIP-G',
  140. description: 'Component file name',
  141. dynamicOptions: {
  142. source: 'node',
  143. node: 'sdcpp-model',
  144. config: { listOnly: true, modelType: 'clip' },
  145. itemsPath: 'models',
  146. valueKey: 'name',
  147. labelKey: 'name',
  148. needs: ['serverUrl', 'credentialId']
  149. }
  150. },
  151. t5xxl: {
  152. type: 'string', title: 'T5-XXL',
  153. description: 'Component file name',
  154. dynamicOptions: {
  155. source: 'node',
  156. node: 'sdcpp-model',
  157. config: { listOnly: true, modelType: 't5' },
  158. itemsPath: 'models',
  159. valueKey: 'name',
  160. labelKey: 'name',
  161. needs: ['serverUrl', 'credentialId']
  162. }
  163. },
  164. llm: {
  165. type: 'string', title: 'LLM',
  166. description: 'Component file name, used by Z-Image, Qwen, Anima and Flux2',
  167. dynamicOptions: {
  168. source: 'node',
  169. node: 'sdcpp-model',
  170. config: { listOnly: true, modelType: 'llm' },
  171. itemsPath: 'models',
  172. valueKey: 'name',
  173. labelKey: 'name',
  174. needs: ['serverUrl', 'credentialId']
  175. }
  176. },
  177. taesd: {
  178. type: 'string', title: 'TAESD',
  179. description: 'Tiny autoencoder for progress previews',
  180. dynamicOptions: {
  181. source: 'node',
  182. node: 'sdcpp-model',
  183. config: { listOnly: true, modelType: 'taesd' },
  184. itemsPath: 'models',
  185. valueKey: 'name',
  186. labelKey: 'name',
  187. needs: ['serverUrl', 'credentialId']
  188. }
  189. },
  190. controlnet: {
  191. type: 'string', title: 'ControlNet',
  192. description: 'Component file name',
  193. dynamicOptions: {
  194. source: 'node',
  195. node: 'sdcpp-model',
  196. config: { listOnly: true, modelType: 'controlnet' },
  197. itemsPath: 'models',
  198. valueKey: 'name',
  199. labelKey: 'name',
  200. needs: ['serverUrl', 'credentialId']
  201. }
  202. },
  203. ipAdapter: {
  204. type: 'string', title: 'IP-Adapter',
  205. description: 'Component file name. An IP-Adapter lets a generation take its style and subject from a reference image - load it here, then set the reference on the generation node. Classic and Plus/Resampler variants both work',
  206. dynamicOptions: {
  207. source: 'node',
  208. node: 'sdcpp-model',
  209. config: { listOnly: true, modelType: 'ip_adapter' },
  210. itemsPath: 'models',
  211. valueKey: 'name',
  212. labelKey: 'name',
  213. needs: ['serverUrl', 'credentialId']
  214. }
  215. },
  216. flashAttn: { type: 'boolean', title: 'Flash Attention', description: 'For CLIP and T5. A large speed and memory win on modern GPUs' },
  217. diffusionFlashAttn: { type: 'boolean', title: 'Flash Attention (diffusion)', description: 'Flash attention for the diffusion model specifically' },
  218. enableMmap: { type: 'boolean', title: 'Memory-map Weights', description: 'Recommended for large files' },
  219. eagerLoad: { type: 'boolean', title: 'Eager Load', description: 'Move every parameter to the compute backend at load time instead of on demand' },
  220. streamLayers: { type: 'boolean', title: 'Stream Layers', description: 'Stream diffusion layers when the model does not fit in VRAM. Pair with a VRAM budget' },
  221. maxVram: { type: 'number', title: 'VRAM Budget (GiB)', description: 'Budget for segmented parameter offload. 0 leaves it to the server' },
  222. nThreads: { type: 'number', title: 'CPU Threads', description: '-1 lets the server decide' },
  223. weightType: {
  224. type: 'string', title: 'Weight Type',
  225. enum: ['', 'f32', 'f16', 'bf16', 'q8_0', 'q5_0', 'q5_1', 'q4_0', 'q4_1', 'q4_k', 'q5_k', 'q6_k', 'q8_k', 'q3_k', 'q2_k', 'mxfp4', 'nvfp4', 'q1_0'],
  226. enumLabels: ['auto - use the weights in the file', 'f32', 'f16', 'bf16', 'q8_0', 'q5_0', 'q5_1', 'q4_0', 'q4_1', 'q4_k', 'q5_k', 'q6_k', 'q8_k', 'q3_k', 'q2_k', 'mxfp4', 'nvfp4', 'q1_0'],
  227. default: '', description: 'Force a quantisation. Left on auto, sd.cpp uses whatever the model file holds'
  228. },
  229. vaeFormat: {
  230. type: 'string', title: 'VAE Format',
  231. enum: ['', 'auto', 'flux', 'sd3', 'flux2', 'wan'],
  232. enumLabels: ['not set - leave it to the server', 'auto - detect from the file', 'flux', 'sd3', 'flux2', 'wan'],
  233. default: '', description: 'Override VAE format detection'
  234. },
  235. prediction: {
  236. type: 'string', title: 'Prediction Type',
  237. enum: ['', 'eps', 'v', 'edm_v', 'sd3_flow', 'flux_flow', 'flux2_flow', 'sefi_flow', 'minit2i_flow'],
  238. enumLabels: ['auto - detect from the model', 'eps', 'v', 'edm_v', 'sd3_flow', 'flux_flow', 'flux2_flow', 'sefi_flow', 'minit2i_flow'],
  239. default: '', description: 'Override the prediction type. Left on auto it is detected from the model'
  240. },
  241. rngType: {
  242. type: 'string', title: 'RNG', enum: ['', 'cuda', 'std_default', 'cpu'],
  243. enumLabels: ['server default', 'cuda', 'std_default', 'cpu'],
  244. default: '', description: 'Affects whether a seed reproduces across backends'
  245. },
  246. samplerRngType: {
  247. type: 'string', title: 'Sampler RNG', enum: ['', 'cuda', 'std_default', 'cpu'],
  248. enumLabels: ['same as RNG above', 'cuda', 'std_default', 'cpu'],
  249. default: '',
  250. description: 'Override the RNG for sampling only. The server does not report this one, so Fill in leaves it alone'
  251. },
  252. loraApplyMode: {
  253. type: 'string', title: 'LoRA Apply Mode', enum: ['', 'auto', 'immediately', 'at_runtime'],
  254. enumLabels: ['server default', 'auto', 'immediately', 'at_runtime'],
  255. default: '', description: 'When a LoRA named in a prompt is applied'
  256. },
  257. vaeConvDirect: { type: 'boolean', title: 'Direct VAE Convolution', description: 'Use the ggml_conv2d_direct path for the VAE' },
  258. diffusionConvDirect: { type: 'boolean', title: 'Direct Diffusion Convolution', description: 'Use the ggml_conv2d_direct path for the diffusion model' },
  259. forceSdxlVaeConvScale: { type: 'boolean', title: 'Force SDXL VAE Conv Scale', description: 'SDXL-specific VAE convolution scaling' },
  260. taePreviewOnly: { type: 'boolean', title: 'TAESD For Preview Only', description: 'Load TAESD purely to render progress previews, skipping the full VAE' },
  261. backend: { type: 'string', title: 'Backend', description: 'Per-component placement, such as te=cpu,vae=cpu,controlnet=cpu' },
  262. paramsBackend: { type: 'string', title: 'Parameter Backend', description: 'Global parameter placement, such as *=cpu to hold weights in system RAM' },
  263. rpcServers: { type: 'string', title: 'RPC Servers', description: 'Comma separated RPC backend endpoints' },
  264. modelArgs: { type: 'string', title: 'Model Args', description: 'Architecture-specific key=value knobs, comma separated' },
  265. tensorTypeRules: { type: 'string', title: 'Tensor Type Rules', description: 'Per-tensor weight overrides using regex, such as ^vae\\.=f16' },
  266. options: {
  267. type: 'object', title: 'Other Load Options',
  268. description: 'Extra load options passed through, such as flash_attn, enable_mmap, weight_type, stream_layers or max_vram'
  269. },
  270. whenDifferent: {
  271. type: 'string', title: 'When A Different Model Is Loaded',
  272. enum: ['load', 'fail'],
  273. default: 'load',
  274. description: 'load swaps it. fail stops the run instead - for a workflow that depends on a particular model already being in place and should not quietly spend minutes swapping it'
  275. },
  276. force: {
  277. type: 'boolean', title: 'Force Reload',
  278. default: false,
  279. description: 'Load again even when the right model is already loaded. Costs the full load time; useful after changing components or options, which this node cannot see from outside',
  280. showWhen: { field: 'whenDifferent', value: 'load' }
  281. },
  282. timeout: {
  283. type: 'number', title: 'Timeout (ms)',
  284. description: 'How long to wait on any one request to the server. Loading reads gigabytes from disk, so the request that starts it is given this long',
  285. default: 300000
  286. },
  287. loadWaitMs: {
  288. type: 'number', title: 'Wait For Loading (ms)',
  289. description: 'How long to keep watching after the load has started. This is separate from the timeout above because a request giving up says nothing about whether the server is still working - it usually is, and the node follows it through /health rather than reporting a failure that is really just impatience',
  290. default: 900000
  291. }
  292. },
  293. required: []
  294. };
  295. const inputSchema = { type: 'object', properties: { data: { type: 'any' } } };
  296. const outputSchema = {
  297. type: 'object',
  298. properties: {
  299. modelName: { type: 'string', description: 'The model that is loaded now' },
  300. modelType: { type: 'string' },
  301. architecture: { type: 'string', description: 'Architecture the server detected, which decides generation defaults' },
  302. loaded: { type: 'boolean', description: 'True when this node performed a load' },
  303. alreadyLoaded: { type: 'boolean', description: 'True when the right model, with the right settings, was already in place' },
  304. reloadedFor: { type: 'array', description: 'Settings that differed on an otherwise-correct model, when that is why it was reloaded' },
  305. previousModel: { type: 'string', description: 'What was loaded before, when this node swapped it' },
  306. loadedComponents: { type: 'object' },
  307. elapsedMs: { type: 'number' }
  308. }
  309. };
  310. function normalizeServer(url) {
  311. const value = String(url || '').trim();
  312. if (!value) {
  313. throw new Error('SD.cpp: a server URL is required, such as http://localhost:8077');
  314. }
  315. return value.replace(/\/+$/, '');
  316. }
  317. function readCredential(credentialId) {
  318. const auth = smartbotic.credentials.get(credentialId);
  319. if (!auth || auth.success !== true) {
  320. throw new Error('SD.cpp: could not read the credential: ' +
  321. ((auth && auth.error) || 'unknown error'));
  322. }
  323. const value = auth.headerValue || '';
  324. if (value.indexOf('Basic ') !== 0) {
  325. throw new Error('SD.cpp: the credential must be a basic one, holding the sdcpp-restapi ' +
  326. 'username and password');
  327. }
  328. const decoded = smartbotic.utils.base64Decode(value.substring(6));
  329. const separator = decoded.indexOf(':');
  330. if (separator < 1) {
  331. throw new Error('SD.cpp: the credential is malformed, expected a username and a password');
  332. }
  333. return {
  334. username: decoded.substring(0, separator),
  335. password: decoded.substring(separator + 1)
  336. };
  337. }
  338. function call(options) {
  339. const response = smartbotic.http.request(options);
  340. let body = response.data;
  341. if (typeof body === 'string' && body.length > 0) {
  342. try {
  343. body = JSON.parse(body);
  344. } catch (e) {
  345. const snippet = body.substring(0, 200).replace(/\s+/g, ' ');
  346. throw new Error('SD.cpp: ' + options.what + ' returned HTTP ' + response.status +
  347. ' with a body that is not JSON: ' + snippet);
  348. }
  349. }
  350. if (response.status < 200 || response.status >= 300) {
  351. const detail = (body && (body.message || body.error)) || ('HTTP ' + response.status);
  352. throw new Error('SD.cpp: ' + options.what + ' failed: ' + detail);
  353. }
  354. return body || {};
  355. }
  356. function login(server, credential, timeout) {
  357. const session = call({
  358. method: 'POST',
  359. url: server + '/auth/login',
  360. headers: { 'Content-Type': 'application/json' },
  361. body: JSON.stringify({
  362. username: credential.username,
  363. password: credential.password
  364. }),
  365. timeout: timeout,
  366. what: 'signing in'
  367. });
  368. if (!session.token) {
  369. throw new Error('SD.cpp: the server accepted the login but returned no token');
  370. }
  371. return session.token;
  372. }
  373. // /health is unauthenticated, and it is the only way to find out what is
  374. // already loaded without asking for a token first.
  375. function readHealth(server, timeout) {
  376. return call({
  377. method: 'GET',
  378. url: server + '/health',
  379. timeout: timeout,
  380. // Reading health is safe to repeat, and the moment it matters most is
  381. // the moment the server is busiest.
  382. retries: 2,
  383. retryDelayMs: 1000,
  384. what: 'reading server health'
  385. });
  386. }
  387. // The same, but a server that does not answer is treated as one that is busy
  388. // rather than one that has failed. A machine part-way through loading eleven
  389. // gigabytes answers /health slowly or not at all - that is what loading looks
  390. // like from outside, and throwing there ended the whole run for the one thing
  391. // the node was waiting for.
  392. function pollHealth(server, timeout) {
  393. try {
  394. return readHealth(server, timeout);
  395. } catch (e) {
  396. smartbotic.log.info('SD.cpp: no answer from /health while loading (' +
  397. ((e && e.message) || e) + '), still waiting');
  398. return null;
  399. }
  400. }
  401. function putIfSet(target, key, value) {
  402. if (value === undefined || value === null || value === '') {
  403. return;
  404. }
  405. target[key] = value;
  406. }
  407. const LOAD_OPTIONS = [
  408. { setting: 'flashAttn', server: 'flash_attn' },
  409. { setting: 'diffusionFlashAttn', server: 'diffusion_flash_attn' },
  410. { setting: 'enableMmap', server: 'enable_mmap' },
  411. { setting: 'eagerLoad', server: 'eager_load' },
  412. { setting: 'streamLayers', server: 'stream_layers' },
  413. { setting: 'maxVram', server: 'max_vram' },
  414. { setting: 'nThreads', server: 'n_threads' },
  415. { setting: 'weightType', server: 'weight_type' },
  416. { setting: 'vaeFormat', server: 'vae_format' },
  417. { setting: 'prediction', server: 'prediction' },
  418. { setting: 'rngType', server: 'rng_type' },
  419. { setting: 'samplerRngType', server: 'sampler_rng_type' },
  420. { setting: 'loraApplyMode', server: 'lora_apply_mode' },
  421. { setting: 'vaeConvDirect', server: 'vae_conv_direct' },
  422. { setting: 'diffusionConvDirect', server: 'diffusion_conv_direct' },
  423. { setting: 'taePreviewOnly', server: 'tae_preview_only' },
  424. { setting: 'forceSdxlVaeConvScale', server: 'force_sdxl_vae_conv_scale' },
  425. { setting: 'backend', server: 'backend' },
  426. { setting: 'paramsBackend', server: 'params_backend' },
  427. { setting: 'rpcServers', server: 'rpc_servers' },
  428. { setting: 'modelArgs', server: 'model_args' },
  429. { setting: 'tensorTypeRules', server: 'tensor_type_rules' },
  430. ];
  431. // The options this node asks for, as the server names them. Only settings that
  432. // were actually filled in are included: an untouched setting means "whatever
  433. // the server does", not "the default", so it is neither sent nor compared.
  434. function wantedOptions(config) {
  435. var wanted = {};
  436. for (var i = 0; i < LOAD_OPTIONS.length; i++) {
  437. var entry = LOAD_OPTIONS[i];
  438. var value = config[entry.setting];
  439. if (value === undefined || value === null || value === '') continue;
  440. if (typeof value === 'number' && !isFinite(value)) continue;
  441. wanted[entry.server] = value;
  442. }
  443. if (config.options && typeof config.options === 'object') {
  444. var keys = Object.keys(config.options);
  445. for (var k = 0; k < keys.length; k++) {
  446. var extra = config.options[keys[k]];
  447. if (extra !== undefined && extra !== null && extra !== '') {
  448. wanted[keys[k]] = extra;
  449. }
  450. }
  451. }
  452. return wanted;
  453. }
  454. // Which of them the server is not currently loaded with.
  455. //
  456. // This is what makes the node able to correct a server someone else changed:
  457. // the same model loaded with streaming off is not the same thing as the model
  458. // this workflow needs, and reloading it is the whole point of saying so here.
  459. function optionsThatDiffer(wanted, current) {
  460. var differing = [];
  461. var keys = Object.keys(wanted);
  462. for (var i = 0; i < keys.length; i++) {
  463. var key = keys[i];
  464. var have = current ? current[key] : undefined;
  465. var want = wanted[key];
  466. // Numbers arrive as 0 or 0.0 depending on the field, and a boolean may
  467. // come back as a string from a form, so compare on value rather than
  468. // on type.
  469. var same = (typeof want === 'number' || typeof have === 'number')
  470. ? Number(have) === Number(want)
  471. : String(have) === String(want);
  472. if (!same) {
  473. differing.push(key + ': server has ' + JSON.stringify(have) +
  474. ', this node wants ' + JSON.stringify(want));
  475. }
  476. }
  477. return differing;
  478. }
  479. async function execute(config, input, context) {
  480. const server = normalizeServer(config.serverUrl);
  481. const timeout = config.timeout > 0 ? config.timeout : 300000;
  482. // Watching costs nothing, so it is allowed to outlast any single request.
  483. const loadWait = config.loadWaitMs > 0 ? config.loadWaitMs : 900000;
  484. const modelName = String(config.modelName || '').trim();
  485. if (!modelName) {
  486. throw new Error('SD.cpp: a model name is required. Connect an SD.cpp Model node, ' +
  487. 'or type the file name');
  488. }
  489. // Ask what is loaded before loading anything. A load takes minutes and
  490. // unloads whatever was there, so doing it when the right model is already
  491. // resident is pure cost - and on a shared server it disrupts other work.
  492. const startedAt = Date.now();
  493. const health = readHealth(server, Math.min(timeout, 15000));
  494. const current = health.model_name || '';
  495. const sameModel = current === modelName;
  496. // The same model loaded with different settings is not the model this
  497. // workflow asked for. The server swaps models between queue items, so what
  498. // is loaded now may have been put there by something else entirely.
  499. const wanted = wantedOptions(config);
  500. const differing = optionsThatDiffer(wanted, health.load_options || {});
  501. if (sameModel && differing.length === 0 && config.force !== true) {
  502. smartbotic.log.info('SD.cpp: ' + modelName + ' is already loaded with the wanted settings');
  503. return {
  504. modelName: current,
  505. modelType: health.model_type || '',
  506. architecture: health.model_architecture || '',
  507. loaded: false,
  508. alreadyLoaded: true,
  509. reloadedFor: [],
  510. previousModel: '',
  511. loadedComponents: health.loaded_components || {},
  512. elapsedMs: Date.now() - startedAt
  513. };
  514. }
  515. if ((config.whenDifferent || 'load') === 'fail' && (!sameModel || differing.length > 0)) {
  516. if (!sameModel) {
  517. throw new Error('SD.cpp: this workflow expects "' + modelName + '" to be loaded, but ' +
  518. (current ? 'the server has "' + current + '"' : 'no model is loaded') +
  519. '. Set When A Different Model Is Loaded to "load" to swap it automatically');
  520. }
  521. throw new Error('SD.cpp: "' + modelName + '" is loaded, but not with the settings this ' +
  522. 'workflow needs - ' + differing.join('; ') +
  523. '. Set When A Different Model Is Loaded to "load" to reload it');
  524. }
  525. if (sameModel && differing.length > 0) {
  526. smartbotic.log.info('SD.cpp: reloading ' + modelName + ' because ' + differing.join('; '));
  527. }
  528. const credential = readCredential(config.credentialId);
  529. const token = login(server, credential, Math.min(timeout, 30000));
  530. const body = { model_name: modelName };
  531. putIfSet(body, 'model_type', config.modelType);
  532. putIfSet(body, 'vae', config.vae);
  533. putIfSet(body, 'clip_l', config.clipL);
  534. putIfSet(body, 'clip_g', config.clipG);
  535. putIfSet(body, 't5xxl', config.t5xxl);
  536. putIfSet(body, 'llm', config.llm);
  537. putIfSet(body, 'taesd', config.taesd);
  538. putIfSet(body, 'controlnet', config.controlnet);
  539. putIfSet(body, 'ip_adapter', config.ipAdapter);
  540. if (Object.keys(wanted).length > 0) {
  541. body.options = wanted;
  542. }
  543. smartbotic.log.info('SD.cpp: loading ' + modelName +
  544. (current ? ' (replacing ' + current + ')' : ''));
  545. // The slot has to be emptied first. The API documentation says a load
  546. // replaces whatever is there, but the server answers 409 "A model is
  547. // already loaded. Call POST /models/unload first" - so it is unloaded here
  548. // rather than leaving every reload to fail on a server that already has a
  549. // model. (A refused load is harmless: the resident model stays put.)
  550. //
  551. // This is also the only way to change the settings of a model that is
  552. // already loaded, which is the case this node exists to handle.
  553. //
  554. // It does mean everything between here and a finished load runs with the
  555. // server holding nothing. The server never unloads on its own, so an empty
  556. // slot afterwards is always something that happened in this window - which
  557. // is why the failure paths below say so rather than leaving the next run to
  558. // discover it.
  559. let emptiedTheSlot = false;
  560. if (health.model_loaded === true) {
  561. call({
  562. method: 'POST',
  563. url: server + '/models/unload',
  564. headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token },
  565. body: '{}',
  566. timeout: Math.min(timeout, 60000),
  567. what: 'unloading ' + (current || 'the current model') + ' before loading ' + modelName
  568. });
  569. smartbotic.log.info('SD.cpp: unloaded ' + (current || 'the previous model'));
  570. emptiedTheSlot = true;
  571. }
  572. // Loading unloads whatever was in the slot first, and the server holds a
  573. // mutex for the duration, so this blocks until the weights are resident.
  574. let loaded;
  575. try {
  576. loaded = call({
  577. method: 'POST',
  578. url: server + '/models/load',
  579. headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token },
  580. body: JSON.stringify(body),
  581. timeout: timeout,
  582. // Deliberately not retried: the server holds a mutex for the whole
  583. // load, so a second request would queue behind the first and load
  584. // the same weights twice.
  585. what: 'loading model ' + modelName
  586. });
  587. } catch (loadError) {
  588. // This call giving up does not mean the server did. If it is still
  589. // loading, that is the answer to what happened - so the wait below
  590. // finds out how it goes rather than reporting a failure that is really
  591. // just impatience.
  592. const probe = pollHealth(server, 15000);
  593. if (!probe || probe.model_loading !== true) {
  594. // The slot was emptied to make room and the load did not take, so
  595. // the server now holds nothing. One more attempt is worth it: there
  596. // is nothing left to lose, the usual cause is a moment of
  597. // slowness, and the alternative is leaving the server worse than it
  598. // was found.
  599. if (emptiedTheSlot && (!probe || probe.model_loaded !== true)) {
  600. smartbotic.log.warn('SD.cpp: the load failed and the server now has no model. ' +
  601. 'Trying once more before giving up');
  602. try {
  603. loaded = call({
  604. method: 'POST',
  605. url: server + '/models/load',
  606. headers: { 'Content-Type': 'application/json',
  607. 'Authorization': 'Bearer ' + token },
  608. body: JSON.stringify(body),
  609. timeout: timeout,
  610. what: 'loading model ' + modelName + ' (second attempt)'
  611. });
  612. // Falls through to the wait below, the same as a first
  613. // attempt that worked - the model still has to finish
  614. // loading either way.
  615. } catch (secondError) {
  616. throw new Error('SD.cpp: could not load ' + modelName + ', and the server ' +
  617. 'is now holding no model at all - it was unloaded to make room. ' +
  618. 'Nothing will generate until a load succeeds. First attempt: ' +
  619. ((loadError && loadError.message) || loadError) + '. Second: ' +
  620. ((secondError && secondError.message) || secondError));
  621. }
  622. }
  623. throw loadError;
  624. }
  625. smartbotic.log.info('SD.cpp: the load request stopped waiting, but the server is still ' +
  626. 'loading ' + (probe.loading_model_name || modelName) + ' - following it through /health');
  627. loaded = {};
  628. }
  629. // The load call comes back before the model is in memory. The API
  630. // documentation describes it as blocking, and it is not: /health reports
  631. // model_loading with a step count for some time afterwards. Returning here
  632. // would tell the workflow the model is ready and let the next node ask it
  633. // to generate, which fails with "no model loaded" - a confusing way to
  634. // learn that this node lied.
  635. const deadline = Date.now() + loadWait;
  636. let after = pollHealth(server, 15000);
  637. let lastStep = -1;
  638. let silentPolls = 0;
  639. while ((after === null || after.model_loading === true) && Date.now() < deadline) {
  640. if (after === null) {
  641. silentPolls++;
  642. } else {
  643. silentPolls = 0;
  644. const step = after.loading_step;
  645. const total = after.loading_total_steps;
  646. if (typeof step === 'number' && step !== lastStep) {
  647. lastStep = step;
  648. smartbotic.log.info('SD.cpp: loading ' + (after.loading_model_name || modelName) +
  649. ' - ' + step + (total ? '/' + total : ''));
  650. }
  651. }
  652. smartbotic.utils.sleep(2000);
  653. after = pollHealth(server, 15000);
  654. }
  655. const waitedSeconds = Math.round((Date.now() - startedAt) / 1000);
  656. if (after === null) {
  657. throw new Error('SD.cpp: ' + modelName + ' was asked for ' + waitedSeconds +
  658. 's ago and the server has stopped answering /health (' + silentPolls +
  659. ' polls in a row went unanswered). It may still be loading - check the ' +
  660. 'server, and raise the timeout on this node if this model is simply slow');
  661. }
  662. if (after.model_loading === true) {
  663. throw new Error('SD.cpp: ' + modelName + ' was still loading after ' + waitedSeconds +
  664. 's' + (typeof after.loading_step === 'number'
  665. ? ' (at step ' + after.loading_step +
  666. (after.loading_total_steps ? ' of ' + after.loading_total_steps : '') + ')'
  667. : '') +
  668. '. It may still finish on the server; raise the timeout on this node if this ' +
  669. 'model is simply slow to load');
  670. }
  671. if (after.model_loaded !== true) {
  672. throw new Error('SD.cpp: the server accepted the load but has no model loaded afterwards' +
  673. (after.last_error ? ': ' + after.last_error : ''));
  674. }
  675. return {
  676. modelName: loaded.model_name || modelName,
  677. modelType: loaded.model_type || config.modelType || '',
  678. architecture: after.model_architecture || '',
  679. loaded: true,
  680. alreadyLoaded: false,
  681. // Empty when the model itself changed; otherwise the settings that
  682. // forced a reload of a model that was already there.
  683. reloadedFor: sameModel ? differing : [],
  684. previousModel: sameModel ? '' : current,
  685. loadedComponents: loaded.loaded_components || after.loaded_components || {},
  686. elapsedMs: Date.now() - startedAt
  687. };
  688. }
  689. module.exports = { configSchema, inputSchema, outputSchema, execute };