sdcpp-model-load.js 24 KB

123456789101112131415161718192021222324252627282930313233343536373839404142434445464748495051525354555657585960616263646566676869707172737475767778798081828384858687888990919293949596979899100101102103104105106107108109110111112113114115116117118119120121122123124125126127128129130131132133134135136137138139140141142143144145146147148149150151152153154155156157158159160161162163164165166167168169170171172173174175176177178179180181182183184185186187188189190191192193194195196197198199200201202203204205206207208209210211212213214215216217218219220221222223224225226227228229230231232233234235236237238239240241242243244245246247248249250251252253254255256257258259260261262263264265266267268269270271272273274275276277278279280281282283284285286287288289290291292293294295296297298299300301302303304305306307308309310311312313314315316317318319320321322323324325326327328329330331332333334335336337338339340341342343344345346347348349350351352353354355356357358359360361362363364365366367368369370371372373374375376377378379380381382383384385386387388389390391392393394395396397398399400401402403404405406407408409410411412413414415416417418419420421422423424425426427428429430431432433434435436437438439440441442443444445446447448449450451452453454455456457458459460461462463464465466467468469470471472473474475476477478479480481482483484485486487488489490491492493494495496497498499500501502503504505506507508509510511512513514515516517518519520521522523524525526527528529530531532533534535536537538539540541542543544545546547548549550
  1. /**
  2. * @node sdcpp-model-load
  3. * @name SD.cpp Load Model
  4. * @category sdcpp
  5. * @version 1.0.0
  6. * @description Make sure a model is loaded, without reloading one that already is
  7. * @icon box
  8. */
  9. // The credential this node wants, named so it can be found. It is stored as a
  10. // plain basic credential - that is what decides how it is encrypted - and this
  11. // only says which basic credential is the SD.cpp one. Anything that accepts a
  12. // basic credential still accepts this, and this node still accepts a plain
  13. // basic credential, because the shape is identical.
  14. const credentialTypes = [
  15. {
  16. id: 'sdcpp',
  17. label: 'SD.cpp Server',
  18. baseType: 'basic',
  19. description: 'The username and password you sign in to sdcpp-restapi with. The node exchanges them for a token before every call',
  20. usernameLabel: 'Username',
  21. passwordLabel: 'Password'
  22. }
  23. ];
  24. const configSchema = {
  25. type: 'object',
  26. uiGroups: [
  27. { title: 'Connection', fields: ['serverUrl', 'credentialId'] },
  28. { title: 'Model', fields: ['modelName', 'modelType', 'whenDifferent', 'force'] },
  29. { title: 'Components', fields: ['vae', 'clipL', 'clipG', 't5xxl', 'llm', 'taesd', 'controlnet'] },
  30. { title: 'Loading', fields: ['flashAttn', 'diffusionFlashAttn', 'enableMmap', 'eagerLoad',
  31. 'streamLayers', 'maxVram', 'nThreads', 'weightType'] },
  32. { title: 'Advanced', fields: ['vaeFormat', 'prediction', 'rngType', 'samplerRngType',
  33. 'loraApplyMode', 'vaeConvDirect', 'diffusionConvDirect',
  34. 'taePreviewOnly', 'backend', 'paramsBackend', 'rpcServers',
  35. 'modelArgs', 'tensorTypeRules', 'options', 'timeout'] }
  36. ],
  37. // Pressing this asks the server what it has loaded and writes it into the
  38. // settings below - the model and every component it reports - so a workflow
  39. // can be built from a server that is already set up the way it should be,
  40. // rather than by typing the same names again.
  41. prefill: {
  42. label: 'Take from the server',
  43. description: 'Fill these in from the model the server currently has loaded',
  44. node: 'sdcpp-health',
  45. needs: ['serverUrl'],
  46. map: {
  47. modelName: 'modelName',
  48. modelType: 'modelType',
  49. 'loadedComponents.vae': 'vae',
  50. 'loadedComponents.clip_l': 'clipL',
  51. 'loadedComponents.clip_g': 'clipG',
  52. 'loadedComponents.t5xxl': 't5xxl',
  53. 'loadedComponents.llm': 'llm',
  54. 'loadedComponents.taesd': 'taesd',
  55. 'loadedComponents.controlnet': 'controlnet',
  56. 'loadOptions.flash_attn': 'flashAttn',
  57. 'loadOptions.diffusion_flash_attn': 'diffusionFlashAttn',
  58. 'loadOptions.enable_mmap': 'enableMmap',
  59. 'loadOptions.eager_load': 'eagerLoad',
  60. 'loadOptions.stream_layers': 'streamLayers',
  61. 'loadOptions.max_vram': 'maxVram',
  62. 'loadOptions.n_threads': 'nThreads',
  63. 'loadOptions.vae_format': 'vaeFormat',
  64. 'loadOptions.rng_type': 'rngType',
  65. 'loadOptions.lora_apply_mode': 'loraApplyMode',
  66. 'loadOptions.vae_conv_direct': 'vaeConvDirect',
  67. 'loadOptions.diffusion_conv_direct': 'diffusionConvDirect',
  68. 'loadOptions.tae_preview_only': 'taePreviewOnly',
  69. 'loadOptions.backend': 'backend',
  70. 'loadOptions.params_backend': 'paramsBackend',
  71. 'loadOptions.rpc_servers': 'rpcServers',
  72. 'loadOptions.model_args': 'modelArgs'
  73. }
  74. },
  75. properties: {
  76. serverUrl: {
  77. type: 'string', title: 'Server URL',
  78. description: 'Base address of the sdcpp-restapi server',
  79. default: 'http://localhost:8077'
  80. },
  81. credentialId: {
  82. type: 'string', title: 'Credential',
  83. description: 'A basic credential holding the sdcpp-restapi username and password',
  84. dynamicOptions: { source: 'credentials', filter: { type: ['sdcpp', 'basic'] } }
  85. },
  86. modelName: {
  87. type: 'string', title: 'Model',
  88. description: 'File name of the model, relative to its type directory. Browse lists what the server has of the Model Type chosen below',
  89. dynamicOptions: {
  90. source: 'node',
  91. node: 'sdcpp-model',
  92. config: { listOnly: true },
  93. itemsPath: 'models',
  94. valueKey: 'name',
  95. labelKey: 'name',
  96. needs: ['serverUrl', 'credentialId']
  97. }
  98. },
  99. modelType: {
  100. type: 'string', title: 'Model Type',
  101. enum: ['', 'checkpoint', 'diffusion'],
  102. default: '',
  103. description: 'checkpoint bundles U-Net, CLIP and VAE and suits SD1, SD2 and SDXL. diffusion holds only the U-Net or DiT and needs its components named separately, which is how Flux, SD3, Qwen, Wan and Z-Image load'
  104. },
  105. vae: {
  106. type: 'string', title: 'VAE',
  107. description: 'Component file name',
  108. dynamicOptions: {
  109. source: 'node',
  110. node: 'sdcpp-model',
  111. config: { listOnly: true, modelType: 'vae' },
  112. itemsPath: 'models',
  113. valueKey: 'name',
  114. labelKey: 'name',
  115. needs: ['serverUrl', 'credentialId']
  116. }
  117. },
  118. clipL: {
  119. type: 'string', title: 'CLIP-L',
  120. description: 'Component file name',
  121. dynamicOptions: {
  122. source: 'node',
  123. node: 'sdcpp-model',
  124. config: { listOnly: true, modelType: 'clip' },
  125. itemsPath: 'models',
  126. valueKey: 'name',
  127. labelKey: 'name',
  128. needs: ['serverUrl', 'credentialId']
  129. }
  130. },
  131. clipG: {
  132. type: 'string', title: 'CLIP-G',
  133. description: 'Component file name',
  134. dynamicOptions: {
  135. source: 'node',
  136. node: 'sdcpp-model',
  137. config: { listOnly: true, modelType: 'clip' },
  138. itemsPath: 'models',
  139. valueKey: 'name',
  140. labelKey: 'name',
  141. needs: ['serverUrl', 'credentialId']
  142. }
  143. },
  144. t5xxl: {
  145. type: 'string', title: 'T5-XXL',
  146. description: 'Component file name',
  147. dynamicOptions: {
  148. source: 'node',
  149. node: 'sdcpp-model',
  150. config: { listOnly: true, modelType: 't5' },
  151. itemsPath: 'models',
  152. valueKey: 'name',
  153. labelKey: 'name',
  154. needs: ['serverUrl', 'credentialId']
  155. }
  156. },
  157. llm: {
  158. type: 'string', title: 'LLM',
  159. description: 'Component file name, used by Z-Image, Qwen, Anima and Flux2',
  160. dynamicOptions: {
  161. source: 'node',
  162. node: 'sdcpp-model',
  163. config: { listOnly: true, modelType: 'llm' },
  164. itemsPath: 'models',
  165. valueKey: 'name',
  166. labelKey: 'name',
  167. needs: ['serverUrl', 'credentialId']
  168. }
  169. },
  170. taesd: {
  171. type: 'string', title: 'TAESD',
  172. description: 'Tiny autoencoder for progress previews',
  173. dynamicOptions: {
  174. source: 'node',
  175. node: 'sdcpp-model',
  176. config: { listOnly: true, modelType: 'taesd' },
  177. itemsPath: 'models',
  178. valueKey: 'name',
  179. labelKey: 'name',
  180. needs: ['serverUrl', 'credentialId']
  181. }
  182. },
  183. controlnet: {
  184. type: 'string', title: 'ControlNet',
  185. description: 'Component file name',
  186. dynamicOptions: {
  187. source: 'node',
  188. node: 'sdcpp-model',
  189. config: { listOnly: true, modelType: 'controlnet' },
  190. itemsPath: 'models',
  191. valueKey: 'name',
  192. labelKey: 'name',
  193. needs: ['serverUrl', 'credentialId']
  194. }
  195. },
  196. flashAttn: { type: 'boolean', title: 'Flash Attention', description: 'For CLIP and T5. A large speed and memory win on modern GPUs' },
  197. diffusionFlashAttn: { type: 'boolean', title: 'Flash Attention (diffusion)', description: 'Flash attention for the diffusion model specifically' },
  198. enableMmap: { type: 'boolean', title: 'Memory-map Weights', description: 'Recommended for large files' },
  199. eagerLoad: { type: 'boolean', title: 'Eager Load', description: 'Move every parameter to the compute backend at load time instead of on demand' },
  200. streamLayers: { type: 'boolean', title: 'Stream Layers', description: 'Stream diffusion layers when the model does not fit in VRAM. Pair with a VRAM budget' },
  201. maxVram: { type: 'number', title: 'VRAM Budget (GiB)', description: 'Budget for segmented parameter offload. 0 leaves it to the server' },
  202. nThreads: { type: 'number', title: 'CPU Threads', description: '-1 lets the server decide' },
  203. weightType: {
  204. type: 'string', title: 'Weight Type',
  205. enum: ['', 'f32', 'f16', 'bf16', 'q8_0', 'q5_0', 'q5_1', 'q4_0', 'q4_1', 'q4_k', 'q5_k', 'q6_k', 'q8_k', 'q3_k', 'q2_k', 'mxfp4', 'nvfp4', 'q1_0'],
  206. default: '', description: 'Force a quantisation. Empty takes it from the file'
  207. },
  208. vaeFormat: {
  209. type: 'string', title: 'VAE Format',
  210. enum: ['', 'auto', 'flux', 'sd3', 'flux2', 'wan'],
  211. default: '', description: 'Override VAE format detection'
  212. },
  213. prediction: {
  214. type: 'string', title: 'Prediction Type',
  215. enum: ['', 'eps', 'v', 'edm_v', 'sd3_flow', 'flux_flow', 'flux2_flow', 'sefi_flow', 'minit2i_flow'],
  216. default: '', description: 'Override the prediction type. Empty is detected'
  217. },
  218. rngType: {
  219. type: 'string', title: 'RNG', enum: ['', 'cuda', 'std_default', 'cpu'],
  220. default: '', description: 'Affects whether a seed reproduces across backends'
  221. },
  222. samplerRngType: {
  223. type: 'string', title: 'Sampler RNG', enum: ['', 'cuda', 'std_default', 'cpu'],
  224. default: '', description: 'Override the RNG for sampling only'
  225. },
  226. loraApplyMode: {
  227. type: 'string', title: 'LoRA Apply Mode', enum: ['', 'auto', 'immediately', 'at_runtime'],
  228. default: '', description: 'When a LoRA named in a prompt is applied'
  229. },
  230. vaeConvDirect: { type: 'boolean', title: 'Direct VAE Convolution', description: 'Use the ggml_conv2d_direct path for the VAE' },
  231. diffusionConvDirect: { type: 'boolean', title: 'Direct Diffusion Convolution', description: 'Use the ggml_conv2d_direct path for the diffusion model' },
  232. taePreviewOnly: { type: 'boolean', title: 'TAESD For Preview Only', description: 'Load TAESD purely to render progress previews, skipping the full VAE' },
  233. backend: { type: 'string', title: 'Backend', description: 'Per-component placement, such as te=cpu,vae=cpu,controlnet=cpu' },
  234. paramsBackend: { type: 'string', title: 'Parameter Backend', description: 'Global parameter placement, such as *=cpu to hold weights in system RAM' },
  235. rpcServers: { type: 'string', title: 'RPC Servers', description: 'Comma separated RPC backend endpoints' },
  236. modelArgs: { type: 'string', title: 'Model Args', description: 'Architecture-specific key=value knobs, comma separated' },
  237. tensorTypeRules: { type: 'string', title: 'Tensor Type Rules', description: 'Per-tensor weight overrides using regex, such as ^vae\\.=f16' },
  238. options: {
  239. type: 'object', title: 'Other Load Options',
  240. description: 'Extra load options passed through, such as flash_attn, enable_mmap, weight_type, stream_layers or max_vram'
  241. },
  242. whenDifferent: {
  243. type: 'string', title: 'When A Different Model Is Loaded',
  244. enum: ['load', 'fail'],
  245. default: 'load',
  246. description: 'load swaps it. fail stops the run instead - for a workflow that depends on a particular model already being in place and should not quietly spend minutes swapping it'
  247. },
  248. force: {
  249. type: 'boolean', title: 'Force Reload',
  250. default: false,
  251. description: 'Load again even when the right model is already loaded. Costs the full load time; useful after changing components or options, which this node cannot see from outside',
  252. showWhen: { field: 'whenDifferent', value: 'load' }
  253. },
  254. timeout: {
  255. type: 'number', title: 'Timeout (ms)',
  256. description: 'Loading reads gigabytes from disk and can take minutes',
  257. default: 300000
  258. }
  259. },
  260. required: []
  261. };
  262. const inputSchema = { type: 'object', properties: { data: { type: 'any' } } };
  263. const outputSchema = {
  264. type: 'object',
  265. properties: {
  266. modelName: { type: 'string', description: 'The model that is loaded now' },
  267. modelType: { type: 'string' },
  268. architecture: { type: 'string', description: 'Architecture the server detected, which decides generation defaults' },
  269. loaded: { type: 'boolean', description: 'True when this node performed a load' },
  270. alreadyLoaded: { type: 'boolean', description: 'True when the right model, with the right settings, was already in place' },
  271. reloadedFor: { type: 'array', description: 'Settings that differed on an otherwise-correct model, when that is why it was reloaded' },
  272. previousModel: { type: 'string', description: 'What was loaded before, when this node swapped it' },
  273. loadedComponents: { type: 'object' },
  274. elapsedMs: { type: 'number' }
  275. }
  276. };
  277. function normalizeServer(url) {
  278. const value = String(url || '').trim();
  279. if (!value) {
  280. throw new Error('SD.cpp: a server URL is required, such as http://localhost:8077');
  281. }
  282. return value.replace(/\/+$/, '');
  283. }
  284. function readCredential(credentialId) {
  285. const auth = smartbotic.credentials.get(credentialId);
  286. if (!auth || auth.success !== true) {
  287. throw new Error('SD.cpp: could not read the credential: ' +
  288. ((auth && auth.error) || 'unknown error'));
  289. }
  290. const value = auth.headerValue || '';
  291. if (value.indexOf('Basic ') !== 0) {
  292. throw new Error('SD.cpp: the credential must be a basic one, holding the sdcpp-restapi ' +
  293. 'username and password');
  294. }
  295. const decoded = smartbotic.utils.base64Decode(value.substring(6));
  296. const separator = decoded.indexOf(':');
  297. if (separator < 1) {
  298. throw new Error('SD.cpp: the credential is malformed, expected a username and a password');
  299. }
  300. return {
  301. username: decoded.substring(0, separator),
  302. password: decoded.substring(separator + 1)
  303. };
  304. }
  305. function call(options) {
  306. const response = smartbotic.http.request(options);
  307. let body = response.data;
  308. if (typeof body === 'string' && body.length > 0) {
  309. try {
  310. body = JSON.parse(body);
  311. } catch (e) {
  312. const snippet = body.substring(0, 200).replace(/\s+/g, ' ');
  313. throw new Error('SD.cpp: ' + options.what + ' returned HTTP ' + response.status +
  314. ' with a body that is not JSON: ' + snippet);
  315. }
  316. }
  317. if (response.status < 200 || response.status >= 300) {
  318. const detail = (body && (body.message || body.error)) || ('HTTP ' + response.status);
  319. throw new Error('SD.cpp: ' + options.what + ' failed: ' + detail);
  320. }
  321. return body || {};
  322. }
  323. function login(server, credential, timeout) {
  324. const session = call({
  325. method: 'POST',
  326. url: server + '/auth/login',
  327. headers: { 'Content-Type': 'application/json' },
  328. body: JSON.stringify({
  329. username: credential.username,
  330. password: credential.password
  331. }),
  332. timeout: timeout,
  333. what: 'signing in'
  334. });
  335. if (!session.token) {
  336. throw new Error('SD.cpp: the server accepted the login but returned no token');
  337. }
  338. return session.token;
  339. }
  340. // /health is unauthenticated, and it is the only way to find out what is
  341. // already loaded without asking for a token first.
  342. function readHealth(server, timeout) {
  343. return call({
  344. method: 'GET',
  345. url: server + '/health',
  346. timeout: timeout,
  347. what: 'reading server health'
  348. });
  349. }
  350. function putIfSet(target, key, value) {
  351. if (value === undefined || value === null || value === '') {
  352. return;
  353. }
  354. target[key] = value;
  355. }
  356. const LOAD_OPTIONS = [
  357. { setting: 'flashAttn', server: 'flash_attn' },
  358. { setting: 'diffusionFlashAttn', server: 'diffusion_flash_attn' },
  359. { setting: 'enableMmap', server: 'enable_mmap' },
  360. { setting: 'eagerLoad', server: 'eager_load' },
  361. { setting: 'streamLayers', server: 'stream_layers' },
  362. { setting: 'maxVram', server: 'max_vram' },
  363. { setting: 'nThreads', server: 'n_threads' },
  364. { setting: 'weightType', server: 'weight_type' },
  365. { setting: 'vaeFormat', server: 'vae_format' },
  366. { setting: 'prediction', server: 'prediction' },
  367. { setting: 'rngType', server: 'rng_type' },
  368. { setting: 'samplerRngType', server: 'sampler_rng_type' },
  369. { setting: 'loraApplyMode', server: 'lora_apply_mode' },
  370. { setting: 'vaeConvDirect', server: 'vae_conv_direct' },
  371. { setting: 'diffusionConvDirect', server: 'diffusion_conv_direct' },
  372. { setting: 'taePreviewOnly', server: 'tae_preview_only' },
  373. { setting: 'backend', server: 'backend' },
  374. { setting: 'paramsBackend', server: 'params_backend' },
  375. { setting: 'rpcServers', server: 'rpc_servers' },
  376. { setting: 'modelArgs', server: 'model_args' },
  377. { setting: 'tensorTypeRules', server: 'tensor_type_rules' },
  378. ];
  379. // The options this node asks for, as the server names them. Only settings that
  380. // were actually filled in are included: an untouched setting means "whatever
  381. // the server does", not "the default", so it is neither sent nor compared.
  382. function wantedOptions(config) {
  383. var wanted = {};
  384. for (var i = 0; i < LOAD_OPTIONS.length; i++) {
  385. var entry = LOAD_OPTIONS[i];
  386. var value = config[entry.setting];
  387. if (value === undefined || value === null || value === '') continue;
  388. if (typeof value === 'number' && !isFinite(value)) continue;
  389. wanted[entry.server] = value;
  390. }
  391. if (config.options && typeof config.options === 'object') {
  392. var keys = Object.keys(config.options);
  393. for (var k = 0; k < keys.length; k++) {
  394. var extra = config.options[keys[k]];
  395. if (extra !== undefined && extra !== null && extra !== '') {
  396. wanted[keys[k]] = extra;
  397. }
  398. }
  399. }
  400. return wanted;
  401. }
  402. // Which of them the server is not currently loaded with.
  403. //
  404. // This is what makes the node able to correct a server someone else changed:
  405. // the same model loaded with streaming off is not the same thing as the model
  406. // this workflow needs, and reloading it is the whole point of saying so here.
  407. function optionsThatDiffer(wanted, current) {
  408. var differing = [];
  409. var keys = Object.keys(wanted);
  410. for (var i = 0; i < keys.length; i++) {
  411. var key = keys[i];
  412. var have = current ? current[key] : undefined;
  413. var want = wanted[key];
  414. // Numbers arrive as 0 or 0.0 depending on the field, and a boolean may
  415. // come back as a string from a form, so compare on value rather than
  416. // on type.
  417. var same = (typeof want === 'number' || typeof have === 'number')
  418. ? Number(have) === Number(want)
  419. : String(have) === String(want);
  420. if (!same) {
  421. differing.push(key + ': server has ' + JSON.stringify(have) +
  422. ', this node wants ' + JSON.stringify(want));
  423. }
  424. }
  425. return differing;
  426. }
  427. async function execute(config, input, context) {
  428. const server = normalizeServer(config.serverUrl);
  429. const timeout = config.timeout > 0 ? config.timeout : 300000;
  430. const modelName = String(config.modelName || '').trim();
  431. if (!modelName) {
  432. throw new Error('SD.cpp: a model name is required. Connect an SD.cpp Model node, ' +
  433. 'or type the file name');
  434. }
  435. // Ask what is loaded before loading anything. A load takes minutes and
  436. // unloads whatever was there, so doing it when the right model is already
  437. // resident is pure cost - and on a shared server it disrupts other work.
  438. const startedAt = Date.now();
  439. const health = readHealth(server, Math.min(timeout, 15000));
  440. const current = health.model_name || '';
  441. const sameModel = current === modelName;
  442. // The same model loaded with different settings is not the model this
  443. // workflow asked for. The server swaps models between queue items, so what
  444. // is loaded now may have been put there by something else entirely.
  445. const wanted = wantedOptions(config);
  446. const differing = optionsThatDiffer(wanted, health.load_options || {});
  447. if (sameModel && differing.length === 0 && config.force !== true) {
  448. smartbotic.log.info('SD.cpp: ' + modelName + ' is already loaded with the wanted settings');
  449. return {
  450. modelName: current,
  451. modelType: health.model_type || '',
  452. architecture: health.model_architecture || '',
  453. loaded: false,
  454. alreadyLoaded: true,
  455. reloadedFor: [],
  456. previousModel: '',
  457. loadedComponents: health.loaded_components || {},
  458. elapsedMs: Date.now() - startedAt
  459. };
  460. }
  461. if ((config.whenDifferent || 'load') === 'fail' && (!sameModel || differing.length > 0)) {
  462. if (!sameModel) {
  463. throw new Error('SD.cpp: this workflow expects "' + modelName + '" to be loaded, but ' +
  464. (current ? 'the server has "' + current + '"' : 'no model is loaded') +
  465. '. Set When A Different Model Is Loaded to "load" to swap it automatically');
  466. }
  467. throw new Error('SD.cpp: "' + modelName + '" is loaded, but not with the settings this ' +
  468. 'workflow needs - ' + differing.join('; ') +
  469. '. Set When A Different Model Is Loaded to "load" to reload it');
  470. }
  471. if (sameModel && differing.length > 0) {
  472. smartbotic.log.info('SD.cpp: reloading ' + modelName + ' because ' + differing.join('; '));
  473. }
  474. const credential = readCredential(config.credentialId);
  475. const token = login(server, credential, Math.min(timeout, 30000));
  476. const body = { model_name: modelName };
  477. putIfSet(body, 'model_type', config.modelType);
  478. putIfSet(body, 'vae', config.vae);
  479. putIfSet(body, 'clip_l', config.clipL);
  480. putIfSet(body, 'clip_g', config.clipG);
  481. putIfSet(body, 't5xxl', config.t5xxl);
  482. putIfSet(body, 'llm', config.llm);
  483. putIfSet(body, 'taesd', config.taesd);
  484. putIfSet(body, 'controlnet', config.controlnet);
  485. if (Object.keys(wanted).length > 0) {
  486. body.options = wanted;
  487. }
  488. smartbotic.log.info('SD.cpp: loading ' + modelName +
  489. (current ? ' (replacing ' + current + ')' : ''));
  490. // Loading unloads whatever was in the slot first, and the server holds a
  491. // mutex for the duration, so this blocks until the weights are resident.
  492. const loaded = call({
  493. method: 'POST',
  494. url: server + '/models/load',
  495. headers: { 'Content-Type': 'application/json', 'Authorization': 'Bearer ' + token },
  496. body: JSON.stringify(body),
  497. timeout: timeout,
  498. what: 'loading model ' + modelName
  499. });
  500. const after = readHealth(server, Math.min(timeout, 15000));
  501. return {
  502. modelName: loaded.model_name || modelName,
  503. modelType: loaded.model_type || config.modelType || '',
  504. architecture: after.model_architecture || '',
  505. loaded: true,
  506. alreadyLoaded: false,
  507. // Empty when the model itself changed; otherwise the settings that
  508. // forced a reload of a model that was already there.
  509. reloadedFor: sameModel ? differing : [],
  510. previousModel: sameModel ? '' : current,
  511. loadedComponents: loaded.loaded_components || after.loaded_components || {},
  512. elapsedMs: Date.now() - startedAt
  513. };
  514. }
  515. module.exports = { configSchema, inputSchema, outputSchema, execute };