|
@@ -30,7 +30,7 @@ const configSchema = {
|
|
|
{ title: 'Model', fields: ['modelName', 'modelType', 'whenDifferent', 'force'] },
|
|
{ title: 'Model', fields: ['modelName', 'modelType', 'whenDifferent', 'force'] },
|
|
|
{ title: 'Components', fields: ['vae', 'clipL', 'clipG', 't5xxl', 'llm', 'taesd', 'controlnet', 'ipAdapter'] },
|
|
{ title: 'Components', fields: ['vae', 'clipL', 'clipG', 't5xxl', 'llm', 'taesd', 'controlnet', 'ipAdapter'] },
|
|
|
{ title: 'Loading', fields: ['flashAttn', 'diffusionFlashAttn', 'enableMmap', 'eagerLoad',
|
|
{ title: 'Loading', fields: ['flashAttn', 'diffusionFlashAttn', 'enableMmap', 'eagerLoad',
|
|
|
- 'streamLayers', 'maxVram', 'nThreads', 'weightType'] },
|
|
|
|
|
|
|
+ 'disablePrefetch', 'disableSegmentedCompute', 'maxVram', 'nThreads', 'weightType'] },
|
|
|
{ title: 'Advanced', fields: ['vaeFormat', 'prediction', 'rngType', 'samplerRngType',
|
|
{ title: 'Advanced', fields: ['vaeFormat', 'prediction', 'rngType', 'samplerRngType',
|
|
|
'loraApplyMode', 'vaeConvDirect', 'diffusionConvDirect',
|
|
'loraApplyMode', 'vaeConvDirect', 'diffusionConvDirect',
|
|
|
'taePreviewOnly', 'forceSdxlVaeConvScale', 'backend', 'paramsBackend', 'rpcServers',
|
|
'taePreviewOnly', 'forceSdxlVaeConvScale', 'backend', 'paramsBackend', 'rpcServers',
|
|
@@ -60,7 +60,8 @@ const configSchema = {
|
|
|
'loadOptions.diffusion_flash_attn': 'diffusionFlashAttn',
|
|
'loadOptions.diffusion_flash_attn': 'diffusionFlashAttn',
|
|
|
'loadOptions.enable_mmap': 'enableMmap',
|
|
'loadOptions.enable_mmap': 'enableMmap',
|
|
|
'loadOptions.eager_load': 'eagerLoad',
|
|
'loadOptions.eager_load': 'eagerLoad',
|
|
|
- 'loadOptions.stream_layers': 'streamLayers',
|
|
|
|
|
|
|
+ 'loadOptions.disable_prefetch': 'disablePrefetch',
|
|
|
|
|
+ 'loadOptions.disable_segmented_compute': 'disableSegmentedCompute',
|
|
|
'loadOptions.max_vram': 'maxVram',
|
|
'loadOptions.max_vram': 'maxVram',
|
|
|
'loadOptions.n_threads': 'nThreads',
|
|
'loadOptions.n_threads': 'nThreads',
|
|
|
'loadOptions.vae_format': 'vaeFormat',
|
|
'loadOptions.vae_format': 'vaeFormat',
|
|
@@ -219,8 +220,14 @@ const configSchema = {
|
|
|
diffusionFlashAttn: { type: 'boolean', title: 'Flash Attention (diffusion)', description: 'Flash attention for the diffusion model specifically' },
|
|
diffusionFlashAttn: { type: 'boolean', title: 'Flash Attention (diffusion)', description: 'Flash attention for the diffusion model specifically' },
|
|
|
enableMmap: { type: 'boolean', title: 'Memory-map Weights', description: 'Recommended for large files' },
|
|
enableMmap: { type: 'boolean', title: 'Memory-map Weights', description: 'Recommended for large files' },
|
|
|
eagerLoad: { type: 'boolean', title: 'Eager Load', description: 'Move every parameter to the compute backend at load time instead of on demand' },
|
|
eagerLoad: { type: 'boolean', title: 'Eager Load', description: 'Move every parameter to the compute backend at load time instead of on demand' },
|
|
|
- streamLayers: { type: 'boolean', title: 'Stream Layers', description: 'Stream diffusion layers when the model does not fit in VRAM. Pair with a VRAM budget' },
|
|
|
|
|
- maxVram: { type: 'number', title: 'VRAM Budget (GiB)', description: 'Budget for segmented parameter offload. 0 leaves it to the server' },
|
|
|
|
|
|
|
+ disablePrefetch: { type: 'boolean', title: 'Disable Prefetch', description: 'Turn off prefetching of the next layer\'s weights. On by default upstream - only switch this off to diagnose a problem, it costs speed' },
|
|
|
|
|
+ disableSegmentedCompute: { type: 'boolean', title: 'Disable Segmented Compute', description: 'Turn off running the diffusion graph in segments. On by default upstream, and what lets a model larger than VRAM run at all - switching it off will OOM on a big model' },
|
|
|
|
|
+ maxVram: {
|
|
|
|
|
+ type: 'number',
|
|
|
|
|
+ title: 'VRAM Budget (GiB)',
|
|
|
|
|
+ default: 0,
|
|
|
|
|
+ description: '0 (the default) lets sd.cpp re-check free VRAM continuously and use what is actually there - the safest setting, and the right one unless you have a specific reason. A positive N caps managed weights and runner buffers at N GiB regardless of what is free, which is what you want when sharing the card with something else and you need a hard ceiling. The old -1 is no longer accepted; the server coerces any negative value to 0, which matches what -1 was asking for'
|
|
|
|
|
+ },
|
|
|
nThreads: { type: 'number', title: 'CPU Threads', description: '-1 lets the server decide' },
|
|
nThreads: { type: 'number', title: 'CPU Threads', description: '-1 lets the server decide' },
|
|
|
weightType: {
|
|
weightType: {
|
|
|
type: 'string', title: 'Weight Type',
|
|
type: 'string', title: 'Weight Type',
|
|
@@ -267,7 +274,7 @@ const configSchema = {
|
|
|
tensorTypeRules: { type: 'string', title: 'Tensor Type Rules', description: 'Per-tensor weight overrides using regex, such as ^vae\\.=f16' },
|
|
tensorTypeRules: { type: 'string', title: 'Tensor Type Rules', description: 'Per-tensor weight overrides using regex, such as ^vae\\.=f16' },
|
|
|
options: {
|
|
options: {
|
|
|
type: 'object', title: 'Other Load Options',
|
|
type: 'object', title: 'Other Load Options',
|
|
|
- description: 'Extra load options passed through, such as flash_attn, enable_mmap, weight_type, stream_layers or max_vram'
|
|
|
|
|
|
|
+ description: 'Extra load options passed through, such as flash_attn, enable_mmap, weight_type, disable_prefetch or max_vram'
|
|
|
},
|
|
},
|
|
|
whenDifferent: {
|
|
whenDifferent: {
|
|
|
type: 'string', title: 'When A Different Model Is Loaded',
|
|
type: 'string', title: 'When A Different Model Is Loaded',
|
|
@@ -428,7 +435,12 @@ const LOAD_OPTIONS = [
|
|
|
{ setting: 'diffusionFlashAttn', server: 'diffusion_flash_attn' },
|
|
{ setting: 'diffusionFlashAttn', server: 'diffusion_flash_attn' },
|
|
|
{ setting: 'enableMmap', server: 'enable_mmap' },
|
|
{ setting: 'enableMmap', server: 'enable_mmap' },
|
|
|
{ setting: 'eagerLoad', server: 'eager_load' },
|
|
{ setting: 'eagerLoad', server: 'eager_load' },
|
|
|
- { setting: 'streamLayers', server: 'stream_layers' },
|
|
|
|
|
|
|
+ // Layer streaming and segmented compute are how sd.cpp works now - always
|
|
|
|
|
+ // on, with nothing to enable. The old stream_layers toggle is gone
|
|
|
|
|
+ // entirely, and a load still carrying it is rejected outright: "Unknown
|
|
|
|
|
+ // field(s) in /models/load options". These two only turn the behaviour OFF.
|
|
|
|
|
+ { setting: 'disablePrefetch', server: 'disable_prefetch' },
|
|
|
|
|
+ { setting: 'disableSegmentedCompute', server: 'disable_segmented_compute' },
|
|
|
{ setting: 'maxVram', server: 'max_vram' },
|
|
{ setting: 'maxVram', server: 'max_vram' },
|
|
|
{ setting: 'nThreads', server: 'n_threads' },
|
|
{ setting: 'nThreads', server: 'n_threads' },
|
|
|
{ setting: 'weightType', server: 'weight_type' },
|
|
{ setting: 'weightType', server: 'weight_type' },
|
|
@@ -440,6 +452,10 @@ const LOAD_OPTIONS = [
|
|
|
{ setting: 'vaeConvDirect', server: 'vae_conv_direct' },
|
|
{ setting: 'vaeConvDirect', server: 'vae_conv_direct' },
|
|
|
{ setting: 'diffusionConvDirect', server: 'diffusion_conv_direct' },
|
|
{ setting: 'diffusionConvDirect', server: 'diffusion_conv_direct' },
|
|
|
{ setting: 'taePreviewOnly', server: 'tae_preview_only' },
|
|
{ setting: 'taePreviewOnly', server: 'tae_preview_only' },
|
|
|
|
|
+ // Absent from /openapi.json but genuinely accepted - it is in the server's
|
|
|
|
|
+ // own allow-list in model_manager.cpp and read into ctx_params. The schema
|
|
|
|
|
+ // is incomplete here, so a field being missing from it is not evidence
|
|
|
|
|
+ // that the server rejects it.
|
|
|
{ setting: 'forceSdxlVaeConvScale', server: 'force_sdxl_vae_conv_scale' },
|
|
{ setting: 'forceSdxlVaeConvScale', server: 'force_sdxl_vae_conv_scale' },
|
|
|
{ setting: 'backend', server: 'backend' },
|
|
{ setting: 'backend', server: 'backend' },
|
|
|
{ setting: 'paramsBackend', server: 'params_backend' },
|
|
{ setting: 'paramsBackend', server: 'params_backend' },
|
|
@@ -459,6 +475,20 @@ function wantedOptions(config) {
|
|
|
var value = config[entry.setting];
|
|
var value = config[entry.setting];
|
|
|
if (value === undefined || value === null || value === '') continue;
|
|
if (value === undefined || value === null || value === '') continue;
|
|
|
if (typeof value === 'number' && !isFinite(value)) continue;
|
|
if (typeof value === 'number' && !isFinite(value)) continue;
|
|
|
|
|
+
|
|
|
|
|
+ // max_vram used to take -1 for "auto". The server's own parser
|
|
|
|
|
+ // (ModelLoadParams::from_json) already coerces any negative to 0, which
|
|
|
|
|
+ // is the closest match to that intent - 0 means sd.cpp re-checks free
|
|
|
|
|
+ // VRAM continuously. Normalised here too so the value the node reports
|
|
|
|
|
+ // sending is the value that takes effect, rather than the caller seeing
|
|
|
|
|
+ // -1 in the request and 0 in the behaviour.
|
|
|
|
|
+ if (entry.server === 'max_vram' && value < 0) {
|
|
|
|
|
+ smartbotic.log.info('SD.cpp: max_vram ' + value +
|
|
|
|
|
+ ' is the old "auto" value; sending 0, which is what the server ' +
|
|
|
|
|
+ 'would coerce it to and means "use whatever VRAM is free".');
|
|
|
|
|
+ value = 0;
|
|
|
|
|
+ }
|
|
|
|
|
+
|
|
|
wanted[entry.server] = value;
|
|
wanted[entry.server] = value;
|
|
|
}
|
|
}
|
|
|
if (config.options && typeof config.options === 'object') {
|
|
if (config.options && typeof config.options === 'object') {
|